ci: parallelize models chunk downloads and split branch publishing (#1955)

* ci: parallelize model chunk downloads and better publish

* ci: download all model chunks in parallel with xargs -P8

* split split

* ew

* must require
This commit is contained in:
Jason Wen
2026-08-24 21:48:53 -04:00
committed by GitHub
parent 6cc5f3aad8
commit d14d0b1dd0
2 changed files with 136 additions and 119 deletions
@@ -0,0 +1,66 @@
name: Download HF model chunks
description: Resolve and download model chunks from HuggingFace in parallel
inputs:
hf_repo:
description: HuggingFace dataset repo
required: true
models:
description: 'JSON array of {hf_path, onnx_hash, canonical} objects'
required: true
dest_dir:
description: Destination directory for downloaded chunks
required: true
runs:
using: composite
steps:
- name: Download model chunks
shell: bash
env:
HF_REPO: ${{ inputs.hf_repo }}
MODELS_JSON: ${{ inputs.models }}
DEST_DIR: ${{ inputs.dest_dir }}
run: |
set -eo pipefail
DOWNLOAD_LIST=$(mktemp)
resolve_chunks() {
local HF_PATH="$1" ONNX_HASH="$2" CANONICAL="$3" DEST_DIR="$4"
local JSON_URL="https://huggingface.co/datasets/${HF_REPO}/resolve/main/${HF_PATH}/default_models.json"
local DEFAULTS BUNDLE ARTIFACT BASE_URL NUM_CHUNKS
DEFAULTS=$(curl -fsSL "$JSON_URL")
BUNDLE=$(echo "$DEFAULTS" | jq --arg hash "$ONNX_HASH" '.bundles[] | select(.onnx_sha256 == $hash)')
ARTIFACT=$(echo "$BUNDLE" | jq -r '.models[0].artifact')
BASE_URL=$(echo "$ARTIFACT" | jq -r '.download_uri.url' | sed 's|/[^/]*$||')
NUM_CHUNKS=$(echo "$ARTIFACT" | jq -r '.chunks | length')
mkdir -p "$DEST_DIR"
while IFS= read -r CHUNK_NAME; do
CHUNK_IDX=$(echo "$CHUNK_NAME" | grep -oP 'chunk\K[0-9]+of[0-9]+' || true)
if [ -z "$CHUNK_IDX" ]; then
echo "::error::Failed to parse chunk index from: $CHUNK_NAME"
return 1
fi
ENCODED_URL=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${BASE_URL}/${CHUNK_NAME}', safe=':/'))")
printf '%s\t%s\n' "$ENCODED_URL" "${DEST_DIR}/${CANONICAL}.chunk${CHUNK_IDX}" >> "$DOWNLOAD_LIST"
done < <(echo "$ARTIFACT" | jq -r '.chunks[].file_name')
echo "$NUM_CHUNKS" > "${DEST_DIR}/${CANONICAL}.chunkmanifest"
}
echo "$MODELS_JSON" | jq -c '.[]' | while IFS= read -r model; do
HF_PATH=$(echo "$model" | jq -r '.hf_path')
ONNX_HASH=$(echo "$model" | jq -r '.onnx_hash')
CANONICAL=$(echo "$model" | jq -r '.canonical')
resolve_chunks "$HF_PATH" "$ONNX_HASH" "$CANONICAL" "$DEST_DIR"
done
TOTAL=$(wc -l < "$DOWNLOAD_LIST")
echo "Downloading $TOTAL chunks with 8 parallel connections..."
xargs -P8 -d'\n' -I{} bash -c '
URL="${1%% *}"
DEST="${1#* }"
echo "Downloading $(basename "$DEST")"
curl -fsSL --retry 3 --retry-delay 5 -o "$DEST" "$URL"
' _ {} < "$DOWNLOAD_LIST"
rm -f "$DOWNLOAD_LIST"
+70 -119
View File
@@ -424,109 +424,16 @@ jobs:
mkdir -p ${{ env.OUTPUT_DIR }}
tar xzf prebuilt.tar.gz -C ${{ env.OUTPUT_DIR }}
- name: Download default model chunks from HF
env:
HF_REPO: sunnypilot/sunnypilot_models_v1
run: |
set -o pipefail
MODELS_DIR="${{ env.OUTPUT_DIR }}/openpilot/selfdrive/modeld/models"
download_model_chunks() {
local DEFAULTS_PATH="$1"
local ONNX_HASH="$2"
local CANONICAL="$3"
local JSON_URL="https://huggingface.co/datasets/${HF_REPO}/resolve/main/${DEFAULTS_PATH}/default_models.json"
local DEFAULTS=$(curl -fsSL "$JSON_URL")
BUNDLE=$(echo "$DEFAULTS" | jq --arg hash "$ONNX_HASH" '.bundles[] | select(.onnx_sha256 == $hash)')
ARTIFACT=$(echo "$BUNDLE" | jq -r '.models[0].artifact')
BASE_URL=$(echo "$ARTIFACT" | jq -r '.download_uri.url' | sed 's|/[^/]*$||')
NUM_CHUNKS=$(echo "$ARTIFACT" | jq -r '.chunks | length')
echo "$ARTIFACT" | jq -r '.chunks[].file_name' | while read CHUNK_NAME; do
CHUNK_IDX=$(echo "$CHUNK_NAME" | grep -oP 'chunk\K[0-9]+of[0-9]+' || true)
if [ -z "$CHUNK_IDX" ]; then
echo "::error::Failed to parse chunk index from: $CHUNK_NAME"
exit 1
fi
CANONICAL_CHUNK="${CANONICAL}.chunk${CHUNK_IDX}"
ENCODED_URL=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${BASE_URL}/${CHUNK_NAME}', safe=':/'))")
echo "Downloading $CHUNK_NAME -> $CANONICAL_CHUNK"
curl -fsSL -o "${MODELS_DIR}/${CANONICAL_CHUNK}" "$ENCODED_URL"
done
echo "$NUM_CHUNKS" > "${MODELS_DIR}/${CANONICAL}.chunkmanifest"
}
download_model_chunks "models/defaults/small" "${{ needs.prepare_small_model.outputs.driving_onnx_sha256 }}" "driving_tinygrad.pkl"
download_model_chunks "models/defaults/dm" "${{ needs.prepare_dm_model.outputs.dm_onnx_sha256 }}" "dmonitoring_model_tinygrad.pkl"
- name: Prepare chestnut output
if: ${{ needs.prepare_chestnut.result == 'success' }}
run: |
mkdir -p "${{ github.workspace }}/chestnut_output"
tar xzf prebuilt.tar.gz -C "${{ github.workspace }}/chestnut_output"
- name: Download big model chunks from HF
if: ${{ needs.prepare_chestnut.result == 'success' }}
env:
HF_REPO: sunnypilot/sunnypilot_models_v1
HF_DEFAULTS_PATH: models/defaults/big
run: |
ONNX_HASH="${{ needs.prepare_chestnut.outputs.onnx_sha256 }}"
JSON_URL="https://huggingface.co/datasets/${HF_REPO}/resolve/main/${HF_DEFAULTS_PATH}/default_models.json"
DEFAULTS=$(curl -fsSL "$JSON_URL")
BUNDLE=$(echo "$DEFAULTS" | jq --arg hash "$ONNX_HASH" '.bundles[] | select(.onnx_sha256 == $hash)')
mkdir -p big_model_chunks
ARTIFACT=$(echo "$BUNDLE" | jq -r '.models[0].artifact')
BASE_URL=$(echo "$ARTIFACT" | jq -r '.download_uri.url' | sed 's|/[^/]*$||')
NUM_CHUNKS=$(echo "$ARTIFACT" | jq -r '.chunks | length')
CANONICAL="big_driving_tinygrad.pkl"
echo "$ARTIFACT" | jq -r '.chunks[].file_name' | while read CHUNK_NAME; do
CHUNK_IDX=$(echo "$CHUNK_NAME" | grep -oP 'chunk\K[0-9]+of[0-9]+')
CANONICAL_CHUNK="${CANONICAL}.chunk${CHUNK_IDX}"
ENCODED_URL=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${BASE_URL}/${CHUNK_NAME}', safe=':/'))")
echo "Downloading $CHUNK_NAME -> $CANONICAL_CHUNK"
curl -fsSL -o "big_model_chunks/${CANONICAL_CHUNK}" "$ENCODED_URL"
done
echo "$NUM_CHUNKS" > "big_model_chunks/${CANONICAL}.chunkmanifest"
- name: Inject models into chestnut
if: ${{ needs.prepare_chestnut.result == 'success' }}
env:
HF_REPO: sunnypilot/sunnypilot_models_v1
run: |
CHESTNUT_MODELS="${{ github.workspace }}/chestnut_output/openpilot/selfdrive/modeld/models"
cp big_model_chunks/* "$CHESTNUT_MODELS/"
download_model_chunks() {
local DEFAULTS_PATH="$1"
local ONNX_HASH="$2"
local CANONICAL="$3"
local JSON_URL="https://huggingface.co/datasets/${HF_REPO}/resolve/main/${DEFAULTS_PATH}/default_models.json"
local DEFAULTS=$(curl -fsSL "$JSON_URL")
BUNDLE=$(echo "$DEFAULTS" | jq --arg hash "$ONNX_HASH" '.bundles[] | select(.onnx_sha256 == $hash)')
ARTIFACT=$(echo "$BUNDLE" | jq -r '.models[0].artifact')
BASE_URL=$(echo "$ARTIFACT" | jq -r '.download_uri.url' | sed 's|/[^/]*$||')
NUM_CHUNKS=$(echo "$ARTIFACT" | jq -r '.chunks | length')
echo "$ARTIFACT" | jq -r '.chunks[].file_name' | while read CHUNK_NAME; do
CHUNK_IDX=$(echo "$CHUNK_NAME" | grep -oP 'chunk\K[0-9]+of[0-9]+' || true)
if [ -z "$CHUNK_IDX" ]; then
echo "::error::Failed to parse chunk index from: $CHUNK_NAME"
exit 1
fi
CANONICAL_CHUNK="${CANONICAL}.chunk${CHUNK_IDX}"
ENCODED_URL=$(python3 -c "import urllib.parse; print(urllib.parse.quote('${BASE_URL}/${CHUNK_NAME}', safe=':/'))")
echo "Downloading $CHUNK_NAME -> $CANONICAL_CHUNK"
curl -fsSL -o "${CHESTNUT_MODELS}/${CANONICAL_CHUNK}" "$ENCODED_URL"
done
echo "$NUM_CHUNKS" > "${CHESTNUT_MODELS}/${CANONICAL}.chunkmanifest"
}
download_model_chunks "models/defaults/small" "${{ needs.prepare_small_model.outputs.driving_onnx_sha256 }}" "driving_tinygrad.pkl"
download_model_chunks "models/defaults/dm" "${{ needs.prepare_dm_model.outputs.dm_onnx_sha256 }}" "dmonitoring_model_tinygrad.pkl"
- name: Download model chunks from HF
uses: ./.github/workflows/download-hf-model-chunks
with:
hf_repo: sunnypilot/sunnypilot_models_v1
dest_dir: ${{ env.OUTPUT_DIR }}/openpilot/selfdrive/modeld/models
models: |
[
{"hf_path": "models/defaults/small", "onnx_hash": "${{ needs.prepare_small_model.outputs.driving_onnx_sha256 }}", "canonical": "driving_tinygrad.pkl"},
{"hf_path": "models/defaults/dm", "onnx_hash": "${{ needs.prepare_dm_model.outputs.dm_onnx_sha256 }}", "canonical": "dmonitoring_model_tinygrad.pkl"}
]
- name: Configure Git
run: |
@@ -548,22 +455,6 @@ jobs:
"https://x-access-token:${{github.token}}@github.com/sunnypilot/sunnypilot.git" \
"${{ needs.prepare_strategy.outputs.extra_version_identifier }}"
- name: Publish chestnut branch
if: ${{ needs.prepare_chestnut.result == 'success' }}
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
CHESTNUT_BRANCH="${{ needs.prepare_strategy.outputs.new_branch }}-chestnut"
CHESTNUT_DIR="${{ github.workspace }}/chestnut_output"
${{ env.CI_DIR }}/publish.sh \
"${{ github.workspace }}" \
"$CHESTNUT_DIR" \
"$CHESTNUT_BRANCH" \
"${{ needs.prepare_strategy.outputs.version }}" \
"https://x-access-token:${{github.token}}@github.com/sunnypilot/sunnypilot.git" \
"${{ needs.prepare_strategy.outputs.extra_version_identifier }}"
- name: Tag ${{ needs.prepare_strategy.outputs.environment }}
if: ${{ needs.prepare_strategy.outputs.is_stable_branch == 'true' && (github.event_name != 'push' || !startsWith(github.ref, 'refs/tags/')) }}
run: |
@@ -571,11 +462,71 @@ jobs:
git tag -f -a ${TAG} -m "${{ needs.prepare_strategy.outputs.environment }} @ ${{ needs.prepare_strategy.outputs.version }} of build ${{ needs.prepare_strategy.outputs.build }}."
git push -f origin ${TAG}
publish_chestnut:
concurrency:
group: ${{ needs.prepare_strategy.outputs.publish_concurrency_group }}-chestnut
cancel-in-progress: ${{ needs.prepare_strategy.outputs.cancel_publish_in_progress == 'true' }}
if: ${{
always() && !cancelled() &&
needs.build.result == 'success' &&
needs.prepare_strategy.result == 'success' &&
needs.prepare_small_model.result == 'success' &&
needs.prepare_dm_model.result == 'success' &&
needs.prepare_chestnut.result == 'success' &&
(!contains(github.event_name, 'pull_request') || (github.event.action == 'labeled' && github.event.label.name == 'prebuilt'))
}}
needs: [ build, prepare_strategy, prepare_chestnut, prepare_small_model, prepare_dm_model ]
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v4
- name: Download prebuilt artifact
uses: actions/download-artifact@v4
with:
name: prebuilt
- name: Untar prebuilt
run: |
mkdir -p ${{ env.OUTPUT_DIR }}
tar xzf prebuilt.tar.gz -C ${{ env.OUTPUT_DIR }}
- name: Download model chunks from HF
uses: ./.github/workflows/download-hf-model-chunks
with:
hf_repo: sunnypilot/sunnypilot_models_v1
dest_dir: ${{ env.OUTPUT_DIR }}/openpilot/selfdrive/modeld/models
models: |
[
{"hf_path": "models/defaults/small", "onnx_hash": "${{ needs.prepare_small_model.outputs.driving_onnx_sha256 }}", "canonical": "driving_tinygrad.pkl"},
{"hf_path": "models/defaults/dm", "onnx_hash": "${{ needs.prepare_dm_model.outputs.dm_onnx_sha256 }}", "canonical": "dmonitoring_model_tinygrad.pkl"},
{"hf_path": "models/defaults/big", "onnx_hash": "${{ needs.prepare_chestnut.outputs.onnx_sha256 }}", "canonical": "big_driving_tinygrad.pkl"}
]
- name: Configure Git
run: |
git config --global user.email "github-actions[bot]@users.noreply.github.com"
git config --global user.name "github-actions[bot]"
- name: Publish chestnut branch
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
CHESTNUT_BRANCH="${{ needs.prepare_strategy.outputs.new_branch }}-chestnut"
${{ env.CI_DIR }}/publish.sh \
"${{ github.workspace }}" \
"${{ env.OUTPUT_DIR }}" \
"$CHESTNUT_BRANCH" \
"${{ needs.prepare_strategy.outputs.version }}" \
"https://x-access-token:${{github.token}}@github.com/sunnypilot/sunnypilot.git" \
"${{ needs.prepare_strategy.outputs.extra_version_identifier }}"
notify:
needs:
- prepare_strategy
- build
- publish
- publish_chestnut
- prepare_chestnut
- prepare_small_model
- prepare_dm_model