From 86a7026ba612b2ce222a726e6df27d6e3dc40729 Mon Sep 17 00:00:00 2001 From: contentis Date: Wed, 23 Sep 2026 10:46:45 +0200 Subject: [PATCH 1/8] Skip draft CI and scope model tests and export reuse --- .github/scripts/find_successful_artifact.py | 23 ++-- .github/scripts/model_ci.py | 112 ++++++++++++++++++ .github/workflows/ci.yml | 5 + .github/workflows/export-nvidia-asr-model.yml | 49 +++----- .github/workflows/export-sam2-model.yml | 69 ++++------- .github/workflows/export-whisper-model.yml | 50 +++----- .github/workflows/model-changes.yml | 32 +++++ .github/workflows/nvidia-asr.yml | 19 ++- .github/workflows/sam2.yml | 31 +++-- .github/workflows/whisper-asr.yml | 7 ++ 10 files changed, 257 insertions(+), 140 deletions(-) create mode 100644 .github/scripts/model_ci.py create mode 100644 .github/workflows/model-changes.yml diff --git a/.github/scripts/find_successful_artifact.py b/.github/scripts/find_successful_artifact.py index 82338ee..4174f0e 100644 --- a/.github/scripts/find_successful_artifact.py +++ b/.github/scripts/find_successful_artifact.py @@ -25,7 +25,10 @@ def github_api(path: str) -> object: return json.load(response) -def write_outputs(*, found: bool, artifact_name: str, run_id: str = "") -> None: +def write_outputs(*, found: bool, artifact_name: str, run_id: str = "", as_json: bool = False) -> None: + if as_json: + print(json.dumps({"found": found, "name": artifact_name if found else "", "run_id": run_id})) + return output_path = Path(os.environ["GITHUB_OUTPUT"]) with output_path.open("a", encoding="utf-8") as output: output.write(f"found={'true' if found else 'false'}\n") @@ -40,16 +43,15 @@ def main() -> None: parser.add_argument("--default-branch", required=True) parser.add_argument("--event-name", required=True) parser.add_argument("--source-branch", required=True) + parser.add_argument("--pull-request", default="") + parser.add_argument("--json", action="store_true") args = parser.parse_args() artifact_name = args.artifact_name allowed_branches = {args.default_branch} - if args.event_name != "pull_request": + if args.event_name != "pull_request" or args.pull_request: allowed_branches.add(args.source_branch) - response = github_api( - f"repos/{args.repository}/actions/artifacts" - f"?name={quote(artifact_name)}&per_page=100" - ) + response = github_api(f"repos/{args.repository}/actions/artifacts?name={quote(artifact_name)}&per_page=100") artifacts = sorted( ( artifact @@ -67,18 +69,23 @@ def main() -> None: run_id = int(artifact["workflow_run"]["id"]) if run_id not in conclusions: run = github_api(f"repos/{args.repository}/actions/runs/{run_id}") + same_pr = args.pull_request and any( + str(pr["number"]) == args.pull_request for pr in run.get("pull_requests", []) + ) + trusted_branch = run["event"] != "pull_request" and run["head_branch"] in allowed_branches conclusions[run_id] = ( - run["status"] == "completed" and run["conclusion"] == "success" + (trusted_branch or same_pr) and run["status"] == "completed" and run["conclusion"] == "success" ) if conclusions[run_id]: write_outputs( found=True, artifact_name=artifact_name, run_id=str(run_id), + as_json=args.json, ) return - write_outputs(found=False, artifact_name=artifact_name) + write_outputs(found=False, artifact_name=artifact_name, as_json=args.json) if __name__ == "__main__": diff --git a/.github/scripts/model_ci.py b/.github/scripts/model_ci.py new file mode 100644 index 0000000..207a4eb --- /dev/null +++ b/.github/scripts/model_ci.py @@ -0,0 +1,112 @@ +#!/usr/bin/env python3 +"""Select affected model tests and fingerprint their ONNX export inputs.""" + +import argparse +import fnmatch +import hashlib +import json +import os +import subprocess +from pathlib import Path + +SHARED_RUNTIME = [ + "CMakeLists.txt", + "CMakePresets.json", + "cmake/*", + "common/*.cpp", + "common/*.h", + "common/*CMakeLists.txt", + ".github/workflows/ci.yml", + ".github/scripts/*", + ".github/workflows/model-changes.yml", +] +EXPORTS = { + "whisper": [ + "asr/whisper/model_export/export_whisper.py", + "common/model_export/*.py", + ".github/workflows/export-whisper-model.yml", + ], + "parakeet": [ + "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", + "asr/rnnt/python/rnnt/parakeet_tdt/nemo_backend.py", + "asr/rnnt/python/rnnt/parakeet_tdt/__init__.py", + ], + "nemotron": [ + "asr/rnnt/python/tools/export_nemotron_onnx.py", + "asr/rnnt/python/rnnt/nemotron_asr/nemo_backend.py", + "asr/rnnt/python/rnnt/nemotron_asr/__init__.py", + ], + "sam2": [ + "vision/sam2/python/export_sam2_onnx.py", + "vision/sam2/python/modeling.py", + ".github/workflows/export-sam2-model.yml", + ], +} +for model in ("parakeet", "nemotron"): + EXPORTS[model] += [ + "asr/rnnt/python/tools/nemo_preprocessor_export.py", + "asr/rnnt/python/rnnt/__init__.py", + "asr/rnnt/python/rnnt/audio.py", + ".github/workflows/export-nvidia-asr-model.yml", + ] +RUNTIME = { + "whisper": ["asr/whisper/*", "assets/sample.wav", ".github/workflows/whisper-asr.yml"], + "sam2": ["vision/sam2/*", "assets/sam2-*", ".github/workflows/sam2.yml"], +} +for model, package in (("parakeet", "parakeet_tdt"), ("nemotron", "nemotron_asr")): + RUNTIME[model] = [ + f"asr/rnnt/cpp/{model}*", + "asr/rnnt/cpp/asr*", + "asr/rnnt/CMakeLists.txt", + f"asr/rnnt/python/rnnt/{package}/*", + "asr/rnnt/python/rnnt/validation/*", + "asr/rnnt/python/tests/*", + "assets/sample.wav", + ".github/workflows/nvidia-asr.yml", + ] + + +def affected(model, files): + patterns = SHARED_RUNTIME + RUNTIME[model] + EXPORTS[model] + return any(fnmatch.fnmatchcase(path, pattern) for path in files if not path.endswith(".md") for pattern in patterns) + + +def export_key(model, model_id, attention): + # Index blob IDs are independent of checkout line endings and include deleted/renamed inputs. + sources = subprocess.check_output(["git", "ls-files", "--stage", "--", *EXPORTS[model]]) + options = json.dumps([model, model_id, attention]).encode() + return hashlib.sha256(sources + options).hexdigest()[:16] + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("mode", choices=["changes", "key"]) + parser.add_argument("--model", choices=EXPORTS) + parser.add_argument("--model-id", default="") + parser.add_argument("--attention", default="") + args = parser.parse_args() + if args.mode == "key": + if not args.model or not args.model_id: + parser.error("key requires --model and --model-id") + print(export_key(args.model, args.model_id, args.attention)) + return + event = json.loads(Path(os.environ["GITHUB_EVENT_PATH"]).read_text()) + pr = event.get("pull_request") + base = pr["base"]["sha"] if pr else event.get("before") + files = None + if base and subprocess.run(["git", "cat-file", "-e", f"{base}^{{commit}}"], capture_output=True).returncode == 0: + if pr: + base = subprocess.check_output(["git", "merge-base", base, "HEAD"], text=True).strip() + files = ( + subprocess.check_output(["git", "diff", "--no-renames", "--name-only", "-z", base, "HEAD"]) + .decode() + .split("\0") + ) + with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: + for model in EXPORTS: + changed = files is None or affected(model, files) + output.write(f"{model}={str(changed).lower()}\n") + + +if __name__ == "__main__": + main() diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5d0965e..aae8c6c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -4,6 +4,7 @@ on: push: branches: [main] pull_request: + types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] workflow_dispatch: permissions: @@ -19,6 +20,7 @@ env: jobs: clang-format: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft name: clang-format runs-on: ubuntu-24.04 @@ -47,6 +49,7 @@ jobs: | xargs -0 -r clang-format-22 --dry-run --Werror -- linux-x64: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft name: Ubuntu 24.04 / Clang / x64 runs-on: ubuntu-24.04 container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -90,6 +93,7 @@ jobs: path: out/build/linux-x64/bin/Release/ linux-arm64: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft name: Ubuntu 24.04 / Clang / ARM64 runs-on: ubuntu-24.04-arm container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -131,6 +135,7 @@ jobs: path: out/build/linux-arm64/bin/Release/ windows-x64: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft name: Windows Server 2025 / MSVC / x64 runs-on: windows-2025 timeout-minutes: 90 diff --git a/.github/workflows/export-nvidia-asr-model.yml b/.github/workflows/export-nvidia-asr-model.yml index dc098d1..7bc2a6c 100644 --- a/.github/workflows/export-nvidia-asr-model.yml +++ b/.github/workflows/export-nvidia-asr-model.yml @@ -51,34 +51,11 @@ jobs: fetch-depth: 0 ref: ${{ inputs.source_ref }} - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'asr/rnnt/python/*.py' \ - 'asr/rnnt/python/**/*.py' \ - '.github/workflows/export-nvidia-asr-model.yml' \ - '.github/workflows/nvidia-asr.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - name: Compute artifact names id: artifact-names shell: bash env: + MODEL_ID: ${{ inputs.model_id }} FALLBACK_PRECISION: ${{ inputs.fallback_precision }} MODEL_SLUG: ${{ inputs.model_slug }} MODEL_TYPE: ${{ inputs.model_type }} @@ -99,13 +76,14 @@ jobs: *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; esac fi - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}" >> "$GITHUB_OUTPUT" + source_hash="$(python .github/scripts/model_ci.py key --model "${MODEL_TYPE}" --model-id "$MODEL_ID")" + echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" + echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}" >> "$GITHUB_OUTPUT" + echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" fi - name: Find successful primary artifact - if: steps.changes.outputs.force != 'true' id: primary-artifact shell: bash env: @@ -113,17 +91,18 @@ jobs: EVENT_NAME: ${{ github.event_name }} GH_TOKEN: ${{ github.token }} SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} run: | python .github/scripts/find_successful_artifact.py \ --repository "$GITHUB_REPOSITORY" \ --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ --default-branch "$DEFAULT_BRANCH" \ --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" + --source-branch "$SOURCE_BRANCH" \ + --pull-request "$PULL_REQUEST" - name: Find successful fallback artifact if: >- - steps.changes.outputs.force != 'true' && steps.primary-artifact.outputs.found != 'true' && inputs.fallback_precision != '' id: fallback-artifact @@ -133,13 +112,15 @@ jobs: EVENT_NAME: ${{ github.event_name }} GH_TOKEN: ${{ github.token }} SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} run: | python .github/scripts/find_successful_artifact.py \ --repository "$GITHUB_REPOSITORY" \ --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ --default-branch "$DEFAULT_BRANCH" \ --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" + --source-branch "$SOURCE_BRANCH" \ + --pull-request "$PULL_REQUEST" - name: Select export action id: selection @@ -148,15 +129,12 @@ jobs: FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - FORCE_EXPORT: ${{ steps.changes.outputs.force }} PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} run: | set -euo pipefail - if [[ "$FORCE_EXPORT" == "true" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - elif [[ "$PRIMARY_FOUND" == "true" ]]; then + if [[ "$PRIMARY_FOUND" == "true" ]]; then echo "export=false" >> "$GITHUB_OUTPUT" echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" @@ -213,6 +191,7 @@ jobs: PRECISION: ${{ inputs.precision }} PYTHONPATH: asr/rnnt/python SOURCE_REF: ${{ inputs.source_ref }} + SOURCE_HASH: ${{ steps.artifact-names.outputs.source_hash }} run: | set -uo pipefail export_dtype() { @@ -288,7 +267,7 @@ jobs: json.dumps(manifest, indent=2) + "\n", encoding="utf-8" ) PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}" >> "$GITHUB_OUTPUT" + echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}-${SOURCE_HASH}" >> "$GITHUB_OUTPUT" - name: Upload NVIDIA ASR ONNX if: steps.selection.outputs.export == 'true' diff --git a/.github/workflows/export-sam2-model.yml b/.github/workflows/export-sam2-model.yml index 227015b..c2df020 100644 --- a/.github/workflows/export-sam2-model.yml +++ b/.github/workflows/export-sam2-model.yml @@ -41,43 +41,15 @@ jobs: fetch-depth: 0 ref: ${{ inputs.source_ref }} - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'vision/sam2/python/*.py' \ - 'vision/sam2/python/**/*.py' \ - '.github/workflows/export-sam2-model.yml' \ - '.github/workflows/sam2.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - name: Compute model artifact key id: model-key shell: bash env: MODEL_SLUG: ${{ inputs.model_slug }} + MODEL_ID: ${{ inputs.model_id }} run: | set -euo pipefail - source_hash="$( - git ls-files -z -- 'vision/sam2/python/*.py' 'vision/sam2/python/**/*.py' \ - | xargs -0 sha256sum \ - | sha256sum \ - | cut -d' ' -f1 - )" + source_hash="$(python .github/scripts/model_ci.py key --model sam2 --model-id "$MODEL_ID")" echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" echo "artifact_name=vision-${MODEL_SLUG}-onnx-v1-${source_hash:0:16}" >> "$GITHUB_OUTPUT" @@ -85,24 +57,23 @@ jobs: id: existing-artifact shell: bash env: - FORCE_EXPORT: ${{ steps.changes.outputs.force }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} GH_TOKEN: ${{ github.token }} + SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} MODEL_ARTIFACT: ${{ steps.model-key.outputs.artifact_name }} run: | - set -euo pipefail - artifact_run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/artifacts?name=${MODEL_ARTIFACT}&per_page=100" \ - --jq '.artifacts | map(select(.expired | not)) | sort_by(.created_at) | last | .workflow_run.id // empty' - )" - if [[ "$FORCE_EXPORT" == "true" || -z "$artifact_run_id" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - else - echo "export=false" >> "$GITHUB_OUTPUT" - fi - echo "run_id=$artifact_run_id" >> "$GITHUB_OUTPUT" + python .github/scripts/find_successful_artifact.py \ + --repository "$GITHUB_REPOSITORY" \ + --artifact-name "$MODEL_ARTIFACT" \ + --default-branch "$DEFAULT_BRANCH" \ + --event-name "$EVENT_NAME" \ + --source-branch "$SOURCE_BRANCH" \ + --pull-request "$PULL_REQUEST" - name: Reclaim runner disk space - if: steps.existing-artifact.outputs.export == 'true' + if: steps.existing-artifact.outputs.found != 'true' shell: bash run: | set -euxo pipefail @@ -115,13 +86,13 @@ jobs: df -h / - name: Set up Python - if: steps.existing-artifact.outputs.export == 'true' + if: steps.existing-artifact.outputs.found != 'true' uses: actions/setup-python@v7 with: python-version: '3.12' - name: Install SAM2 export dependencies - if: steps.existing-artifact.outputs.export == 'true' + if: steps.existing-artifact.outputs.found != 'true' shell: bash run: | set -euxo pipefail @@ -138,7 +109,7 @@ jobs: "sam-2 @ git+https://github.com/facebookresearch/sam2.git" - name: Export SAM2 ONNX - if: steps.existing-artifact.outputs.export == 'true' + if: steps.existing-artifact.outputs.found != 'true' shell: bash env: MODEL_ID: ${{ inputs.model_id }} @@ -187,7 +158,7 @@ jobs: PY - name: Upload SAM2 ONNX - if: steps.existing-artifact.outputs.export == 'true' + if: steps.existing-artifact.outputs.found != 'true' uses: actions/upload-artifact@v7 with: name: ${{ steps.model-key.outputs.artifact_name }} @@ -199,11 +170,11 @@ jobs: id: final-artifact shell: bash env: - EXPORTED: ${{ steps.existing-artifact.outputs.export }} + FOUND: ${{ steps.existing-artifact.outputs.found }} EXISTING_RUN_ID: ${{ steps.existing-artifact.outputs.run_id }} run: | set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then + if [[ "$FOUND" != "true" ]]; then echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" else echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" diff --git a/.github/workflows/export-whisper-model.yml b/.github/workflows/export-whisper-model.yml index 0f895e8..8f5ddf4 100644 --- a/.github/workflows/export-whisper-model.yml +++ b/.github/workflows/export-whisper-model.yml @@ -52,34 +52,12 @@ jobs: fetch-depth: 0 ref: ${{ inputs.source_ref }} - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'asr/whisper/*.py' \ - 'asr/whisper/**/*.py' \ - '.github/workflows/export-whisper-model.yml' \ - '.github/workflows/whisper-asr.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - name: Compute artifact names id: artifact-names shell: bash env: + MODEL_ID: ${{ inputs.model_id }} + ATTENTION: ${{ inputs.attention }} FALLBACK_PRECISION: ${{ inputs.fallback_precision }} MODEL_SLUG: ${{ inputs.model_slug }} PRECISION: ${{ inputs.precision }} @@ -95,13 +73,14 @@ jobs: *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; esac fi - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}" >> "$GITHUB_OUTPUT" + source_hash="$(python .github/scripts/model_ci.py key --model "whisper" --model-id "$MODEL_ID" --attention "$ATTENTION")" + echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" + echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}" >> "$GITHUB_OUTPUT" + echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" fi - name: Find successful primary artifact - if: steps.changes.outputs.force != 'true' id: primary-artifact shell: bash env: @@ -109,17 +88,18 @@ jobs: EVENT_NAME: ${{ github.event_name }} GH_TOKEN: ${{ github.token }} SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} run: | python .github/scripts/find_successful_artifact.py \ --repository "$GITHUB_REPOSITORY" \ --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ --default-branch "$DEFAULT_BRANCH" \ --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" + --source-branch "$SOURCE_BRANCH" \ + --pull-request "$PULL_REQUEST" - name: Find successful fallback artifact if: >- - steps.changes.outputs.force != 'true' && steps.primary-artifact.outputs.found != 'true' && inputs.fallback_precision != '' id: fallback-artifact @@ -129,13 +109,15 @@ jobs: EVENT_NAME: ${{ github.event_name }} GH_TOKEN: ${{ github.token }} SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} run: | python .github/scripts/find_successful_artifact.py \ --repository "$GITHUB_REPOSITORY" \ --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ --default-branch "$DEFAULT_BRANCH" \ --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" + --source-branch "$SOURCE_BRANCH" \ + --pull-request "$PULL_REQUEST" - name: Select export action id: selection @@ -144,15 +126,12 @@ jobs: FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - FORCE_EXPORT: ${{ steps.changes.outputs.force }} PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} run: | set -euo pipefail - if [[ "$FORCE_EXPORT" == "true" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - elif [[ "$PRIMARY_FOUND" == "true" ]]; then + if [[ "$PRIMARY_FOUND" == "true" ]]; then echo "export=false" >> "$GITHUB_OUTPUT" echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" @@ -204,6 +183,7 @@ jobs: MODEL_SLUG: ${{ inputs.model_slug }} PRECISION: ${{ inputs.precision }} SOURCE_REF: ${{ inputs.source_ref }} + SOURCE_HASH: ${{ steps.artifact-names.outputs.source_hash }} run: | set -uo pipefail run_export() { @@ -250,7 +230,7 @@ jobs: json.dumps(manifest, indent=2) + "\n", encoding="utf-8" ) PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}" >> "$GITHUB_OUTPUT" + echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}-${SOURCE_HASH}" >> "$GITHUB_OUTPUT" - name: Upload Whisper ONNX if: steps.selection.outputs.export == 'true' diff --git a/.github/workflows/model-changes.yml b/.github/workflows/model-changes.yml new file mode 100644 index 0000000..d945b3e --- /dev/null +++ b/.github/workflows/model-changes.yml @@ -0,0 +1,32 @@ +name: Detect model changes + +on: + workflow_call: + outputs: + whisper: + value: ${{ jobs.changes.outputs.whisper }} + parakeet: + value: ${{ jobs.changes.outputs.parakeet }} + nemotron: + value: ${{ jobs.changes.outputs.nemotron }} + sam2: + value: ${{ jobs.changes.outputs.sam2 }} + +permissions: + contents: read + +jobs: + changes: + runs-on: ubuntu-24.04 + outputs: + whisper: ${{ steps.changes.outputs.whisper }} + parakeet: ${{ steps.changes.outputs.parakeet }} + nemotron: ${{ steps.changes.outputs.nemotron }} + sam2: ${{ steps.changes.outputs.sam2 }} + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + ref: ${{ github.event.pull_request.head.sha || github.sha }} + - id: changes + run: python .github/scripts/model_ci.py changes diff --git a/.github/workflows/nvidia-asr.yml b/.github/workflows/nvidia-asr.yml index 780708c..0eb6940 100644 --- a/.github/workflows/nvidia-asr.yml +++ b/.github/workflows/nvidia-asr.yml @@ -4,6 +4,7 @@ on: push: branches: [main] pull_request: + types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] workflow_dispatch: inputs: build_run_id: @@ -20,7 +21,13 @@ concurrency: cancel-in-progress: true jobs: + changes: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft + uses: ./.github/workflows/model-changes.yml + requirements: + needs: changes + if: needs.changes.outputs.parakeet == 'true' || needs.changes.outputs.nemotron == 'true' name: Wait for Build runs-on: ubuntu-24.04 timeout-minutes: 100 @@ -62,8 +69,9 @@ jobs: echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" export-parakeet: + if: needs.changes.outputs.parakeet == 'true' name: Export Parakeet TDT ONNX - needs: requirements + needs: [requirements, changes] uses: ./.github/workflows/export-nvidia-asr-model.yml with: model_slug: parakeet-tdt @@ -75,8 +83,9 @@ jobs: source_ref: ${{ needs.requirements.outputs.source_ref }} export-nemotron: + if: needs.changes.outputs.nemotron == 'true' name: Export Nemotron 3.5 ASR Streaming ONNX - needs: requirements + needs: [requirements, changes] uses: ./.github/workflows/export-nvidia-asr-model.yml with: model_slug: nemotron-3.5-asr-streaming-0.6b @@ -89,7 +98,10 @@ jobs: artifacts: name: Collect NVIDIA ASR artifacts - needs: [export-parakeet, export-nemotron] + needs: [requirements, export-parakeet, export-nemotron] + if: >- + !cancelled() && !failure() && + (needs.export-parakeet.result == 'success' || needs.export-nemotron.result == 'success') runs-on: ubuntu-24.04 outputs: matrix: ${{ steps.matrix.outputs.matrix }} @@ -127,6 +139,7 @@ jobs: }, ] } + matrix["include"] = [model for model in matrix["include"] if model["artifact_name"]] with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: output.write(f"matrix={json.dumps(matrix, separators=(',', ':'))}\n") PY diff --git a/.github/workflows/sam2.yml b/.github/workflows/sam2.yml index cb12fa8..d6b6cee 100644 --- a/.github/workflows/sam2.yml +++ b/.github/workflows/sam2.yml @@ -4,6 +4,7 @@ on: push: branches: [main] pull_request: + types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] workflow_dispatch: inputs: build_run_id: @@ -20,7 +21,13 @@ concurrency: cancel-in-progress: true jobs: + changes: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft + uses: ./.github/workflows/model-changes.yml + requirements: + needs: changes + if: needs.changes.outputs.sam2 == 'true' name: Wait for Build runs-on: ubuntu-24.04 timeout-minutes: 100 @@ -105,25 +112,29 @@ jobs: shell: bash env: GH_TOKEN: ${{ github.token }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} + SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} run: | set -euo pipefail - source_hash="$( - git ls-files -z -- 'vision/sam2/python/*.py' 'vision/sam2/python/**/*.py' \ - | xargs -0 sha256sum \ - | sha256sum \ - | cut -d' ' -f1 - )" models='{}' for slug in \ sam2.1-hiera-tiny \ sam2.1-hiera-small \ sam2.1-hiera-base-plus \ sam2.1-hiera-large; do + source_hash="$(python .github/scripts/model_ci.py key --model sam2 --model-id "facebook/${slug}")" artifact_name="vision-${slug}-onnx-v1-${source_hash:0:16}" - artifact_run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/artifacts?name=${artifact_name}&per_page=100" \ - --jq '.artifacts | map(select(.expired | not)) | sort_by(.created_at) | last | .workflow_run.id // empty' - )" + artifact_run_id="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/artifacts?per_page=100" \ + --jq ".artifacts[] | select(.name == \"$artifact_name\" and (.expired | not)) | \"$GITHUB_RUN_ID\"")" + if [[ -z "$artifact_run_id" ]]; then + artifact="$(python .github/scripts/find_successful_artifact.py \ + --repository "$GITHUB_REPOSITORY" --artifact-name "$artifact_name" \ + --default-branch "$DEFAULT_BRANCH" --event-name "$EVENT_NAME" \ + --source-branch "$SOURCE_BRANCH" --pull-request "$PULL_REQUEST" --json)" + artifact_run_id="$(jq -r '.run_id' <<< "$artifact")" + fi test -n "$artifact_run_id" models="$(jq \ --arg slug "$slug" \ diff --git a/.github/workflows/whisper-asr.yml b/.github/workflows/whisper-asr.yml index 6d7d64f..0c1e59f 100644 --- a/.github/workflows/whisper-asr.yml +++ b/.github/workflows/whisper-asr.yml @@ -4,6 +4,7 @@ on: push: branches: [main] pull_request: + types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] workflow_dispatch: inputs: build_run_id: @@ -20,7 +21,13 @@ concurrency: cancel-in-progress: true jobs: + changes: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft + uses: ./.github/workflows/model-changes.yml + requirements: + needs: changes + if: needs.changes.outputs.whisper == 'true' name: Wait for Build runs-on: ubuntu-24.04 timeout-minutes: 100 From e6b9577f5d80fc15396f87cca49f714b0899111c Mon Sep 17 00:00:00 2001 From: contentis Date: Wed, 23 Sep 2026 10:48:30 +0200 Subject: [PATCH 2/8] Ignore skipped draft builds when resolving model test inputs --- .github/workflows/nvidia-asr.yml | 2 +- .github/workflows/sam2.yml | 2 +- .github/workflows/whisper-asr.yml | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/nvidia-asr.yml b/.github/workflows/nvidia-asr.yml index 0eb6940..501f3d0 100644 --- a/.github/workflows/nvidia-asr.yml +++ b/.github/workflows/nvidia-asr.yml @@ -50,7 +50,7 @@ jobs: while [[ -z "$run_id" ]]; do run_id="$( gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" + --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" )" [[ -n "$run_id" ]] || sleep 10 done diff --git a/.github/workflows/sam2.yml b/.github/workflows/sam2.yml index d6b6cee..3149e44 100644 --- a/.github/workflows/sam2.yml +++ b/.github/workflows/sam2.yml @@ -50,7 +50,7 @@ jobs: while [[ -z "$run_id" ]]; do run_id="$( gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" + --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" )" [[ -n "$run_id" ]] || sleep 10 done diff --git a/.github/workflows/whisper-asr.yml b/.github/workflows/whisper-asr.yml index 0c1e59f..3fdd8de 100644 --- a/.github/workflows/whisper-asr.yml +++ b/.github/workflows/whisper-asr.yml @@ -50,7 +50,7 @@ jobs: while [[ -z "$run_id" ]]; do run_id="$( gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" + --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" )" [[ -n "$run_id" ]] || sleep 10 done From 224c6ed07106b4d6461934913d156dd74fece24f Mon Sep 17 00:00:00 2001 From: lspindler Date: Thu, 24 Sep 2026 09:35:29 +0200 Subject: [PATCH 3/8] Drive model CI from shared recipes and conservative dependency rules Signed-off-by: lspindler --- .coderabbit.yaml | 6 + .github/model-ci.json | 146 ++++++++ .github/model-tests/nemotron.json | 76 +++++ .github/model-tests/parakeet.json | 74 ++++ .github/model-tests/sam2.json | 100 ++++++ .github/model-tests/whisper.json | 105 ++++++ .github/scripts/model_ci.py | 318 ++++++++++++------ .github/scripts/run_model_ci.py | 106 ++++++ .github/tests/test_model_ci.py | 273 +++++++++++++++ .github/workflows/ci.yml | 61 +++- .github/workflows/export-nvidia-asr-model.yml | 297 ---------------- .github/workflows/export-sam2-model.yml | 181 ---------- .github/workflows/export-whisper-model.yml | 260 -------------- .github/workflows/model-changes.yml | 32 -- .github/workflows/model-test.yml | 109 ++++++ .github/workflows/nvidia-asr.yml | 233 ------------- .github/workflows/sam2.yml | 263 --------------- .github/workflows/whisper-asr.yml | 292 ---------------- REVIEW_GUIDELINES.md | 26 ++ docs/model-ci.md | 106 ++++++ 20 files changed, 1404 insertions(+), 1660 deletions(-) create mode 100644 .github/model-ci.json create mode 100644 .github/model-tests/nemotron.json create mode 100644 .github/model-tests/parakeet.json create mode 100644 .github/model-tests/sam2.json create mode 100644 .github/model-tests/whisper.json create mode 100644 .github/scripts/run_model_ci.py create mode 100644 .github/tests/test_model_ci.py delete mode 100644 .github/workflows/export-nvidia-asr-model.yml delete mode 100644 .github/workflows/export-sam2-model.yml delete mode 100644 .github/workflows/export-whisper-model.yml delete mode 100644 .github/workflows/model-changes.yml create mode 100644 .github/workflows/model-test.yml delete mode 100644 .github/workflows/nvidia-asr.yml delete mode 100644 .github/workflows/sam2.yml delete mode 100644 .github/workflows/whisper-asr.yml create mode 100644 REVIEW_GUIDELINES.md create mode 100644 docs/model-ci.md diff --git a/.coderabbit.yaml b/.coderabbit.yaml index a96fa08..d68b9e5 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -21,3 +21,9 @@ reviews: base_branches: [".*"] auto_incremental_review: true auto_pause_after_reviewed_commits: 0 + +knowledge_base: + code_guidelines: + enabled: true + filePatterns: + - "REVIEW_GUIDELINES.md" diff --git a/.github/model-ci.json b/.github/model-ci.json new file mode 100644 index 0000000..7ff5868 --- /dev/null +++ b/.github/model-ci.json @@ -0,0 +1,146 @@ +{ + "version": 1, + "ignore": [ + "**/*.md", + "*.md", + ".coderabbit.yaml", + ".coderabbit.yml", + "LICENSE", + "THIRD_PARTY_NOTICES.md" + ], + "groups": { + "native": [ + "CMakeLists.txt", + "CMakePresets.json", + "cmake/**", + "common/**/*.cpp", + "common/**/*.h", + "common/*.cpp", + "common/*.h", + "common/**/CMakeLists.txt", + "common/CMakeLists.txt" + ], + "nemo-export": [ + "asr/rnnt/python/tools/nemo_preprocessor_export.py", + "asr/rnnt/python/rnnt/__init__.py", + "asr/rnnt/python/rnnt/audio.py" + ], + "rnnt-runtime": [ + "asr/rnnt/cpp/asr*", + "asr/rnnt/CMakeLists.txt", + "asr/rnnt/python/rnnt/validation/**", + "asr/rnnt/python/tests/**" + ], + "audio-test": [ + "assets/sample.wav" + ] + }, + "models": { + "parakeet": { + "export": { + "paths": [ + "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", + "asr/rnnt/python/rnnt/parakeet_tdt/nemo_backend.py", + "asr/rnnt/python/rnnt/parakeet_tdt/__init__.py" + ], + "groups": [ + "nemo-export" + ] + }, + "runtime": { + "paths": [ + "asr/rnnt/cpp/parakeet*", + "asr/rnnt/python/rnnt/parakeet_tdt/**" + ], + "groups": [ + "native", + "rnnt-runtime", + "audio-test" + ] + } + }, + "nemotron": { + "export": { + "paths": [ + "asr/rnnt/python/tools/export_nemotron_onnx.py", + "asr/rnnt/python/rnnt/nemotron_asr/nemo_backend.py", + "asr/rnnt/python/rnnt/nemotron_asr/__init__.py" + ], + "groups": [ + "nemo-export" + ] + }, + "runtime": { + "paths": [ + "asr/rnnt/cpp/nemotron*", + "asr/rnnt/python/rnnt/nemotron_asr/**" + ], + "groups": [ + "native", + "rnnt-runtime", + "audio-test" + ] + } + }, + "whisper": { + "export": { + "paths": [ + "asr/whisper/model_export/**", + "common/model_export/**" + ] + }, + "runtime": { + "paths": [ + "asr/whisper/**" + ], + "groups": [ + "native", + "audio-test" + ] + } + }, + "sam2": { + "export": { + "paths": [ + "vision/sam2/python/export_sam2_onnx.py", + "vision/sam2/python/modeling.py" + ] + }, + "runtime": { + "paths": [ + "vision/sam2/**", + "assets/sam2-*" + ], + "groups": [ + "native" + ] + } + } + }, + "build_only": { + "flux2": { + "targets": [ + "din_flux2_cli" + ], + "paths": [ + "image_gen/flux2/**" + ], + "groups": [ + "native" + ], + "reason": "GPU inference requires model assets and a GPU runner; existing hosted CI only builds this sample." + }, + "base": { + "targets": [ + "din_base_onnx" + ], + "paths": [ + "tests/base/**" + ], + "groups": [ + "native" + ], + "reason": "Existing hosted CI builds this sample; it has no portable CPU smoke recipe." + } + } +} diff --git a/.github/model-tests/nemotron.json b/.github/model-tests/nemotron.json new file mode 100644 index 0000000..9dfd0f4 --- /dev/null +++ b/.github/model-tests/nemotron.json @@ -0,0 +1,76 @@ +{ + "name": "nemotron", + "source_roots": [ + "asr/rnnt" + ], + "targets": [ + "din_asr_nemotron_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "soundfile", + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" + ] + ], + "command": [ + "{python}", + "asr/rnnt/python/tools/export_nemotron_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--external-data" + ], + "required_files": [ + "preprocessor.onnx", + "encoder_step.onnx", + "predict_step.onnx", + "metadata.json", + "vocab.txt", + "decoder_init_hidden.f32", + "decoder_init_cell.f32" + ], + "env": { + "PYTHONPATH": "asr/rnnt/python" + } + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "nemotron-3.5-asr-streaming-0.6b", + "model_id": "nvidia/nemotron-3.5-asr-streaming-0.6b", + "precision": "fp16", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/model-tests/parakeet.json b/.github/model-tests/parakeet.json new file mode 100644 index 0000000..8a82029 --- /dev/null +++ b/.github/model-tests/parakeet.json @@ -0,0 +1,74 @@ +{ + "name": "parakeet", + "source_roots": [ + "asr/rnnt" + ], + "targets": [ + "din_asr_parakeet_tdt_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "soundfile", + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" + ] + ], + "command": [ + "{python}", + "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--external-data" + ], + "required_files": [ + "preprocessor.onnx", + "encoder.onnx", + "predict_step.onnx", + "metadata.json", + "tokenizer.json" + ], + "env": { + "PYTHONPATH": "asr/rnnt/python" + } + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "parakeet-tdt", + "model_id": "nvidia/parakeet-tdt-0.6b-v3", + "precision": "fp16", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/model-tests/sam2.json b/.github/model-tests/sam2.json new file mode 100644 index 0000000..411171d --- /dev/null +++ b/.github/model-tests/sam2.json @@ -0,0 +1,100 @@ +{ + "name": "sam2", + "source_roots": [ + "vision/sam2" + ], + "targets": [ + "din_sam2_cli", + "din_sam2_verify_masks" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch", + "torchvision" + ], + [ + "{python}", + "-m", + "pip", + "install", + "huggingface_hub", + "numpy", + "onnx", + "onnxscript", + "sam-2 @ git+https://github.com/facebookresearch/sam2.git" + ] + ], + "command": [ + "{python}", + "vision/sam2/python/export_sam2_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--max-points", + "3" + ], + "required_files": [ + "image_preprocess.onnx", + "image_encoder.onnx", + "sam2_decoder.onnx", + "sam2_decoder_propagate.onnx", + "memory_attention.onnx", + "memory_encoder.onnx", + "mask_postprocess.onnx", + "metadata.json" + ] + }, + "smoke": { + "decode_base64": { + "assets/sam2-ci.png.base64": "assets/sam2-ci.png" + }, + "mkdir": [ + "out/{slug}" + ], + "command": [ + "{executable}", + "assets/sam2-ci.png", + "assets/sam2-ci-prompt.json", + "out/{slug}", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "sam2.1-hiera-tiny", + "model_id": "facebook/sam2.1-hiera-tiny", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-small", + "model_id": "facebook/sam2.1-hiera-small", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-base-plus", + "model_id": "facebook/sam2.1-hiera-base-plus", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-large", + "model_id": "facebook/sam2.1-hiera-large", + "precision": "fp16" + } + ] +} diff --git a/.github/model-tests/whisper.json b/.github/model-tests/whisper.json new file mode 100644 index 0000000..55ee04f --- /dev/null +++ b/.github/model-tests/whisper.json @@ -0,0 +1,105 @@ +{ + "name": "whisper", + "source_roots": [ + "asr/whisper" + ], + "targets": [ + "din_asr_whisper_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "safetensors", + "transformers" + ] + ], + "command": [ + "{python}", + "asr/whisper/model_export/export_whisper.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{precision}", + "--attention", + "{attention}" + ], + "required_files": [ + "encoder.onnx", + "decoder.onnx", + "mel.onnx", + "vocab.json" + ] + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "whisper-tiny", + "model_id": "openai/whisper-tiny", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-base", + "model_id": "openai/whisper-base", + "precision": "fp32", + "attention": "sdpa" + }, + { + "slug": "whisper-small", + "model_id": "openai/whisper-small", + "precision": "fp16", + "attention": "math", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-medium", + "model_id": "openai/whisper-medium", + "precision": "fp32", + "attention": "sdpa" + }, + { + "slug": "whisper-large-v3", + "model_id": "openai/whisper-large-v3", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-large-v3-turbo", + "model_id": "openai/whisper-large-v3-turbo", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/scripts/model_ci.py b/.github/scripts/model_ci.py index 207a4eb..deee66f 100644 --- a/.github/scripts/model_ci.py +++ b/.github/scripts/model_ci.py @@ -1,112 +1,240 @@ #!/usr/bin/env python3 -"""Select affected model tests and fingerprint their ONNX export inputs.""" +"""Plan model CI from optional dependency declarations; uncertainty runs more work.""" import argparse import fnmatch import hashlib import json import os +import re import subprocess from pathlib import Path -SHARED_RUNTIME = [ - "CMakeLists.txt", - "CMakePresets.json", - "cmake/*", - "common/*.cpp", - "common/*.h", - "common/*CMakeLists.txt", - ".github/workflows/ci.yml", - ".github/scripts/*", - ".github/workflows/model-changes.yml", -] -EXPORTS = { - "whisper": [ - "asr/whisper/model_export/export_whisper.py", - "common/model_export/*.py", - ".github/workflows/export-whisper-model.yml", - ], - "parakeet": [ - "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", - "asr/rnnt/python/rnnt/parakeet_tdt/nemo_backend.py", - "asr/rnnt/python/rnnt/parakeet_tdt/__init__.py", - ], - "nemotron": [ - "asr/rnnt/python/tools/export_nemotron_onnx.py", - "asr/rnnt/python/rnnt/nemotron_asr/nemo_backend.py", - "asr/rnnt/python/rnnt/nemotron_asr/__init__.py", - ], - "sam2": [ - "vision/sam2/python/export_sam2_onnx.py", - "vision/sam2/python/modeling.py", - ".github/workflows/export-sam2-model.yml", - ], -} -for model in ("parakeet", "nemotron"): - EXPORTS[model] += [ - "asr/rnnt/python/tools/nemo_preprocessor_export.py", - "asr/rnnt/python/rnnt/__init__.py", - "asr/rnnt/python/rnnt/audio.py", - ".github/workflows/export-nvidia-asr-model.yml", - ] -RUNTIME = { - "whisper": ["asr/whisper/*", "assets/sample.wav", ".github/workflows/whisper-asr.yml"], - "sam2": ["vision/sam2/*", "assets/sam2-*", ".github/workflows/sam2.yml"], -} -for model, package in (("parakeet", "parakeet_tdt"), ("nemotron", "nemotron_asr")): - RUNTIME[model] = [ - f"asr/rnnt/cpp/{model}*", - "asr/rnnt/cpp/asr*", - "asr/rnnt/CMakeLists.txt", - f"asr/rnnt/python/rnnt/{package}/*", - "asr/rnnt/python/rnnt/validation/*", - "asr/rnnt/python/tests/*", - "assets/sample.wav", - ".github/workflows/nvidia-asr.yml", - ] - - -def affected(model, files): - patterns = SHARED_RUNTIME + RUNTIME[model] + EXPORTS[model] - return any(fnmatch.fnmatchcase(path, pattern) for path in files if not path.endswith(".md") for pattern in patterns) - - -def export_key(model, model_id, attention): - # Index blob IDs are independent of checkout line endings and include deleted/renamed inputs. - sources = subprocess.check_output(["git", "ls-files", "--stage", "--", *EXPORTS[model]]) - options = json.dumps([model, model_id, attention]).encode() - return hashlib.sha256(sources + options).hexdigest()[:16] +ROOT = Path(__file__).resolve().parents[2] +CONFIG = '.github/model-ci.json' +RECIPES = '.github/model-tests' +CONTROL = ['.github/scripts/**', '.github/workflows/**', '.github/tests/**', CONFIG] +NATIVE_SUFFIXES = {'.c', '.cc', '.cpp', '.cxx', '.h', '.hh', '.hpp', '.hxx', '.cu', '.cuh'} + + +def git(*args, root=ROOT): + return subprocess.check_output(['git', *args], cwd=root) + + +def matches(path, patterns): + # fnmatch intentionally lets * span directories. Both common/*.h and ** work. + return any(fnmatch.fnmatchcase(path, pattern) for pattern in patterns) + + +def read_config(root=ROOT): + try: + config = json.loads((root / CONFIG).read_text()) + if not isinstance(config, dict) or config.get('version') != 1: + raise ValueError('unsupported dependency configuration version') + for field in ('groups', 'models', 'build_only'): + if not isinstance(config.get(field, {}), dict): + raise ValueError(f'{field} must be an object') + def patterns(value): + if not isinstance(value, list) or not all(isinstance(p, str) for p in value): + raise ValueError('dependency patterns and group names must be string lists') + patterns(config.get('ignore', [])) + for value in config.get('groups', {}).values(): + patterns(value) + sections = list(config.get('build_only', {}).values()) + for model in config.get('models', {}).values(): + if not isinstance(model, dict): + raise ValueError('model dependencies must be objects') + sections.extend(model.values()) + for section in sections: + if not isinstance(section, dict): + raise ValueError('dependency sections must be objects') + patterns(section.get('paths', [])) + patterns(section.get('groups', [])) + return config + except (OSError, ValueError) as error: + print(f'::warning::Dependency optimization disabled: {error}') + return {} + + +def recipes(root=ROOT): + result = {} + for path in sorted((root / RECIPES).glob('*.json')): + recipe = json.loads(path.read_text()) + name = recipe['name'] + if name in result or path.stem != name or not re.fullmatch(r'[a-z0-9-]+', name): + raise ValueError(f'invalid or duplicate recipe name: {path}') + if not recipe['targets'] or not recipe['variants'] or not recipe['source_roots']: + raise ValueError(f'incomplete execution recipe: {name}') + slugs = [v['slug'] for v in recipe['variants']] + if len(set(slugs)) != len(slugs) or any(not re.fullmatch(r'[a-z0-9.-]+', s) for s in slugs): + raise ValueError(f'invalid or duplicate variant slug: {name}') + for target in recipe['targets']: + if not re.fullmatch(r'[A-Za-z0-9_-]+', target): + raise ValueError(f'invalid CMake target: {target}') + for variant in recipe['variants']: + if variant['precision'] not in ('fp16', 'fp32') or not variant['model_id']: + raise ValueError(f'invalid export variant: {variant}') + result[name] = recipe + if not result: + raise ValueError('No model execution recipes found; refusing to report empty coverage') + return result + + +def inputs(section, config): + if not isinstance(section, dict): + return None + patterns = section.get('paths', []).copy() + for group in section.get('groups', []): + if group not in config.get('groups', {}): + return None + patterns.extend(config['groups'][group]) + return patterns or None + + +def discover_unregistered(config, entries, root=ROOT): + """Discover literal sample targets independently of the optimization registry. + + CMake remains responsible for actual build dependencies. Unknown/dynamic + declarations get a full build, never a guessed partial target list. + """ + known = {t for entry in entries.values() for t in entry['targets']} + known.update(t for item in config.get('build_only', {}).values() for t in item['targets']) + unknown = [] + tracked = git('ls-files', '-z', root=root).decode().split('\0') + for path in tracked: + if not path.endswith('CMakeLists.txt') or path == 'CMakeLists.txt': + continue + text = (root / path).read_text() + text = re.sub(r'#[^\n]*', '', text) + for target in re.findall(r'\badd_(?:din_)?executable\s*\(\s*([^\s)]+)', text, re.I): + target = target.strip('"') + if target not in known: + unknown.append(f'{path}: {target}') + return sorted(unknown) + + +def plan(files, config, entries, unknown_targets=()): + reasons = [] + force_all = files is None or not config + force_export = files is None or not config + if files is None: + reasons.append('No reliable change base (manual/initial run): run all') + files = [p for p in files or [] if p] + # Positive exemptions, not a blanket "all YAML is harmless" rule. + relevant = [p for p in files if not matches(p, config.get('ignore', []))] + if config and files and not relevant: + return dict(build=False, targets=[], matrix={'include': []}, format=False, + reasons=['Only explicitly ignored documentation/review files changed'], unknown_targets=[]) + if unknown_targets: + force_all = force_export = True + reasons.append('Unregistered CMake targets: full build and all available smoke recipes') + paths_by_model = {} + for name in entries: + dependency = config.get('models', {}).get(name, {}) + paths_by_model[name] = (inputs(dependency.get('export'), config), inputs(dependency.get('runtime'), config)) + owned = [p for pair in paths_by_model.values() for patterns in pair if patterns for p in patterns] + for item in config.get('build_only', {}).values(): + owned.extend(inputs(item, config) or []) + recipe_paths = [f'{RECIPES}/{name}.json' for name in entries] + unknown_files = [p for p in relevant if not matches(p, owned + CONTROL + recipe_paths)] + if unknown_files: + force_all = force_export = True + reasons.append('Unowned changes: ' + ', '.join(unknown_files)) + control_changed = any(matches(p, CONTROL) for p in relevant) + selected, targets = [], set() + full_build = force_all or control_changed + for name, entry in entries.items(): + export_paths, runtime_paths = paths_by_model[name] + undeclared = export_paths is None or runtime_paths is None + changed = any(matches(p, (export_paths or []) + (runtime_paths or []) + [f'{RECIPES}/{name}.json']) for p in relevant) + if force_all or control_changed or undeclared or changed: + reasons.append(f'{name}: ' + ('missing dependency rules; run conservatively' if undeclared else 'affected or full validation')) + targets.update(entry['targets']) + for variant in entry['variants']: + selected.append({'model': name, 'variant': variant['slug'], 'force_export': force_export or export_paths is None}) + for name, item in config.get('build_only', {}).items(): + patterns = inputs(item, config) + if force_all or control_changed or patterns is None or any(matches(p, patterns) for p in relevant): + targets.update(item['targets']) + reasons.append(f'{name}: build only ({item["reason"]})') + return dict(build=bool(targets) or full_build, targets=[] if full_build else sorted(targets), + matrix={'include': selected}, format=files is None or any(Path(p).suffix in NATIVE_SUFFIXES or p == '.clang-format' for p in files), + reasons=reasons, unknown_targets=list(unknown_targets)) + + +def changed_files(event, root=ROOT): + pr = event.get('pull_request') + base = pr['base']['sha'] if pr else event.get('before') + if not base or not base.strip('0'): + return None + try: + git('cat-file', '-e', f'{base}^{{commit}}', root=root) + if pr: + base = git('merge-base', base, 'HEAD', root=root).decode().strip() + return git('diff', '--no-renames', '--name-only', '-z', base, 'HEAD', root=root).decode().split('\0') + except subprocess.CalledProcessError: + return None + + +def export_key(name, variant, config, entries, root=ROOT, force=False): + entry = entries[name] + patterns = inputs(config.get('models', {}).get(name, {}).get('export'), config) + fingerprint = {'recipe': entry['export'], 'variant': variant, 'inputs': patterns, + 'environment': 'ubuntu-24.04-python-3.12-cpu'} + # The runner and export workflow affect all export recipes, unlike runtime code. + patterns = (patterns or []) + ['.github/scripts/model_ci.py', '.github/scripts/run_model_ci.py', '.github/workflows/model-test.yml'] + owned = CONTROL + config.get('ignore', []) + [RECIPES + '/**'] + for dependency in config.get('models', {}).values(): + for stage in ('export', 'runtime'): + owned.extend(inputs(dependency.get(stage), config) or []) + for component in config.get('build_only', {}).values(): + owned.extend(inputs(component, config) or []) + sources = [] + for record in git('ls-files', '--stage', '-z', root=root).decode().split('\0'): + if not record: + continue + metadata, path = record.split('\t', 1) + if matches(path, patterns) or not matches(path, owned): + sources.append([path, metadata]) + fingerprint['sources'] = sources + if force or not inputs(config.get('models', {}).get(name, {}).get('export'), config): + # Unknown export dependencies must not reuse an earlier run's model. + fingerprint['uncertain_run'] = [os.getenv('GITHUB_RUN_ID', 'local'), os.getenv('GITHUB_RUN_ATTEMPT', '1'), git('rev-parse', 'HEAD', root=root).decode().strip()] + return hashlib.sha256(json.dumps(fingerprint, sort_keys=True).encode()).hexdigest()[:24] + + +def write_outputs(values): + with Path(os.environ['GITHUB_OUTPUT']).open('a') as output: + for key, value in values.items(): + output.write(f'{key}={json.dumps(value, separators=(",", ":")) if not isinstance(value, str) else value}\n') def main(): parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("mode", choices=["changes", "key"]) - parser.add_argument("--model", choices=EXPORTS) - parser.add_argument("--model-id", default="") - parser.add_argument("--attention", default="") + parser.add_argument('mode', choices=['plan', 'build']) + parser.add_argument('--preset') args = parser.parse_args() - if args.mode == "key": - if not args.model or not args.model_id: - parser.error("key requires --model and --model-id") - print(export_key(args.model, args.model_id, args.attention)) + if args.mode == 'build': + targets = json.loads(os.environ.get('BUILD_TARGETS', '[]')) + command = ['cmake', '--build', '--preset', args.preset, '--parallel'] + if targets: + command += ['--target', *targets] + subprocess.run(command, check=True) return - event = json.loads(Path(os.environ["GITHUB_EVENT_PATH"]).read_text()) - pr = event.get("pull_request") - base = pr["base"]["sha"] if pr else event.get("before") - files = None - if base and subprocess.run(["git", "cat-file", "-e", f"{base}^{{commit}}"], capture_output=True).returncode == 0: - if pr: - base = subprocess.check_output(["git", "merge-base", base, "HEAD"], text=True).strip() - files = ( - subprocess.check_output(["git", "diff", "--no-renames", "--name-only", "-z", base, "HEAD"]) - .decode() - .split("\0") - ) - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - for model in EXPORTS: - changed = files is None or affected(model, files) - output.write(f"{model}={str(changed).lower()}\n") - - -if __name__ == "__main__": + entries = recipes() + config = read_config() + event = json.loads(Path(os.environ['GITHUB_EVENT_PATH']).read_text()) + result = plan(changed_files(event), config, entries, discover_unregistered(config, entries)) + print(json.dumps(result, indent=2)) + write_outputs({'build': result['build'], 'targets': result['targets'], 'matrix': result['matrix'], + 'models': bool(result['matrix']['include']), 'format': result['format']}) + if os.getenv('GITHUB_STEP_SUMMARY'): + with Path(os.environ['GITHUB_STEP_SUMMARY']).open('a') as summary: + summary.write('## CI selection\n\n' + '\n'.join('- ' + reason for reason in result['reasons']) + '\n') + if result['unknown_targets']: + summary.write('\nUnregistered executables are included in the full build. Add an execution recipe to run their inference tests:\n\n') + summary.write('\n'.join('- ' + name for name in result['unknown_targets']) + '\n') + + +if __name__ == '__main__': main() diff --git a/.github/scripts/run_model_ci.py b/.github/scripts/run_model_ci.py new file mode 100644 index 0000000..237ed2c --- /dev/null +++ b/.github/scripts/run_model_ci.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python3 +"""Execute model recipes without model-specific workflow branches.""" + +import argparse +import base64 +import json +import os +import shutil +import subprocess +import sys +from pathlib import Path + +from model_ci import ROOT, export_key, read_config, recipes, write_outputs + + +def expand(command, values): + return [argument.format_map(values) for argument in command] + + +def run(command, values, env=None): + subprocess.run(expand(command, values), cwd=ROOT, env=env, check=True) + + +def artifact_name(slug, precision, key): + return f'model-{slug}-{precision}-{key}' + + +def find_artifact(name): + command = [sys.executable, str(ROOT / '.github/scripts/find_successful_artifact.py'), + '--repository', os.environ['GITHUB_REPOSITORY'], '--artifact-name', name, + '--default-branch', os.environ['DEFAULT_BRANCH'], '--event-name', os.environ['EVENT_NAME'], + '--source-branch', os.environ['SOURCE_BRANCH'], '--pull-request', os.environ.get('PULL_REQUEST', ''), '--json'] + return json.loads(subprocess.check_output(command, text=True)) + + +def export_model(entry, variant, values, key): + definition = entry['export'] + env = dict(os.environ, **definition.get('env', {})) + for command in definition['install']: + run(command, values, env) + output = Path(values['output']) + precisions = [variant['precision']] + if variant.get('fallback_precision'): + precisions.append(variant['fallback_precision']) + for index, precision in enumerate(precisions): + current = dict(values, precision=precision, dtype={'fp16': 'float16', 'fp32': 'float32'}[precision]) + if output.exists(): + shutil.rmtree(output) + try: + run(definition['command'], current, env) + except subprocess.CalledProcessError: + if index == len(precisions) - 1: + raise + print(f'::warning::{precision} export failed; trying {precisions[index + 1]}', flush=True) + continue + missing = [name for name in definition['required_files'] if not (output / name).is_file()] + if missing: + raise RuntimeError(f'Export did not produce required files: {missing}') + manifest = {'model_id': variant['model_id'], 'precision': precision, 'source_hash': key, + 'commit': os.getenv('GITHUB_SHA'), 'variant': variant} + (output / 'ci-export-manifest.json').write_text(json.dumps(manifest, indent=2) + '\n') + write_outputs({'name': artifact_name(variant['slug'], precision, key), 'run_id': os.environ['GITHUB_RUN_ID']}) + return + + +def smoke_test(entry, values): + definition = entry['smoke'] + for source, destination in definition.get('decode_base64', {}).items(): + (ROOT / destination).write_bytes(base64.b64decode((ROOT / source).read_text())) + for directory in definition.get('mkdir', []): + (ROOT / directory.format_map(values)).mkdir(parents=True, exist_ok=True) + executable = ROOT / 'runtime' / (entry['targets'][0] + ('.exe' if os.name == 'nt' else '')) + if os.name != 'nt': + executable.chmod(executable.stat().st_mode | 0o111) + env = dict(os.environ) + env['LD_LIBRARY_PATH'] = str(ROOT / 'runtime') + os.pathsep + env.get('LD_LIBRARY_PATH', '') + run(definition['command'], dict(values, executable=str(executable)), env) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('mode', choices=['resolve', 'export', 'smoke']) + args = parser.parse_args() + entries = recipes() + entry = entries[os.environ['MODEL']] + variant = next(v for v in entry['variants'] if v['slug'] == os.environ['VARIANT']) + values = dict(variant, python=sys.executable, output=str(ROOT / 'artifacts/ci' / variant['slug'])) + if args.mode == 'smoke': + smoke_test(entry, values) + return + key = export_key(entry['name'], variant, read_config(), entries, force=os.getenv('FORCE_EXPORT') == 'true') + if args.mode == 'export': + export_model(entry, variant, values, key) + return + for precision in dict.fromkeys([variant['precision'], variant.get('fallback_precision')]): + if not precision: + continue + found = find_artifact(artifact_name(variant['slug'], precision, key)) + if found['found']: + write_outputs(found) + return + write_outputs({'found': False, 'name': '', 'run_id': ''}) + + +if __name__ == '__main__': + main() diff --git a/.github/tests/test_model_ci.py b/.github/tests/test_model_ci.py new file mode 100644 index 0000000..a2819c1 --- /dev/null +++ b/.github/tests/test_model_ci.py @@ -0,0 +1,273 @@ +import copy +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +SCRIPTS = Path(__file__).resolve().parents[1] / 'scripts' +sys.path.insert(0, str(SCRIPTS)) +import model_ci as ci +import run_model_ci as runner + + +class SelectionTests(unittest.TestCase): + def setUp(self): + self.config = ci.read_config() + self.entries = {name: value for name, value in ci.recipes().items() if name in {"whisper", "parakeet", "nemotron", "sam2"}} + + def plan(self, *paths, **kwargs): + return ci.plan(list(paths), self.config, self.entries, **kwargs) + + def selected(self, result): + return {row['model'] for row in result['matrix']['include']} + + def test_review_config_and_docs_do_not_build_or_export(self): + for path in ['.coderabbit.yaml', 'README.md', 'asr/rnnt/README.md']: + with self.subTest(path=path): + result = self.plan(path) + self.assertFalse(result['build']) + self.assertEqual(self.selected(result), set()) + + def test_parakeet_runtime_selects_only_parakeet(self): + result = self.plan('asr/rnnt/cpp/parakeet_tdt.cpp') + self.assertEqual(self.selected(result), {'parakeet'}) + self.assertEqual(result['targets'], ['din_asr_parakeet_tdt_cli']) + self.assertFalse(result['matrix']['include'][0]['force_export']) + + def test_nemotron_runtime_selects_only_nemotron(self): + self.assertEqual(self.selected(self.plan('asr/rnnt/cpp/nemotron.cpp')), {'nemotron'}) + + def test_export_change_also_selects_inference(self): + self.assertEqual(self.selected(self.plan('asr/whisper/model_export/export_whisper.py')), {'whisper'}) + + def test_shared_audio_runs_all_models(self): + self.assertEqual(self.selected(self.plan('common/io/audio.cpp')), set(self.entries)) + + def test_shared_export_selects_only_consumers(self): + self.assertEqual(self.selected(self.plan('asr/rnnt/python/tools/nemo_preprocessor_export.py')), {'parakeet', 'nemotron'}) + + def test_test_asset_selects_consumers_without_forcing_export(self): + result = self.plan('assets/sample.wav') + self.assertEqual(self.selected(result), {'whisper', 'parakeet', 'nemotron'}) + self.assertFalse(any(row['force_export'] for row in result['matrix']['include'])) + + def test_unknown_file_runs_everything_and_forces_export(self): + result = self.plan('new_component/kernel.cu') + self.assertEqual(self.selected(result), set(self.entries)) + self.assertEqual(result['targets'], []) + self.assertTrue(all(row['force_export'] for row in result['matrix']['include'])) + + def test_unknown_yaml_is_not_silently_ignored(self): + self.assertEqual(self.selected(self.plan('model-settings.yaml')), set(self.entries)) + + def test_missing_dependency_entry_always_runs_existing_recipe(self): + del self.config['models']['parakeet'] + result = self.plan('vision/sam2/cpp/main.cpp') + self.assertIn('parakeet', self.selected(result)) + self.assertTrue(next(row['force_export'] for row in result['matrix']['include'] if row['model'] == 'parakeet')) + + def test_new_recipe_requires_no_planner_change(self): + self.entries['new-model'] = copy.deepcopy(self.entries['parakeet']) + self.entries['new-model']['targets'] = ['din_new_cli'] + self.entries['new-model']['variants'] = [{'slug': 'new-variant'}] + result = self.plan('assets/sample.wav') + self.assertIn({'model': 'new-model', 'variant': 'new-variant', 'force_export': True}, result['matrix']['include']) + self.assertIn('din_new_cli', result['targets']) + + def test_unknown_group_runs_component_conservatively(self): + self.config['models']['whisper']['export']['groups'] = ['missing'] + result = self.plan('asr/rnnt/cpp/parakeet_tdt.cpp') + self.assertIn('whisper', self.selected(result)) + + def test_unknown_target_forces_full_build(self): + result = self.plan('assets/sample.wav', unknown_targets=['models/new/CMakeLists.txt: din_new']) + self.assertEqual(result['targets'], []) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_manual_or_unavailable_base_runs_all(self): + result = ci.plan(None, self.config, self.entries) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_missing_configuration_runs_all(self): + result = ci.plan(['README.md'], {}, self.entries) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_workflow_change_validates_all_models(self): + self.assertEqual(self.selected(self.plan('.github/workflows/ci.yml')), set(self.entries)) + + def test_flux_changes_build_without_unrelated_inference(self): + result = self.plan('image_gen/flux2/cpp/flux.cpp') + self.assertEqual(result['targets'], ['din_flux2_cli']) + self.assertEqual(self.selected(result), set()) + + def test_variant_matrix_preserves_existing_coverage(self): + result = ci.plan(None, self.config, self.entries) + self.assertEqual(len(result['matrix']['include']), 12) + + +class GitTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.git('init', '-q') + self.git('config', 'user.name', 'CI Test') + self.git('config', 'user.email', 'ci@example.invalid') + self.write('export.py', 'initial') + self.write('runtime.cpp', 'initial') + self.git('add', '.') + self.git('commit', '-qm', 'initial') + self.base = self.git('rev-parse', 'HEAD').strip() + self.config = {'models': {'demo': {'export': {'paths': ['export.py']}, 'runtime': {'paths': ['runtime.cpp']}}}} + self.entries = {'demo': {'export': {'command': ['export.py']}}} + self.variant = {'model_id': 'model/revision', 'precision': 'fp16'} + + def git(self, *args): + return ci.git(*args, root=self.root).decode() + + def write(self, path, content): + target = self.root / path + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + + def key(self, **kwargs): + return ci.export_key('demo', self.variant, self.config, self.entries, root=self.root, **kwargs) + + def test_runtime_change_preserves_export_key(self): + before = self.key() + self.write('runtime.cpp', 'changed') + self.git('add', '.') + self.assertEqual(before, self.key()) + + def test_exporter_edit_delete_and_rename_change_key(self): + previous = self.key() + self.write('export.py', 'changed') + self.git('add', '.') + self.assertNotEqual(previous, self.key()) + previous = self.key() + self.git('mv', 'export.py', 'renamed.py') + self.assertNotEqual(previous, self.key()) + previous = self.key() + self.git('rm', '-f', 'renamed.py') + self.assertNotEqual(previous, self.key()) + + def test_unknown_file_stays_in_key_on_future_runs(self): + before = self.key() + self.write('unowned_helper.py', 'new export dependency') + self.git('add', '.') + self.assertNotEqual(before, self.key()) + + def test_export_settings_change_key(self): + before = self.key() + self.entries['demo']['export']['install'] = ['changed dependency'] + self.assertNotEqual(before, self.key()) + before = self.key() + self.variant['model_id'] = 'different/model' + self.assertNotEqual(before, self.key()) + + def test_fallback_cannot_reuse_previous_run(self): + with patch.dict(os.environ, {'GITHUB_RUN_ID': '1'}): + before = self.key(force=True) + with patch.dict(os.environ, {'GITHUB_RUN_ID': '2'}): + self.assertNotEqual(before, self.key(force=True)) + + def test_deletion_and_rename_list_both_paths(self): + self.git('mv', 'runtime.cpp', 'renamed.cpp') + self.git('commit', '-qm', 'rename') + changed = ci.changed_files({'before': self.base}, root=self.root) + self.assertIn('runtime.cpp', changed) + self.assertIn('renamed.cpp', changed) + + def test_pr_uses_merge_base(self): + self.git('checkout', '-qb', 'base-update') + self.write('base-only.cpp', 'base change') + self.git('add', '.') + self.git('commit', '-qm', 'base update') + base_tip = self.git('rev-parse', 'HEAD').strip() + self.git('checkout', '-qb', 'feature', self.base) + self.write('runtime.cpp', 'feature change') + self.git('add', '.') + self.git('commit', '-qm', 'feature') + changed = ci.changed_files({'pull_request': {'base': {'sha': base_tip}}}, root=self.root) + self.assertIn('runtime.cpp', changed) + self.assertNotIn('base-only.cpp', changed) + + def test_unknown_target_discovery_is_independent_of_registry(self): + self.write('new/CMakeLists.txt', 'add_din_executable(din_new_cli main.cpp)') + self.git('add', '.') + found = ci.discover_unregistered({}, {}, root=self.root) + self.assertEqual(found, ['new/CMakeLists.txt: din_new_cli']) + + +class ExecutionTests(unittest.TestCase): + def test_subprocess_failures_propagate(self): + with patch.object(subprocess, 'run', side_effect=subprocess.CalledProcessError(1, ['cli'])): + with self.assertRaises(subprocess.CalledProcessError): + runner.run(['{executable}', '--model-dir', '{output}'], {'executable': 'cli', 'output': 'model'}) + + def test_command_arguments_do_not_go_through_shell(self): + with patch.object(subprocess, 'run') as run: + runner.run(['cli', '{output}'], {'output': 'directory with spaces'}) + self.assertEqual(run.call_args.args[0], ['cli', 'directory with spaces']) + self.assertNotIn('shell', run.call_args.kwargs) + + def test_all_current_recipes_expand(self): + for entry in ci.recipes().values(): + for variant in entry['variants']: + values = dict(variant, python='python', output='model', executable='cli', dtype='float16') + runner.expand(entry['export']['command'], values) + runner.expand(entry['smoke']['command'], values) + for command in entry['export']['install']: + runner.expand(command, values) + + +class ExportExecutionTests(unittest.TestCase): + def test_export_retries_fallback_and_records_precision(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + outputs = Path(directory) / 'outputs' + definition = {'export': {'install': [['install']], 'command': ['export', '{precision}'], 'required_files': ['model.onnx']}} + variant = {'slug': 'example', 'model_id': 'org/model', 'precision': 'fp16', 'fallback_precision': 'fp32'} + values = dict(variant, output=str(output)) + def execute(command, expanded, env): + if command[0] == 'install': + return + output.mkdir(parents=True) + if expanded['precision'] == 'fp16': + (output / 'partial').write_text('partial export') + raise subprocess.CalledProcessError(1, command) + self.assertFalse((output / 'partial').exists()) + (output / 'model.onnx').write_text('valid model') + with patch.object(runner, 'run', side_effect=execute), patch.dict(os.environ, {'GITHUB_OUTPUT': str(outputs), 'GITHUB_RUN_ID': '123'}): + runner.export_model(definition, variant, values, 'fingerprint') + self.assertIn('name=model-example-fp32-fingerprint', outputs.read_text()) + self.assertEqual(json.loads((output / 'ci-export-manifest.json').read_text())['precision'], 'fp32') + + def test_successful_command_with_missing_artifacts_fails(self): + with tempfile.TemporaryDirectory() as directory: + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['missing.onnx']}} + variant = {'precision': 'fp32'} + with patch.object(runner, 'run'): + with self.assertRaisesRegex(RuntimeError, 'required files'): + runner.export_model(entry, variant, {'output': str(Path(directory) / 'model')}, 'key') + + +class ConfigurationTests(unittest.TestCase): + def test_malformed_optional_configuration_disables_optimization(self): + for content in ['{', '[]', '{"version": 1, "models": {"demo": []}}', '{"version": 1, "groups": {"shared": "not-a-list"}}']: + with self.subTest(content=content), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + (root / '.github').mkdir() + (root / ci.CONFIG).write_text(content) + self.assertEqual(ci.read_config(root), {}) + + +if __name__ == '__main__': + unittest.main() diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index aae8c6c..a55bb53 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -19,8 +19,31 @@ env: CMAKE_BUILD_PARALLEL_LEVEL: "4" jobs: - clang-format: + plan: if: github.event_name != 'pull_request' || !github.event.pull_request.draft + runs-on: ubuntu-24.04 + outputs: + build: ${{ steps.plan.outputs.build }} + targets: ${{ steps.plan.outputs.targets }} + models: ${{ steps.plan.outputs.models }} + matrix: ${{ steps.plan.outputs.matrix }} + format: ${{ steps.plan.outputs.format }} + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + - uses: actions/setup-python@v7 + with: + python-version: '3.12' + - name: Validate CI selection and execution recipes + run: python -m unittest discover -s .github/tests -v + - name: Select affected components + id: plan + run: python .github/scripts/model_ci.py plan + + clang-format: + needs: plan + if: needs.plan.outputs.format == 'true' name: clang-format runs-on: ubuntu-24.04 @@ -49,7 +72,8 @@ jobs: | xargs -0 -r clang-format-22 --dry-run --Werror -- linux-x64: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft + needs: plan + if: needs.plan.outputs.build == 'true' name: Ubuntu 24.04 / Clang / x64 runs-on: ubuntu-24.04 container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -79,7 +103,9 @@ jobs: -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset linux-x64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python3 .github/scripts/model_ci.py build --preset linux-x64-release - name: List registered CTest tests run: ctest --test-dir out/build/linux-x64 --build-config Release --show-only=human @@ -93,7 +119,8 @@ jobs: path: out/build/linux-x64/bin/Release/ linux-arm64: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft + needs: plan + if: needs.plan.outputs.build == 'true' name: Ubuntu 24.04 / Clang / ARM64 runs-on: ubuntu-24.04-arm container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -121,7 +148,9 @@ jobs: run: cmake --preset linux-arm64 -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset linux-arm64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python3 .github/scripts/model_ci.py build --preset linux-arm64-release - name: List registered CTest tests run: ctest --test-dir out/build/linux-arm64 --build-config Release --show-only=human @@ -135,7 +164,8 @@ jobs: path: out/build/linux-arm64/bin/Release/ windows-x64: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft + needs: plan + if: needs.plan.outputs.build == 'true' name: Windows Server 2025 / MSVC / x64 runs-on: windows-2025 timeout-minutes: 90 @@ -167,7 +197,9 @@ jobs: -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset windows-x64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python .github/scripts/model_ci.py build --preset windows-x64-release - name: List registered CTest tests run: ctest --test-dir out/build/windows-x64 --build-config Release --show-only=human @@ -179,3 +211,18 @@ jobs: name: windows-x64-release if-no-files-found: error path: out/build/windows-x64/bin/Release/ + + model-tests: + needs: [plan, linux-x64, windows-x64] + if: >- + !cancelled() && needs.plan.outputs.models == 'true' && + needs.linux-x64.result == 'success' && needs.windows-x64.result == 'success' + name: ${{ matrix.variant }} + strategy: + fail-fast: false + matrix: ${{ fromJSON(needs.plan.outputs.matrix) }} + uses: ./.github/workflows/model-test.yml + with: + model: ${{ matrix.model }} + variant: ${{ matrix.variant }} + force_export: ${{ matrix.force_export }} diff --git a/.github/workflows/export-nvidia-asr-model.yml b/.github/workflows/export-nvidia-asr-model.yml deleted file mode 100644 index 7bc2a6c..0000000 --- a/.github/workflows/export-nvidia-asr-model.yml +++ /dev/null @@ -1,297 +0,0 @@ -name: Export NVIDIA ASR model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - model_type: - required: true - type: string - precision: - required: true - type: string - fallback_precision: - required: false - default: '' - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.final-artifact.outputs.name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Compute artifact names - id: artifact-names - shell: bash - env: - MODEL_ID: ${{ inputs.model_id }} - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_SLUG: ${{ inputs.model_slug }} - MODEL_TYPE: ${{ inputs.model_type }} - PRECISION: ${{ inputs.precision }} - run: | - set -euo pipefail - case "$MODEL_TYPE" in - parakeet|nemotron) ;; - *) echo "Unsupported NVIDIA ASR model type: $MODEL_TYPE" >&2; exit 2 ;; - esac - case "$PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported precision: $PRECISION" >&2; exit 2 ;; - esac - if [[ -n "$FALLBACK_PRECISION" ]]; then - case "$FALLBACK_PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; - esac - fi - source_hash="$(python .github/scripts/model_ci.py key --model "${MODEL_TYPE}" --model-id "$MODEL_ID")" - echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" - if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" - fi - - - name: Find successful primary artifact - id: primary-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" \ - --pull-request "$PULL_REQUEST" - - - name: Find successful fallback artifact - if: >- - steps.primary-artifact.outputs.found != 'true' && - inputs.fallback_precision != '' - id: fallback-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" \ - --pull-request "$PULL_REQUEST" - - - name: Select export action - id: selection - shell: bash - env: - FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} - FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} - FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} - PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} - PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$PRIMARY_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" - elif [[ "$FALLBACK_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$FALLBACK_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$FALLBACK_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "export=true" >> "$GITHUB_OUTPUT" - fi - - - name: Reclaim runner disk space - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.selection.outputs.export == 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install NeMo export dependencies - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install --index-url https://download.pytorch.org/whl/cpu torch - python -m pip install \ - onnx \ - onnxscript \ - soundfile \ - "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" - - - name: Export NVIDIA ASR ONNX - if: steps.selection.outputs.export == 'true' - id: exported - shell: bash - env: - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - MODEL_SLUG: ${{ inputs.model_slug }} - MODEL_TYPE: ${{ inputs.model_type }} - PRECISION: ${{ inputs.precision }} - PYTHONPATH: asr/rnnt/python - SOURCE_REF: ${{ inputs.source_ref }} - SOURCE_HASH: ${{ steps.artifact-names.outputs.source_hash }} - run: | - set -uo pipefail - export_dtype() { - case "$1" in - fp16) echo float16 ;; - fp32) echo float32 ;; - *) return 2 ;; - esac - } - run_export() { - local dtype - dtype="$(export_dtype "$1")" - case "$MODEL_TYPE" in - parakeet) - python asr/rnnt/python/tools/export_parakeet_tdt_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$dtype" \ - --external-data - ;; - nemotron) - python asr/rnnt/python/tools/export_nemotron_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$dtype" \ - --external-data - ;; - esac - } - - selected_precision="$PRECISION" - set +e - run_export "$selected_precision" - export_status=$? - set -e - if [[ $export_status -ne 0 ]]; then - if [[ -z "$FALLBACK_PRECISION" ]]; then - exit "$export_status" - fi - echo "::warning::$PRECISION export failed; retrying once with $FALLBACK_PRECISION." - rm -rf "$MODEL_OUTPUT" - selected_precision="$FALLBACK_PRECISION" - run_export "$selected_precision" - fi - - case "$MODEL_TYPE" in - parakeet) - required_files=(preprocessor.onnx encoder.onnx predict_step.onnx metadata.json tokenizer.json) - ;; - nemotron) - required_files=(preprocessor.onnx encoder_step.onnx predict_step.onnx metadata.json vocab.txt decoder_init_hidden.f32 decoder_init_cell.f32) - ;; - esac - for required_file in "${required_files[@]}"; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SELECTED_PRECISION="$selected_precision" SOURCE_REF - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": os.environ["SELECTED_PRECISION"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}-${SOURCE_HASH}" >> "$GITHUB_OUTPUT" - - - name: Upload NVIDIA ASR ONNX - if: steps.selection.outputs.export == 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.exported.outputs.name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - EXISTING_NAME: ${{ steps.selection.outputs.name }} - EXISTING_RUN_ID: ${{ steps.selection.outputs.run_id }} - EXPORTED: ${{ steps.selection.outputs.export }} - EXPORTED_NAME: ${{ steps.exported.outputs.name }} - run: | - set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then - echo "name=$EXPORTED_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "name=$EXISTING_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/export-sam2-model.yml b/.github/workflows/export-sam2-model.yml deleted file mode 100644 index c2df020..0000000 --- a/.github/workflows/export-sam2-model.yml +++ /dev/null @@ -1,181 +0,0 @@ -name: Export SAM2 model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.model-key.outputs.artifact_name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Compute model artifact key - id: model-key - shell: bash - env: - MODEL_SLUG: ${{ inputs.model_slug }} - MODEL_ID: ${{ inputs.model_id }} - run: | - set -euo pipefail - source_hash="$(python .github/scripts/model_ci.py key --model sam2 --model-id "$MODEL_ID")" - echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" - echo "artifact_name=vision-${MODEL_SLUG}-onnx-v1-${source_hash:0:16}" >> "$GITHUB_OUTPUT" - - - name: Resolve reusable model artifact - id: existing-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - MODEL_ARTIFACT: ${{ steps.model-key.outputs.artifact_name }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "$MODEL_ARTIFACT" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" \ - --pull-request "$PULL_REQUEST" - - - name: Reclaim runner disk space - if: steps.existing-artifact.outputs.found != 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.existing-artifact.outputs.found != 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install SAM2 export dependencies - if: steps.existing-artifact.outputs.found != 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install \ - --index-url https://download.pytorch.org/whl/cpu \ - torch \ - torchvision - python -m pip install \ - huggingface_hub \ - numpy \ - onnx \ - onnxscript \ - "sam-2 @ git+https://github.com/facebookresearch/sam2.git" - - - name: Export SAM2 ONNX - if: steps.existing-artifact.outputs.found != 'true' - shell: bash - env: - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - SOURCE_HASH: ${{ steps.model-key.outputs.source_hash }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euxo pipefail - python vision/sam2/python/export_sam2_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype float16 \ - --max-points 3 - - required_files=( - image_preprocess.onnx - image_encoder.onnx - sam2_decoder.onnx - sam2_decoder_propagate.onnx - memory_attention.onnx - memory_encoder.onnx - mask_postprocess.onnx - metadata.json - ) - for required_file in "${required_files[@]}"; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SOURCE_HASH - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": "float16", - "source_hash": os.environ["SOURCE_HASH"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - - - name: Upload SAM2 ONNX - if: steps.existing-artifact.outputs.found != 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.model-key.outputs.artifact_name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - FOUND: ${{ steps.existing-artifact.outputs.found }} - EXISTING_RUN_ID: ${{ steps.existing-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$FOUND" != "true" ]]; then - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/export-whisper-model.yml b/.github/workflows/export-whisper-model.yml deleted file mode 100644 index 8f5ddf4..0000000 --- a/.github/workflows/export-whisper-model.yml +++ /dev/null @@ -1,260 +0,0 @@ -name: Export Whisper model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - precision: - required: true - type: string - fallback_precision: - required: false - default: '' - type: string - attention: - required: false - default: sdpa - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.final-artifact.outputs.name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Compute artifact names - id: artifact-names - shell: bash - env: - MODEL_ID: ${{ inputs.model_id }} - ATTENTION: ${{ inputs.attention }} - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_SLUG: ${{ inputs.model_slug }} - PRECISION: ${{ inputs.precision }} - run: | - set -euo pipefail - case "$PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported precision: $PRECISION" >&2; exit 2 ;; - esac - if [[ -n "$FALLBACK_PRECISION" ]]; then - case "$FALLBACK_PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; - esac - fi - source_hash="$(python .github/scripts/model_ci.py key --model "whisper" --model-id "$MODEL_ID" --attention "$ATTENTION")" - echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" - if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}-${source_hash}" >> "$GITHUB_OUTPUT" - fi - - - name: Find successful primary artifact - id: primary-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" \ - --pull-request "$PULL_REQUEST" - - - name: Find successful fallback artifact - if: >- - steps.primary-artifact.outputs.found != 'true' && - inputs.fallback_precision != '' - id: fallback-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" \ - --pull-request "$PULL_REQUEST" - - - name: Select export action - id: selection - shell: bash - env: - FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} - FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} - FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} - PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} - PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$PRIMARY_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" - elif [[ "$FALLBACK_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$FALLBACK_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$FALLBACK_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "export=true" >> "$GITHUB_OUTPUT" - fi - - - name: Reclaim runner disk space - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.selection.outputs.export == 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install Whisper export dependencies - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install --index-url https://download.pytorch.org/whl/cpu torch - python -m pip install onnx onnxscript safetensors transformers - - - name: Export Whisper ONNX - if: steps.selection.outputs.export == 'true' - id: exported - shell: bash - env: - ATTENTION: ${{ inputs.attention }} - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - MODEL_SLUG: ${{ inputs.model_slug }} - PRECISION: ${{ inputs.precision }} - SOURCE_REF: ${{ inputs.source_ref }} - SOURCE_HASH: ${{ steps.artifact-names.outputs.source_hash }} - run: | - set -uo pipefail - run_export() { - python asr/whisper/model_export/export_whisper.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$1" \ - --attention "$ATTENTION" - } - - selected_precision="$PRECISION" - set +e - run_export "$selected_precision" - export_status=$? - set -e - if [[ $export_status -ne 0 ]]; then - if [[ -z "$FALLBACK_PRECISION" ]]; then - exit "$export_status" - fi - echo "::warning::$PRECISION export failed; retrying once with $FALLBACK_PRECISION." - rm -rf "$MODEL_OUTPUT" - selected_precision="$FALLBACK_PRECISION" - run_export "$selected_precision" - fi - - for required_file in encoder.onnx decoder.onnx mel.onnx vocab.json; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SELECTED_PRECISION="$selected_precision" SOURCE_REF - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": os.environ["SELECTED_PRECISION"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}-${SOURCE_HASH}" >> "$GITHUB_OUTPUT" - - - name: Upload Whisper ONNX - if: steps.selection.outputs.export == 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.exported.outputs.name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - EXISTING_NAME: ${{ steps.selection.outputs.name }} - EXISTING_RUN_ID: ${{ steps.selection.outputs.run_id }} - EXPORTED: ${{ steps.selection.outputs.export }} - EXPORTED_NAME: ${{ steps.exported.outputs.name }} - run: | - set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then - echo "name=$EXPORTED_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "name=$EXISTING_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/model-changes.yml b/.github/workflows/model-changes.yml deleted file mode 100644 index d945b3e..0000000 --- a/.github/workflows/model-changes.yml +++ /dev/null @@ -1,32 +0,0 @@ -name: Detect model changes - -on: - workflow_call: - outputs: - whisper: - value: ${{ jobs.changes.outputs.whisper }} - parakeet: - value: ${{ jobs.changes.outputs.parakeet }} - nemotron: - value: ${{ jobs.changes.outputs.nemotron }} - sam2: - value: ${{ jobs.changes.outputs.sam2 }} - -permissions: - contents: read - -jobs: - changes: - runs-on: ubuntu-24.04 - outputs: - whisper: ${{ steps.changes.outputs.whisper }} - parakeet: ${{ steps.changes.outputs.parakeet }} - nemotron: ${{ steps.changes.outputs.nemotron }} - sam2: ${{ steps.changes.outputs.sam2 }} - steps: - - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ github.event.pull_request.head.sha || github.sha }} - - id: changes - run: python .github/scripts/model_ci.py changes diff --git a/.github/workflows/model-test.yml b/.github/workflows/model-test.yml new file mode 100644 index 0000000..85f2847 --- /dev/null +++ b/.github/workflows/model-test.yml @@ -0,0 +1,109 @@ +name: Model export and inference + +on: + workflow_call: + inputs: + model: + required: true + type: string + variant: + required: true + type: string + force_export: + required: false + default: false + type: boolean + +permissions: + contents: read + actions: read + +jobs: + export: + runs-on: ubuntu-24.04 + timeout-minutes: 240 + outputs: + name: ${{ steps.resolve.outputs.name || steps.export.outputs.name }} + run_id: ${{ steps.resolve.outputs.run_id || steps.export.outputs.run_id }} + env: + MODEL: ${{ inputs.model }} + VARIANT: ${{ inputs.variant }} + FORCE_EXPORT: ${{ inputs.force_export }} + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-python@v7 + with: + python-version: '3.12' + - name: Find a successful matching export + id: resolve + env: + GH_TOKEN: ${{ github.token }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} + SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} + run: python .github/scripts/run_model_ci.py resolve + - name: Reclaim export disk space + if: steps.resolve.outputs.found != 'true' + run: sudo rm -rf /opt/ghc /usr/local/.ghcup /usr/local/lib/android /usr/share/dotnet + - name: Install dependencies and export model + if: steps.resolve.outputs.found != 'true' + id: export + run: python .github/scripts/run_model_ci.py export + - uses: actions/upload-artifact@v7 + if: steps.resolve.outputs.found != 'true' + with: + name: ${{ steps.export.outputs.name }} + path: artifacts/ci/${{ inputs.variant }}/ + if-no-files-found: error + compression-level: 0 + + smoke: + needs: export + name: ${{ matrix.platform }} / CPU + runs-on: ${{ matrix.runner }} + container: ${{ matrix.container }} + timeout-minutes: 120 + strategy: + fail-fast: false + matrix: + include: + - platform: linux-x64 + runner: ubuntu-24.04 + container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 + - platform: windows-x64 + runner: windows-2025 + container: '' + env: + MODEL: ${{ inputs.model }} + VARIANT: ${{ inputs.variant }} + steps: + - name: Install Python in Linux runtime container + if: runner.os == 'Linux' + run: apt-get update && apt-get install -y --no-install-recommends python3 + - uses: actions/checkout@v7 + - uses: actions/setup-python@v7 + if: runner.os == 'Windows' + with: + python-version: '3.12' + - name: Download native build from this run + uses: actions/download-artifact@v8 + with: + name: ${{ matrix.platform }}-release + path: runtime + - name: Download matching model export + uses: actions/download-artifact@v8 + with: + name: ${{ needs.export.outputs.name }} + path: artifacts/ci/${{ inputs.variant }}/ + github-token: ${{ github.token }} + repository: ${{ github.repository }} + run-id: ${{ needs.export.outputs.run_id }} + - name: Run model inference + shell: bash + run: | + if [ "$RUNNER_OS" = Linux ]; then + python3 .github/scripts/run_model_ci.py smoke + else + python .github/scripts/run_model_ci.py smoke + fi diff --git a/.github/workflows/nvidia-asr.yml b/.github/workflows/nvidia-asr.yml deleted file mode 100644 index 501f3d0..0000000 --- a/.github/workflows/nvidia-asr.yml +++ /dev/null @@ -1,233 +0,0 @@ -name: NVIDIA ASR - -on: - push: - branches: [main] - pull_request: - types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: nvidia-asr-${{ github.ref }} - cancel-in-progress: true - -jobs: - changes: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft - uses: ./.github/workflows/model-changes.yml - - requirements: - needs: changes - if: needs.changes.outputs.parakeet == 'true' || needs.changes.outputs.nemotron == 'true' - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export-parakeet: - if: needs.changes.outputs.parakeet == 'true' - name: Export Parakeet TDT ONNX - needs: [requirements, changes] - uses: ./.github/workflows/export-nvidia-asr-model.yml - with: - model_slug: parakeet-tdt - model_label: Parakeet TDT - model_id: nvidia/parakeet-tdt-0.6b-v3 - model_type: parakeet - precision: fp16 - fallback_precision: fp32 - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-nemotron: - if: needs.changes.outputs.nemotron == 'true' - name: Export Nemotron 3.5 ASR Streaming ONNX - needs: [requirements, changes] - uses: ./.github/workflows/export-nvidia-asr-model.yml - with: - model_slug: nemotron-3.5-asr-streaming-0.6b - model_label: Nemotron 3.5 ASR Streaming 0.6B - model_id: nvidia/nemotron-3.5-asr-streaming-0.6b - model_type: nemotron - precision: fp16 - fallback_precision: fp32 - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Collect NVIDIA ASR artifacts - needs: [requirements, export-parakeet, export-nemotron] - if: >- - !cancelled() && !failure() && - (needs.export-parakeet.result == 'success' || needs.export-nemotron.result == 'success') - runs-on: ubuntu-24.04 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - - steps: - - name: Build runtime matrix - id: matrix - shell: bash - env: - NEMOTRON_NAME: ${{ needs.export-nemotron.outputs.artifact_name }} - NEMOTRON_RUN_ID: ${{ needs.export-nemotron.outputs.artifact_run_id }} - PARAKEET_NAME: ${{ needs.export-parakeet.outputs.artifact_name }} - PARAKEET_RUN_ID: ${{ needs.export-parakeet.outputs.artifact_run_id }} - run: | - python - <<'PY' - import json - import os - from pathlib import Path - - matrix = { - "include": [ - { - "slug": "parakeet-tdt", - "label": "Parakeet TDT", - "executable": "din_asr_parakeet_tdt_cli", - "artifact_name": os.environ["PARAKEET_NAME"], - "artifact_run_id": os.environ["PARAKEET_RUN_ID"], - }, - { - "slug": "nemotron-3.5-asr-streaming-0.6b", - "label": "Nemotron 3.5 ASR Streaming 0.6B", - "executable": "din_asr_nemotron_cli", - "artifact_name": os.environ["NEMOTRON_NAME"], - "artifact_run_id": os.environ["NEMOTRON_RUN_ID"], - }, - ] - } - matrix["include"] = [model for model in matrix["include"] if model["artifact_name"]] - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - output.write(f"matrix={json.dumps(matrix, separators=(',', ':'))}\n") - PY - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - chmod +x "runtime/${{ matrix.executable }}" - "runtime/${{ matrix.executable }}" \ - assets/sample.wav \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - & "runtime/${{ matrix.executable }}.exe" ` - "assets/sample.wav" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/.github/workflows/sam2.yml b/.github/workflows/sam2.yml deleted file mode 100644 index 3149e44..0000000 --- a/.github/workflows/sam2.yml +++ /dev/null @@ -1,263 +0,0 @@ -name: SAM2 - -on: - push: - branches: [main] - pull_request: - types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: sam2-${{ github.ref }} - cancel-in-progress: true - -jobs: - changes: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft - uses: ./.github/workflows/model-changes.yml - - requirements: - needs: changes - if: needs.changes.outputs.sam2 == 'true' - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export: - name: Export ${{ matrix.label }} ONNX - needs: requirements - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - model_id: facebook/sam2.1-hiera-tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - model_id: facebook/sam2.1-hiera-small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - model_id: facebook/sam2.1-hiera-base-plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - model_id: facebook/sam2.1-hiera-large - uses: ./.github/workflows/export-sam2-model.yml - with: - model_slug: ${{ matrix.slug }} - model_label: ${{ matrix.label }} - model_id: ${{ matrix.model_id }} - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Resolve SAM2 artifacts - needs: [requirements, export] - runs-on: ubuntu-24.04 - outputs: - models: ${{ steps.resolve.outputs.models }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Resolve model artifacts - id: resolve - shell: bash - env: - GH_TOKEN: ${{ github.token }} - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - PULL_REQUEST: ${{ github.event.pull_request.number }} - run: | - set -euo pipefail - models='{}' - for slug in \ - sam2.1-hiera-tiny \ - sam2.1-hiera-small \ - sam2.1-hiera-base-plus \ - sam2.1-hiera-large; do - source_hash="$(python .github/scripts/model_ci.py key --model sam2 --model-id "facebook/${slug}")" - artifact_name="vision-${slug}-onnx-v1-${source_hash:0:16}" - artifact_run_id="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/artifacts?per_page=100" \ - --jq ".artifacts[] | select(.name == \"$artifact_name\" and (.expired | not)) | \"$GITHUB_RUN_ID\"")" - if [[ -z "$artifact_run_id" ]]; then - artifact="$(python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" --artifact-name "$artifact_name" \ - --default-branch "$DEFAULT_BRANCH" --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" --pull-request "$PULL_REQUEST" --json)" - artifact_run_id="$(jq -r '.run_id' <<< "$artifact")" - fi - test -n "$artifact_run_id" - models="$(jq \ - --arg slug "$slug" \ - --arg name "$artifact_name" \ - --arg run_id "$artifact_run_id" \ - '. + {($slug): {name: $name, run_id: $run_id}}' \ - <<< "$models")" - done - echo "models=$(jq -c . <<< "$models")" >> "$GITHUB_OUTPUT" - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - base64 --decode assets/sam2-ci.png.base64 > assets/sam2-ci.png - mkdir -p "out/${{ matrix.slug }}" - chmod +x runtime/din_sam2_cli - runtime/din_sam2_cli \ - assets/sam2-ci.png \ - assets/sam2-ci-prompt.json \ - "out/${{ matrix.slug }}" \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - $imageBytes = [Convert]::FromBase64String( - (Get-Content -LiteralPath "assets/sam2-ci.png.base64" -Raw).Trim() - ) - [IO.File]::WriteAllBytes("assets/sam2-ci.png", $imageBytes) - New-Item -ItemType Directory -Force -Path "out/${{ matrix.slug }}" | Out-Null - & "runtime/din_sam2_cli.exe" ` - "assets/sam2-ci.png" ` - "assets/sam2-ci-prompt.json" ` - "out/${{ matrix.slug }}" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/.github/workflows/whisper-asr.yml b/.github/workflows/whisper-asr.yml deleted file mode 100644 index 3fdd8de..0000000 --- a/.github/workflows/whisper-asr.yml +++ /dev/null @@ -1,292 +0,0 @@ -name: Whisper ASR - -on: - push: - branches: [main] - pull_request: - types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: whisper-asr-${{ github.ref }} - cancel-in-progress: true - -jobs: - changes: - if: github.event_name != 'pull_request' || !github.event.pull_request.draft - uses: ./.github/workflows/model-changes.yml - - requirements: - needs: changes - if: needs.changes.outputs.whisper == 'true' - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\" and .conclusion != \"skipped\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export-tiny: - name: Export Whisper Tiny ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-tiny - model_label: Whisper Tiny - model_id: openai/whisper-tiny - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-base: - name: Export Whisper Base ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-base - model_label: Whisper Base - model_id: openai/whisper-base - precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-small: - name: Export Whisper Small ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-small - model_label: Whisper Small - model_id: openai/whisper-small - precision: fp16 - fallback_precision: fp32 - attention: math - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-medium: - name: Export Whisper Medium ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-medium - model_label: Whisper Medium - model_id: openai/whisper-medium - precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-large-v3: - name: Export Whisper Large v3 ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-large-v3 - model_label: Whisper Large v3 - model_id: openai/whisper-large-v3 - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-large-v3-turbo: - name: Export Whisper Large v3 Turbo ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-large-v3-turbo - model_label: Whisper Large v3 Turbo - model_id: openai/whisper-large-v3-turbo - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Collect Whisper artifacts - needs: - - export-tiny - - export-base - - export-small - - export-medium - - export-large-v3 - - export-large-v3-turbo - runs-on: ubuntu-24.04 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - - steps: - - name: Build runtime matrix - id: matrix - shell: bash - env: - BASE_NAME: ${{ needs.export-base.outputs.artifact_name }} - BASE_RUN_ID: ${{ needs.export-base.outputs.artifact_run_id }} - LARGE_NAME: ${{ needs.export-large-v3.outputs.artifact_name }} - LARGE_RUN_ID: ${{ needs.export-large-v3.outputs.artifact_run_id }} - MEDIUM_NAME: ${{ needs.export-medium.outputs.artifact_name }} - MEDIUM_RUN_ID: ${{ needs.export-medium.outputs.artifact_run_id }} - SMALL_NAME: ${{ needs.export-small.outputs.artifact_name }} - SMALL_RUN_ID: ${{ needs.export-small.outputs.artifact_run_id }} - TINY_NAME: ${{ needs.export-tiny.outputs.artifact_name }} - TINY_RUN_ID: ${{ needs.export-tiny.outputs.artifact_run_id }} - TURBO_NAME: ${{ needs.export-large-v3-turbo.outputs.artifact_name }} - TURBO_RUN_ID: ${{ needs.export-large-v3-turbo.outputs.artifact_run_id }} - run: | - python - <<'PY' - import json - import os - from pathlib import Path - - models = [ - ("whisper-tiny", "Whisper Tiny", "TINY"), - ("whisper-base", "Whisper Base", "BASE"), - ("whisper-small", "Whisper Small", "SMALL"), - ("whisper-medium", "Whisper Medium", "MEDIUM"), - ("whisper-large-v3", "Whisper Large v3", "LARGE"), - ("whisper-large-v3-turbo", "Whisper Large v3 Turbo", "TURBO"), - ] - matrix = { - "include": [ - { - "slug": slug, - "label": label, - "artifact_name": os.environ[f"{key}_NAME"], - "artifact_run_id": os.environ[f"{key}_RUN_ID"], - } - for slug, label, key in models - ] - } - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - output.write(f"matrix={json.dumps(matrix, separators=(',', ':'))}\n") - PY - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - chmod +x runtime/din_asr_whisper_cli - runtime/din_asr_whisper_cli \ - assets/sample.wav \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - & "runtime/din_asr_whisper_cli.exe" ` - "assets/sample.wav" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/REVIEW_GUIDELINES.md b/REVIEW_GUIDELINES.md new file mode 100644 index 0000000..f630acb --- /dev/null +++ b/REVIEW_GUIDELINES.md @@ -0,0 +1,26 @@ +# Review guidelines + +DIN Deploy provides practical examples for exporting models to ONNX and running +local inference on client PCs. Prioritize straightforward setup, understandable +code, inference correctness, and useful performance. + +- Keep model-specific exporters and runtime code together. Put shared + functionality in `common/` when multiple samples need it. +- Preserve supported Windows and Linux builds, including supported ARM64 + configurations. Preserve CPU execution where supported while optimizing CUDA + and TensorRT RTX paths. +- Keep exporter outputs and runtime expectations consistent: filenames, tensor + shapes, dtypes, preprocessing, tokenizers, and metadata. +- Prefer focused changes and simple implementations. Suggest abstractions when + they solve a concrete maintenance problem. +- Keep CI efficient without silently skipping relevant tests. Missing dependency + declarations must broaden execution. Export-cache inputs must cover the + sources and settings that affect the model. See `docs/model-ci.md`. +- Update model documentation when setup, commands, dependencies, or supported + behavior changes. +- Prioritize demonstrable bugs, compatibility regressions, incorrect outputs, + and significant performance problems. Explain each finding's trigger and + impact. Avoid speculative warnings, cosmetic preferences, and requests to + rewrite unrelated code. +- Treat these principles as guidance; identify tradeoffs rather than enforcing + blanket rules. Follow the contributor sign-off requirements in CONTRIBUTING.md. diff --git a/docs/model-ci.md b/docs/model-ci.md new file mode 100644 index 0000000..3dc2749 --- /dev/null +++ b/docs/model-ci.md @@ -0,0 +1,106 @@ +# Model CI + +CI has one planner and one reusable export/inference workflow. Model-specific +commands are data; adding a model does not require editing workflow logic. + +## Two kinds of configuration + +- `.github/model-tests/*.json` describes **how** to export and test a model family: + CMake targets, model variants, dependency installation, exporter command, + required output files, and CPU smoke-test command. Every recipe is discovered + automatically. The first CMake target is the executable used for inference. +- `.github/model-ci.json` optionally describes **when it is safe to skip** work: + export inputs, runtime/test inputs, reusable dependency groups, and explicitly + ignored files. A recipe without complete dependency rules runs conservatively. + +The execution recipe is necessary: CI cannot infer model IDs, CLI arguments, +credentials, or hardware requirements from a new C++ executable. Dependency +optimization is optional. + +## Behavior + +| Change | Native build | Export | Inference | +| --- | --- | --- | --- | +| Model exporter or its declared dependencies | Model targets | New fingerprint; reuse only an exact match | Affected model variants | +| Model runtime | Model targets | Reuse matching export, or create one on cache miss | Affected model variants | +| Shared runtime | All consuming targets | Reuse matching exports | All consuming models | +| Test asset | Consuming targets | Reuse matching exports | Consuming models | +| README or `.coderabbit.yaml` only | Skip | Skip | Skip | +| Unknown file | Full build | Conservative fresh exports | All available recipes | +| Missing model dependency rules | That model always runs on non-exempt changes | Fresh export if export rules are missing | That model | +| CI control logic or dependency registry | Full build | Fingerprint determines reuse | All models | +| Manual run or unavailable Git base | Full build | Fresh exports | All models | + +Draft PRs skip the pipeline. Marking a PR ready starts it; returning it to draft +cancels an in-progress run through workflow concurrency. Lightweight planner tests +run on every non-draft PR, including documentation changes. C++ formatting runs +when native source or `.clang-format` changes. + +All native platforms use the same selected CMake targets. CMake/Ninja resolves +their actual compilation and linking dependencies. Linux x64 and Windows x64 +smoke tests consume build artifacts from **the same workflow run and checkout**; +there is no polling for a separate Build run. ARM64 keeps its build coverage. +No inference-test success or native binary cache is introduced here. + +## Conservative fallback + +Unowned changed files trigger all known recipes and a full build. Literal CMake +executable declarations outside the root helper file are independently scanned: +an executable absent from the recipes/build-only declarations forces a full +build and all available smoke tests on non-exempt changes. The job summary lists +those executables and calls out the missing inference recipe. This is deliberately +not a general-purpose CMake parser; unusual dynamically declared targets should +get a recipe or an explicit build-only declaration. + +A new recipe with no entry in the dependency registry is always selected. It does +not reuse an earlier run's export. Missing or invalid dependency configuration +falls back to all recipes. Invalid **execution** recipes fail visibly: continuing +without knowing how to execute the test would hide missing coverage. + +FLUX and the base ONNX sample retain their existing build-only hosted coverage, +with explicit reasons in the registry. GPU/local-asset CTest tests are still listed, +not executed on CPU hosted runners. A full-build fallback cannot manufacture the +hardware or model artifacts needed to run those tests. + +## Export reuse + +Export names include a fingerprint of tracked export input blob IDs, export +recipe, model variant/options, and shared exporter runner/workflow. Runtime-only +changes do not invalidate an export. Unowned tracked files are also fingerprinted, +so their changes cannot accidentally restore an old export in a later run. +Unknown dependencies use a run/attempt-specific key to prevent cross-run reuse. + +Only unexpired artifacts from successful allowed runs (default branch, same PR, +or allowed non-PR source branch) are reused. A missing/expired artifact produces a +new export. Precision fallback is preserved: try the requested precision, then +its configured fallback if export fails. Smoke tests still have to pass. + +The existing dependency installation recipes use some floating package/model +versions. Cached exports represent the versions downloaded when they were made, +not a continuous check for upstream updates. Change `export.cache_epoch` to force +refreshes; pin dependency/model revisions in recipes when reproducibility requires +it. A complete package lock and model revision pinning are separate follow-up work. + +## Adding a model + +1. Add its normal CMake executable target and exporter. +2. Add `.github/model-tests/.json`, using an existing recipe as an example. + Commands are argument arrays, executed without a shell. Available placeholders + include `{python}`, `{model_id}`, `{output}`, `{precision}`, `{dtype}`, `{slug}`, + and variant fields. Smoke tests additionally receive `{executable}`. +3. Run `python -m unittest discover -s .github/tests -v`. The new recipe is now + discoverable and runs without dependency optimization. +4. Optionally add `models..export` and `.runtime` input patterns to + `.github/model-ci.json`. Reference shared groups rather than copying them. + Include transitive Python helpers, preprocessing, tokenizer code, build + configuration, and test assets. These declarations are correctness contracts. +5. Inspect the CI selection summary and confirm the new variant passes on both + runtime platforms before relying on selective execution. + +Patterns use Python `fnmatch`: `*` matches across directory separators. Renames +are treated as a deletion plus an addition. Unknown ownership always broadens +work. Add ignore patterns only for files known not to affect builds or tests. + +The first run after this migration creates new export artifacts because the +fingerprint format changed. Existing required-check rules referring to the old +standalone model workflows must be updated for the consolidated workflow. From dd3b41c7ae1ca144b6c210a6e0eced32434b5ff8 Mon Sep 17 00:00:00 2001 From: lspindler Date: Thu, 24 Sep 2026 09:41:26 +0200 Subject: [PATCH 4/8] Add repository agent guidance and disable docstring coverage checks Signed-off-by: lspindler --- .coderabbit.yaml | 5 +++ AGENTS.md | 83 ++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 88 insertions(+) create mode 100644 AGENTS.md diff --git a/.coderabbit.yaml b/.coderabbit.yaml index d68b9e5..5e67233 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -6,6 +6,10 @@ reviews: request_changes_workflow: false fail_commit_status: false + pre_merge_checks: + docstrings: + mode: "off" + high_level_summary: true high_level_summary_in_walkthrough: true collapse_walkthrough: true @@ -26,4 +30,5 @@ knowledge_base: code_guidelines: enabled: true filePatterns: + - "AGENTS.md" - "REVIEW_GUIDELINES.md" diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..b1f650b --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,83 @@ +# DIN Deploy: repository guidance for coding agents + +## Goal + +Build practical, high-performance local inference samples that others can understand, +reuse, and integrate into commercial applications. Keep this a sample repository, +not a framework: as small as possible and as complex as necessary, without sacrificing +correctness, functionality, or performance. + +Apply these rules to new and changed code; do not rewrite unrelated samples to enforce them. + +## Portable models and execution providers + +- Export standard ONNX operators. Do not introduce ONNX Runtime contrib operators + (for example `com.microsoft::*`), ATen fallbacks, custom operator libraries, or + vendor-specific graph operators. Use standard ONNX decompositions instead. +- Inspect exported graphs, including subgraphs and local functions, for nonstandard + operators; ONNX checker success alone does not demonstrate EP compatibility. +- Keep graphs and core algorithms EP-independent; EP-internal fusion is allowed. + Maintain a usable CPU path and isolate optional backend-specific optimizations. +- Target Windows and Linux on x64 and ARM64. Hardware acceleration depends on the + device and driver, not just the OS/CPU architecture; identify those requirements. +- Validate supported EPs and precisions explicitly; standard ONNX does not guarantee + kernel support. Do not silently move expensive operations to CPU or claim that + accelerator-only precisions work there. Provide a compatible configuration or + explain the limitation. + +## Performance and memory ownership + +- Optimize measured bottlenecks in latency, throughput, memory, and startup/compilation. +- Keep intermediate tensors on their device through preprocessing and inference. + Use device-buffer binding and reusable allocations where supported. Avoid host + round trips and per-inference allocation in hot paths. +- Make memory location, ownership, layout/stride, and lifetime explicit at API + boundaries. Keep buffers alive until asynchronous consumers finish. Bound queues + and caches. +- Reuse the application's CUDA context for cooperating components where supported. + Do not create extra contexts or processes casually; document any required boundary. +- Prefer stream/event dependencies over CPU waits and device-wide synchronization. + Preserve producer/consumer dependencies when removing waits. +- GPU-resident is not synonymous with zero-copy or wait-free. Account for device + copies and preprocessing when evaluating performance. + +## Minimal code and useful reuse + +- Reuse existing export, runtime, and I/O helpers. Extract small shared functions + for concrete repetition; keep model-specific behavior near its model. Avoid + speculative abstractions, registries, and frameworks for hypothetical reuse. +- Simplify control flow and remove obsolete code. Do not shorten code at the expense + of readability, buffer reuse, or asynchronous execution. +- Follow surrounding conventions and repository formatting, including `.clang-format`. +- Preserve error causes; do not present failures or unsupported settings as success. + +## Licensing and commercial reuse + +- Keep code contributions compatible with this repository's Apache-2.0 license and + commercial reuse. Prefer permissive dependencies with clear redistribution terms. +- Review licenses for new code, transitive dependencies, model weights, and bundled + assets. Do not assume that open source or public availability permits every use. +- Discuss downstream impact with the user before introducing non-commercial + restrictions or copyleft obligations. Flag unclear terms; do not claim unverified + commercial compatibility or assume model weights share the code's license. +- Preserve required attribution and license notices, and document redistribution + obligations. Consider relevant patent and SDK terms separately from code licenses. + +## Validation and working practices + +- Read the relevant README, build configuration, and CONTRIBUTING.md. Preserve + unrelated user/agent changes. Use the existing project environment and build paths. +- Run focused validation: export parity and graph checks for model changes, + numerical/task-quality checks for inference changes, and regression checks for + state, timing, cancellation, and ownership bugs. +- Benchmark performance changes before/after with matching hardware, EP, precision, + input, and warmup. Measure completed GPU work, not just submission time. Inspect + copies/waits when claiming to reduce them; a smoke test is not performance evidence. +- Separate cold startup from steady-state timings. End-to-end timings include all + enabled steps. Label model-only timings separately and state the RTF convention. +- Build/test the affected available targets. Report untested platform/EP combinations + and unavailable prerequisites honestly. Do not add redundant tests or expand + testing without a concrete reason. +- Update nearby documentation when behavior, requirements, or commands change. + Report what changed, what was verified, and material limitations concisely. +- Commit/push only when requested; follow CONTRIBUTING.md's sign-off requirement. From dec05204ec1fc9a5f3bf3f67d839ffa5400f1012 Mon Sep 17 00:00:00 2001 From: lspindler Date: Thu, 24 Sep 2026 09:42:59 +0200 Subject: [PATCH 5/8] Use AGENTS.md as the sole CodeRabbit guidance file Signed-off-by: lspindler --- .coderabbit.yaml | 1 - REVIEW_GUIDELINES.md | 26 -------------------------- 2 files changed, 27 deletions(-) delete mode 100644 REVIEW_GUIDELINES.md diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 5e67233..66a4727 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -31,4 +31,3 @@ knowledge_base: enabled: true filePatterns: - "AGENTS.md" - - "REVIEW_GUIDELINES.md" diff --git a/REVIEW_GUIDELINES.md b/REVIEW_GUIDELINES.md deleted file mode 100644 index f630acb..0000000 --- a/REVIEW_GUIDELINES.md +++ /dev/null @@ -1,26 +0,0 @@ -# Review guidelines - -DIN Deploy provides practical examples for exporting models to ONNX and running -local inference on client PCs. Prioritize straightforward setup, understandable -code, inference correctness, and useful performance. - -- Keep model-specific exporters and runtime code together. Put shared - functionality in `common/` when multiple samples need it. -- Preserve supported Windows and Linux builds, including supported ARM64 - configurations. Preserve CPU execution where supported while optimizing CUDA - and TensorRT RTX paths. -- Keep exporter outputs and runtime expectations consistent: filenames, tensor - shapes, dtypes, preprocessing, tokenizers, and metadata. -- Prefer focused changes and simple implementations. Suggest abstractions when - they solve a concrete maintenance problem. -- Keep CI efficient without silently skipping relevant tests. Missing dependency - declarations must broaden execution. Export-cache inputs must cover the - sources and settings that affect the model. See `docs/model-ci.md`. -- Update model documentation when setup, commands, dependencies, or supported - behavior changes. -- Prioritize demonstrable bugs, compatibility regressions, incorrect outputs, - and significant performance problems. Explain each finding's trigger and - impact. Avoid speculative warnings, cosmetic preferences, and requests to - rewrite unrelated code. -- Treat these principles as guidance; identify tradeoffs rather than enforcing - blanket rules. Follow the contributor sign-off requirements in CONTRIBUTING.md. From 31f686f8f5141e57b49304fcbb390bb4968db92a Mon Sep 17 00:00:00 2001 From: lspindler Date: Thu, 24 Sep 2026 09:46:14 +0200 Subject: [PATCH 6/8] Condense model CI documentation Signed-off-by: lspindler --- docs/model-ci.md | 129 ++++++++++++----------------------------------- 1 file changed, 32 insertions(+), 97 deletions(-) diff --git a/docs/model-ci.md b/docs/model-ci.md index 3dc2749..057eaa3 100644 --- a/docs/model-ci.md +++ b/docs/model-ci.md @@ -1,106 +1,41 @@ # Model CI -CI has one planner and one reusable export/inference workflow. Model-specific -commands are data; adding a model does not require editing workflow logic. +Model recipes in `.github/model-tests/*.json` define exports, CMake targets, and +CPU smoke tests. Optional dependency rules in `.github/model-ci.json` determine +which models need to run. Adding a model requires no workflow changes. -## Two kinds of configuration +## What runs -- `.github/model-tests/*.json` describes **how** to export and test a model family: - CMake targets, model variants, dependency installation, exporter command, - required output files, and CPU smoke-test command. Every recipe is discovered - automatically. The first CMake target is the executable used for inference. -- `.github/model-ci.json` optionally describes **when it is safe to skip** work: - export inputs, runtime/test inputs, reusable dependency groups, and explicitly - ignored files. A recipe without complete dependency rules runs conservatively. +| Change | CI behavior | +| --- | --- | +| Exporter or export dependencies | Build and test affected models; reuse only matching exports | +| Runtime or test inputs | Build and test affected models using cached exports | +| Shared code | Run all affected consumers | +| Documentation or `.coderabbit.yaml` only | Lightweight checks only | +| Unknown files or CI control logic | Full build and all model recipes | -The execution recipe is necessary: CI cannot infer model IDs, CLI arguments, -credentials, or hardware requirements from a new C++ executable. Dependency -optimization is optional. +Draft PRs skip CI. Missing dependency rules make a model run conservatively; +missing export rules prevent reuse across runs. Missing or expired artifacts are +re-exported. Unknown files also force fresh exports. -## Behavior - -| Change | Native build | Export | Inference | -| --- | --- | --- | --- | -| Model exporter or its declared dependencies | Model targets | New fingerprint; reuse only an exact match | Affected model variants | -| Model runtime | Model targets | Reuse matching export, or create one on cache miss | Affected model variants | -| Shared runtime | All consuming targets | Reuse matching exports | All consuming models | -| Test asset | Consuming targets | Reuse matching exports | Consuming models | -| README or `.coderabbit.yaml` only | Skip | Skip | Skip | -| Unknown file | Full build | Conservative fresh exports | All available recipes | -| Missing model dependency rules | That model always runs on non-exempt changes | Fresh export if export rules are missing | That model | -| CI control logic or dependency registry | Full build | Fingerprint determines reuse | All models | -| Manual run or unavailable Git base | Full build | Fresh exports | All models | - -Draft PRs skip the pipeline. Marking a PR ready starts it; returning it to draft -cancels an in-progress run through workflow concurrency. Lightweight planner tests -run on every non-draft PR, including documentation changes. C++ formatting runs -when native source or `.clang-format` changes. - -All native platforms use the same selected CMake targets. CMake/Ninja resolves -their actual compilation and linking dependencies. Linux x64 and Windows x64 -smoke tests consume build artifacts from **the same workflow run and checkout**; -there is no polling for a separate Build run. ARM64 keeps its build coverage. -No inference-test success or native binary cache is introduced here. - -## Conservative fallback - -Unowned changed files trigger all known recipes and a full build. Literal CMake -executable declarations outside the root helper file are independently scanned: -an executable absent from the recipes/build-only declarations forces a full -build and all available smoke tests on non-exempt changes. The job summary lists -those executables and calls out the missing inference recipe. This is deliberately -not a general-purpose CMake parser; unusual dynamically declared targets should -get a recipe or an explicit build-only declaration. - -A new recipe with no entry in the dependency registry is always selected. It does -not reuse an earlier run's export. Missing or invalid dependency configuration -falls back to all recipes. Invalid **execution** recipes fail visibly: continuing -without knowing how to execute the test would hide missing coverage. - -FLUX and the base ONNX sample retain their existing build-only hosted coverage, -with explicit reasons in the registry. GPU/local-asset CTest tests are still listed, -not executed on CPU hosted runners. A full-build fallback cannot manufacture the -hardware or model artifacts needed to run those tests. - -## Export reuse - -Export names include a fingerprint of tracked export input blob IDs, export -recipe, model variant/options, and shared exporter runner/workflow. Runtime-only -changes do not invalidate an export. Unowned tracked files are also fingerprinted, -so their changes cannot accidentally restore an old export in a later run. -Unknown dependencies use a run/attempt-specific key to prevent cross-run reuse. - -Only unexpired artifacts from successful allowed runs (default branch, same PR, -or allowed non-PR source branch) are reused. A missing/expired artifact produces a -new export. Precision fallback is preserved: try the requested precision, then -its configured fallback if export fails. Smoke tests still have to pass. - -The existing dependency installation recipes use some floating package/model -versions. Cached exports represent the versions downloaded when they were made, -not a continuous check for upstream updates. Change `export.cache_epoch` to force -refreshes; pin dependency/model revisions in recipes when reproducibility requires -it. A complete package lock and model revision pinning are separate follow-up work. +Unregistered CMake executables trigger a full build and all available recipes; +new inference tests still need an execution recipe. Linux and Windows x64 run +CPU smoke tests; ARM64, FLUX, and the base ONNX sample retain build-only coverage. ## Adding a model -1. Add its normal CMake executable target and exporter. -2. Add `.github/model-tests/.json`, using an existing recipe as an example. - Commands are argument arrays, executed without a shell. Available placeholders - include `{python}`, `{model_id}`, `{output}`, `{precision}`, `{dtype}`, `{slug}`, - and variant fields. Smoke tests additionally receive `{executable}`. -3. Run `python -m unittest discover -s .github/tests -v`. The new recipe is now - discoverable and runs without dependency optimization. -4. Optionally add `models..export` and `.runtime` input patterns to - `.github/model-ci.json`. Reference shared groups rather than copying them. - Include transitive Python helpers, preprocessing, tokenizer code, build - configuration, and test assets. These declarations are correctness contracts. -5. Inspect the CI selection summary and confirm the new variant passes on both - runtime platforms before relying on selective execution. - -Patterns use Python `fnmatch`: `*` matches across directory separators. Renames -are treated as a deletion plus an addition. Unknown ownership always broadens -work. Add ignore patterns only for files known not to affect builds or tests. - -The first run after this migration creates new export artifacts because the -fingerprint format changed. Existing required-check rules referring to the old -standalone model workflows must be updated for the consolidated workflow. +1. Add its CMake target and exporter. +2. Copy an existing `.github/model-tests/.json` recipe and adapt its + variants, commands, and required outputs. The first target is the inference CLI. +3. Optionally declare export and runtime/test inputs in `.github/model-ci.json`, + including shared dependencies. Without these rules, the recipe always runs on + non-exempt changes. Patterns use `fnmatch`; `*` spans directories. +4. Run `python -m unittest discover -s .github/tests -v`, then verify the CI + selection summary and model results on both runtime platforms. + +Exports are fingerprinted from their declared inputs, recipe, and model options. +Bump `export.cache_epoch` to refresh floating upstream dependencies or weights; +pin versions when reproducibility is required. + +When migrating, update required-check rules that reference the old standalone +model workflows. The first run creates exports under the new cache keys. From 687a54ea3f6f90b71e11e20d149ac69c31968997 Mon Sep 17 00:00:00 2001 From: "coderabbitai[bot]" <136622811+coderabbitai[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 08:16:00 +0000 Subject: [PATCH 7/8] Retry incomplete model CI exports with fallback precision and pin NeMo and SAM 2 revisions --- .github/model-tests/nemotron.json | 2 +- .github/model-tests/parakeet.json | 2 +- .github/model-tests/sam2.json | 2 +- .github/scripts/run_model_ci.py | 5 +++- .github/tests/test_model_ci.py | 41 +++++++++++++++++++++++++++++++ 5 files changed, 48 insertions(+), 4 deletions(-) diff --git a/.github/model-tests/nemotron.json b/.github/model-tests/nemotron.json index 9dfd0f4..0f38a2a 100644 --- a/.github/model-tests/nemotron.json +++ b/.github/model-tests/nemotron.json @@ -26,7 +26,7 @@ "onnx", "onnxscript", "soundfile", - "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@cf724ac337d1ebc7d0dda1e23fb80916f52927a5" ] ], "command": [ diff --git a/.github/model-tests/parakeet.json b/.github/model-tests/parakeet.json index 8a82029..cd56005 100644 --- a/.github/model-tests/parakeet.json +++ b/.github/model-tests/parakeet.json @@ -26,7 +26,7 @@ "onnx", "onnxscript", "soundfile", - "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@cf724ac337d1ebc7d0dda1e23fb80916f52927a5" ] ], "command": [ diff --git a/.github/model-tests/sam2.json b/.github/model-tests/sam2.json index 411171d..159a2e0 100644 --- a/.github/model-tests/sam2.json +++ b/.github/model-tests/sam2.json @@ -29,7 +29,7 @@ "numpy", "onnx", "onnxscript", - "sam-2 @ git+https://github.com/facebookresearch/sam2.git" + "sam-2 @ git+https://github.com/facebookresearch/sam2.git@2b90b9f5ceec907a1c18123530e92e794ad901a4" ] ], "command": [ diff --git a/.github/scripts/run_model_ci.py b/.github/scripts/run_model_ci.py index 237ed2c..ab82270 100644 --- a/.github/scripts/run_model_ci.py +++ b/.github/scripts/run_model_ci.py @@ -55,7 +55,10 @@ def export_model(entry, variant, values, key): continue missing = [name for name in definition['required_files'] if not (output / name).is_file()] if missing: - raise RuntimeError(f'Export did not produce required files: {missing}') + if index == len(precisions) - 1: + raise RuntimeError(f'Export did not produce required files: {missing}') + print(f'::warning::{precision} export missing required files: {missing}; trying {precisions[index + 1]}', flush=True) + continue manifest = {'model_id': variant['model_id'], 'precision': precision, 'source_hash': key, 'commit': os.getenv('GITHUB_SHA'), 'variant': variant} (output / 'ci-export-manifest.json').write_text(json.dumps(manifest, indent=2) + '\n') diff --git a/.github/tests/test_model_ci.py b/.github/tests/test_model_ci.py index a2819c1..6e47690 100644 --- a/.github/tests/test_model_ci.py +++ b/.github/tests/test_model_ci.py @@ -258,6 +258,47 @@ def test_successful_command_with_missing_artifacts_fails(self): with self.assertRaisesRegex(RuntimeError, 'required files'): runner.export_model(entry, variant, {'output': str(Path(directory) / 'model')}, 'key') + def test_missing_artifacts_retry_fallback_and_clear_partial_output(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + outputs = Path(directory) / 'outputs' + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx']}} + variant = {'slug': 'example', 'model_id': 'org/model', 'precision': 'fp16', 'fallback_precision': 'fp32'} + attempted = [] + + def execute(command, values, env): + attempted.append(values['precision']) + output.mkdir(parents=True) + if values['precision'] == 'fp16': + (output / 'partial').write_text('partial export') + else: + self.assertFalse((output / 'partial').exists()) + (output / 'model.onnx').write_text('valid model') + + with patch.object(runner, 'run', side_effect=execute), patch.dict(os.environ, {'GITHUB_OUTPUT': str(outputs), 'GITHUB_RUN_ID': '123'}): + runner.export_model(entry, variant, {'output': str(output)}, 'fingerprint') + self.assertEqual(attempted, ['fp16', 'fp32']) + self.assertIn('name=model-example-fp32-fingerprint', outputs.read_text()) + self.assertEqual(json.loads((output / 'ci-export-manifest.json').read_text())['precision'], 'fp32') + + def test_missing_artifacts_after_fallback_fail(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx']}} + variant = {'precision': 'fp16', 'fallback_precision': 'fp32'} + attempted = [] + + def execute(command, values, env): + attempted.append(values['precision']) + output.mkdir(parents=True) + + with patch.object(runner, 'run', side_effect=execute), patch.object(runner, 'write_outputs') as write_outputs: + with self.assertRaisesRegex(RuntimeError, 'required files.*model.onnx'): + runner.export_model(entry, variant, {'output': str(output)}, 'key') + self.assertEqual(attempted, ['fp16', 'fp32']) + self.assertFalse((output / 'ci-export-manifest.json').exists()) + write_outputs.assert_not_called() + class ConfigurationTests(unittest.TestCase): def test_malformed_optional_configuration_disables_optimization(self): From a968bf677f46ed3f5d23976f64dae13ddfbe1f2c Mon Sep 17 00:00:00 2001 From: "coderabbitai[bot]" <136622811+coderabbitai[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:01:34 +0000 Subject: [PATCH 8/8] Strengthen model CI fallback test to verify stale model cleanup before retry --- .github/tests/test_model_ci.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/.github/tests/test_model_ci.py b/.github/tests/test_model_ci.py index 6e47690..e6dd24c 100644 --- a/.github/tests/test_model_ci.py +++ b/.github/tests/test_model_ci.py @@ -262,7 +262,7 @@ def test_missing_artifacts_retry_fallback_and_clear_partial_output(self): with tempfile.TemporaryDirectory() as directory: output = Path(directory) / 'model' outputs = Path(directory) / 'outputs' - entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx']}} + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx', 'metadata.json']}} variant = {'slug': 'example', 'model_id': 'org/model', 'precision': 'fp16', 'fallback_precision': 'fp32'} attempted = [] @@ -271,9 +271,12 @@ def execute(command, values, env): output.mkdir(parents=True) if values['precision'] == 'fp16': (output / 'partial').write_text('partial export') + (output / 'model.onnx').write_text('stale model') else: self.assertFalse((output / 'partial').exists()) + self.assertFalse((output / 'model.onnx').exists()) (output / 'model.onnx').write_text('valid model') + (output / 'metadata.json').write_text('valid metadata') with patch.object(runner, 'run', side_effect=execute), patch.dict(os.environ, {'GITHUB_OUTPUT': str(outputs), 'GITHUB_RUN_ID': '123'}): runner.export_model(entry, variant, {'output': str(output)}, 'fingerprint')