diff --git a/.coderabbit.yaml b/.coderabbit.yaml index a96fa08..66a4727 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -6,6 +6,10 @@ reviews: request_changes_workflow: false fail_commit_status: false + pre_merge_checks: + docstrings: + mode: "off" + high_level_summary: true high_level_summary_in_walkthrough: true collapse_walkthrough: true @@ -21,3 +25,9 @@ reviews: base_branches: [".*"] auto_incremental_review: true auto_pause_after_reviewed_commits: 0 + +knowledge_base: + code_guidelines: + enabled: true + filePatterns: + - "AGENTS.md" diff --git a/.github/model-ci.json b/.github/model-ci.json new file mode 100644 index 0000000..7ff5868 --- /dev/null +++ b/.github/model-ci.json @@ -0,0 +1,146 @@ +{ + "version": 1, + "ignore": [ + "**/*.md", + "*.md", + ".coderabbit.yaml", + ".coderabbit.yml", + "LICENSE", + "THIRD_PARTY_NOTICES.md" + ], + "groups": { + "native": [ + "CMakeLists.txt", + "CMakePresets.json", + "cmake/**", + "common/**/*.cpp", + "common/**/*.h", + "common/*.cpp", + "common/*.h", + "common/**/CMakeLists.txt", + "common/CMakeLists.txt" + ], + "nemo-export": [ + "asr/rnnt/python/tools/nemo_preprocessor_export.py", + "asr/rnnt/python/rnnt/__init__.py", + "asr/rnnt/python/rnnt/audio.py" + ], + "rnnt-runtime": [ + "asr/rnnt/cpp/asr*", + "asr/rnnt/CMakeLists.txt", + "asr/rnnt/python/rnnt/validation/**", + "asr/rnnt/python/tests/**" + ], + "audio-test": [ + "assets/sample.wav" + ] + }, + "models": { + "parakeet": { + "export": { + "paths": [ + "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", + "asr/rnnt/python/rnnt/parakeet_tdt/nemo_backend.py", + "asr/rnnt/python/rnnt/parakeet_tdt/__init__.py" + ], + "groups": [ + "nemo-export" + ] + }, + "runtime": { + "paths": [ + "asr/rnnt/cpp/parakeet*", + "asr/rnnt/python/rnnt/parakeet_tdt/**" + ], + "groups": [ + "native", + "rnnt-runtime", + "audio-test" + ] + } + }, + "nemotron": { + "export": { + "paths": [ + "asr/rnnt/python/tools/export_nemotron_onnx.py", + "asr/rnnt/python/rnnt/nemotron_asr/nemo_backend.py", + "asr/rnnt/python/rnnt/nemotron_asr/__init__.py" + ], + "groups": [ + "nemo-export" + ] + }, + "runtime": { + "paths": [ + "asr/rnnt/cpp/nemotron*", + "asr/rnnt/python/rnnt/nemotron_asr/**" + ], + "groups": [ + "native", + "rnnt-runtime", + "audio-test" + ] + } + }, + "whisper": { + "export": { + "paths": [ + "asr/whisper/model_export/**", + "common/model_export/**" + ] + }, + "runtime": { + "paths": [ + "asr/whisper/**" + ], + "groups": [ + "native", + "audio-test" + ] + } + }, + "sam2": { + "export": { + "paths": [ + "vision/sam2/python/export_sam2_onnx.py", + "vision/sam2/python/modeling.py" + ] + }, + "runtime": { + "paths": [ + "vision/sam2/**", + "assets/sam2-*" + ], + "groups": [ + "native" + ] + } + } + }, + "build_only": { + "flux2": { + "targets": [ + "din_flux2_cli" + ], + "paths": [ + "image_gen/flux2/**" + ], + "groups": [ + "native" + ], + "reason": "GPU inference requires model assets and a GPU runner; existing hosted CI only builds this sample." + }, + "base": { + "targets": [ + "din_base_onnx" + ], + "paths": [ + "tests/base/**" + ], + "groups": [ + "native" + ], + "reason": "Existing hosted CI builds this sample; it has no portable CPU smoke recipe." + } + } +} diff --git a/.github/model-tests/nemotron.json b/.github/model-tests/nemotron.json new file mode 100644 index 0000000..0f38a2a --- /dev/null +++ b/.github/model-tests/nemotron.json @@ -0,0 +1,76 @@ +{ + "name": "nemotron", + "source_roots": [ + "asr/rnnt" + ], + "targets": [ + "din_asr_nemotron_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "soundfile", + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@cf724ac337d1ebc7d0dda1e23fb80916f52927a5" + ] + ], + "command": [ + "{python}", + "asr/rnnt/python/tools/export_nemotron_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--external-data" + ], + "required_files": [ + "preprocessor.onnx", + "encoder_step.onnx", + "predict_step.onnx", + "metadata.json", + "vocab.txt", + "decoder_init_hidden.f32", + "decoder_init_cell.f32" + ], + "env": { + "PYTHONPATH": "asr/rnnt/python" + } + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "nemotron-3.5-asr-streaming-0.6b", + "model_id": "nvidia/nemotron-3.5-asr-streaming-0.6b", + "precision": "fp16", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/model-tests/parakeet.json b/.github/model-tests/parakeet.json new file mode 100644 index 0000000..cd56005 --- /dev/null +++ b/.github/model-tests/parakeet.json @@ -0,0 +1,74 @@ +{ + "name": "parakeet", + "source_roots": [ + "asr/rnnt" + ], + "targets": [ + "din_asr_parakeet_tdt_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "soundfile", + "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@cf724ac337d1ebc7d0dda1e23fb80916f52927a5" + ] + ], + "command": [ + "{python}", + "asr/rnnt/python/tools/export_parakeet_tdt_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--external-data" + ], + "required_files": [ + "preprocessor.onnx", + "encoder.onnx", + "predict_step.onnx", + "metadata.json", + "tokenizer.json" + ], + "env": { + "PYTHONPATH": "asr/rnnt/python" + } + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "parakeet-tdt", + "model_id": "nvidia/parakeet-tdt-0.6b-v3", + "precision": "fp16", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/model-tests/sam2.json b/.github/model-tests/sam2.json new file mode 100644 index 0000000..159a2e0 --- /dev/null +++ b/.github/model-tests/sam2.json @@ -0,0 +1,100 @@ +{ + "name": "sam2", + "source_roots": [ + "vision/sam2" + ], + "targets": [ + "din_sam2_cli", + "din_sam2_verify_masks" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch", + "torchvision" + ], + [ + "{python}", + "-m", + "pip", + "install", + "huggingface_hub", + "numpy", + "onnx", + "onnxscript", + "sam-2 @ git+https://github.com/facebookresearch/sam2.git@2b90b9f5ceec907a1c18123530e92e794ad901a4" + ] + ], + "command": [ + "{python}", + "vision/sam2/python/export_sam2_onnx.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{dtype}", + "--max-points", + "3" + ], + "required_files": [ + "image_preprocess.onnx", + "image_encoder.onnx", + "sam2_decoder.onnx", + "sam2_decoder_propagate.onnx", + "memory_attention.onnx", + "memory_encoder.onnx", + "mask_postprocess.onnx", + "metadata.json" + ] + }, + "smoke": { + "decode_base64": { + "assets/sam2-ci.png.base64": "assets/sam2-ci.png" + }, + "mkdir": [ + "out/{slug}" + ], + "command": [ + "{executable}", + "assets/sam2-ci.png", + "assets/sam2-ci-prompt.json", + "out/{slug}", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "sam2.1-hiera-tiny", + "model_id": "facebook/sam2.1-hiera-tiny", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-small", + "model_id": "facebook/sam2.1-hiera-small", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-base-plus", + "model_id": "facebook/sam2.1-hiera-base-plus", + "precision": "fp16" + }, + { + "slug": "sam2.1-hiera-large", + "model_id": "facebook/sam2.1-hiera-large", + "precision": "fp16" + } + ] +} diff --git a/.github/model-tests/whisper.json b/.github/model-tests/whisper.json new file mode 100644 index 0000000..55ee04f --- /dev/null +++ b/.github/model-tests/whisper.json @@ -0,0 +1,105 @@ +{ + "name": "whisper", + "source_roots": [ + "asr/whisper" + ], + "targets": [ + "din_asr_whisper_cli" + ], + "export": { + "cache_epoch": 1, + "install": [ + [ + "{python}", + "-m", + "pip", + "install", + "--index-url", + "https://download.pytorch.org/whl/cpu", + "torch" + ], + [ + "{python}", + "-m", + "pip", + "install", + "onnx", + "onnxscript", + "safetensors", + "transformers" + ] + ], + "command": [ + "{python}", + "asr/whisper/model_export/export_whisper.py", + "--model", + "{model_id}", + "--output", + "{output}", + "--device", + "cpu", + "--dtype", + "{precision}", + "--attention", + "{attention}" + ], + "required_files": [ + "encoder.onnx", + "decoder.onnx", + "mel.onnx", + "vocab.json" + ] + }, + "smoke": { + "command": [ + "{executable}", + "assets/sample.wav", + "--model-dir", + "{output}", + "--provider", + "cpu" + ] + }, + "variants": [ + { + "slug": "whisper-tiny", + "model_id": "openai/whisper-tiny", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-base", + "model_id": "openai/whisper-base", + "precision": "fp32", + "attention": "sdpa" + }, + { + "slug": "whisper-small", + "model_id": "openai/whisper-small", + "precision": "fp16", + "attention": "math", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-medium", + "model_id": "openai/whisper-medium", + "precision": "fp32", + "attention": "sdpa" + }, + { + "slug": "whisper-large-v3", + "model_id": "openai/whisper-large-v3", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + }, + { + "slug": "whisper-large-v3-turbo", + "model_id": "openai/whisper-large-v3-turbo", + "precision": "fp16", + "attention": "sdpa", + "fallback_precision": "fp32" + } + ] +} diff --git a/.github/scripts/find_successful_artifact.py b/.github/scripts/find_successful_artifact.py index 82338ee..4174f0e 100644 --- a/.github/scripts/find_successful_artifact.py +++ b/.github/scripts/find_successful_artifact.py @@ -25,7 +25,10 @@ def github_api(path: str) -> object: return json.load(response) -def write_outputs(*, found: bool, artifact_name: str, run_id: str = "") -> None: +def write_outputs(*, found: bool, artifact_name: str, run_id: str = "", as_json: bool = False) -> None: + if as_json: + print(json.dumps({"found": found, "name": artifact_name if found else "", "run_id": run_id})) + return output_path = Path(os.environ["GITHUB_OUTPUT"]) with output_path.open("a", encoding="utf-8") as output: output.write(f"found={'true' if found else 'false'}\n") @@ -40,16 +43,15 @@ def main() -> None: parser.add_argument("--default-branch", required=True) parser.add_argument("--event-name", required=True) parser.add_argument("--source-branch", required=True) + parser.add_argument("--pull-request", default="") + parser.add_argument("--json", action="store_true") args = parser.parse_args() artifact_name = args.artifact_name allowed_branches = {args.default_branch} - if args.event_name != "pull_request": + if args.event_name != "pull_request" or args.pull_request: allowed_branches.add(args.source_branch) - response = github_api( - f"repos/{args.repository}/actions/artifacts" - f"?name={quote(artifact_name)}&per_page=100" - ) + response = github_api(f"repos/{args.repository}/actions/artifacts?name={quote(artifact_name)}&per_page=100") artifacts = sorted( ( artifact @@ -67,18 +69,23 @@ def main() -> None: run_id = int(artifact["workflow_run"]["id"]) if run_id not in conclusions: run = github_api(f"repos/{args.repository}/actions/runs/{run_id}") + same_pr = args.pull_request and any( + str(pr["number"]) == args.pull_request for pr in run.get("pull_requests", []) + ) + trusted_branch = run["event"] != "pull_request" and run["head_branch"] in allowed_branches conclusions[run_id] = ( - run["status"] == "completed" and run["conclusion"] == "success" + (trusted_branch or same_pr) and run["status"] == "completed" and run["conclusion"] == "success" ) if conclusions[run_id]: write_outputs( found=True, artifact_name=artifact_name, run_id=str(run_id), + as_json=args.json, ) return - write_outputs(found=False, artifact_name=artifact_name) + write_outputs(found=False, artifact_name=artifact_name, as_json=args.json) if __name__ == "__main__": diff --git a/.github/scripts/model_ci.py b/.github/scripts/model_ci.py new file mode 100644 index 0000000..deee66f --- /dev/null +++ b/.github/scripts/model_ci.py @@ -0,0 +1,240 @@ +#!/usr/bin/env python3 +"""Plan model CI from optional dependency declarations; uncertainty runs more work.""" + +import argparse +import fnmatch +import hashlib +import json +import os +import re +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +CONFIG = '.github/model-ci.json' +RECIPES = '.github/model-tests' +CONTROL = ['.github/scripts/**', '.github/workflows/**', '.github/tests/**', CONFIG] +NATIVE_SUFFIXES = {'.c', '.cc', '.cpp', '.cxx', '.h', '.hh', '.hpp', '.hxx', '.cu', '.cuh'} + + +def git(*args, root=ROOT): + return subprocess.check_output(['git', *args], cwd=root) + + +def matches(path, patterns): + # fnmatch intentionally lets * span directories. Both common/*.h and ** work. + return any(fnmatch.fnmatchcase(path, pattern) for pattern in patterns) + + +def read_config(root=ROOT): + try: + config = json.loads((root / CONFIG).read_text()) + if not isinstance(config, dict) or config.get('version') != 1: + raise ValueError('unsupported dependency configuration version') + for field in ('groups', 'models', 'build_only'): + if not isinstance(config.get(field, {}), dict): + raise ValueError(f'{field} must be an object') + def patterns(value): + if not isinstance(value, list) or not all(isinstance(p, str) for p in value): + raise ValueError('dependency patterns and group names must be string lists') + patterns(config.get('ignore', [])) + for value in config.get('groups', {}).values(): + patterns(value) + sections = list(config.get('build_only', {}).values()) + for model in config.get('models', {}).values(): + if not isinstance(model, dict): + raise ValueError('model dependencies must be objects') + sections.extend(model.values()) + for section in sections: + if not isinstance(section, dict): + raise ValueError('dependency sections must be objects') + patterns(section.get('paths', [])) + patterns(section.get('groups', [])) + return config + except (OSError, ValueError) as error: + print(f'::warning::Dependency optimization disabled: {error}') + return {} + + +def recipes(root=ROOT): + result = {} + for path in sorted((root / RECIPES).glob('*.json')): + recipe = json.loads(path.read_text()) + name = recipe['name'] + if name in result or path.stem != name or not re.fullmatch(r'[a-z0-9-]+', name): + raise ValueError(f'invalid or duplicate recipe name: {path}') + if not recipe['targets'] or not recipe['variants'] or not recipe['source_roots']: + raise ValueError(f'incomplete execution recipe: {name}') + slugs = [v['slug'] for v in recipe['variants']] + if len(set(slugs)) != len(slugs) or any(not re.fullmatch(r'[a-z0-9.-]+', s) for s in slugs): + raise ValueError(f'invalid or duplicate variant slug: {name}') + for target in recipe['targets']: + if not re.fullmatch(r'[A-Za-z0-9_-]+', target): + raise ValueError(f'invalid CMake target: {target}') + for variant in recipe['variants']: + if variant['precision'] not in ('fp16', 'fp32') or not variant['model_id']: + raise ValueError(f'invalid export variant: {variant}') + result[name] = recipe + if not result: + raise ValueError('No model execution recipes found; refusing to report empty coverage') + return result + + +def inputs(section, config): + if not isinstance(section, dict): + return None + patterns = section.get('paths', []).copy() + for group in section.get('groups', []): + if group not in config.get('groups', {}): + return None + patterns.extend(config['groups'][group]) + return patterns or None + + +def discover_unregistered(config, entries, root=ROOT): + """Discover literal sample targets independently of the optimization registry. + + CMake remains responsible for actual build dependencies. Unknown/dynamic + declarations get a full build, never a guessed partial target list. + """ + known = {t for entry in entries.values() for t in entry['targets']} + known.update(t for item in config.get('build_only', {}).values() for t in item['targets']) + unknown = [] + tracked = git('ls-files', '-z', root=root).decode().split('\0') + for path in tracked: + if not path.endswith('CMakeLists.txt') or path == 'CMakeLists.txt': + continue + text = (root / path).read_text() + text = re.sub(r'#[^\n]*', '', text) + for target in re.findall(r'\badd_(?:din_)?executable\s*\(\s*([^\s)]+)', text, re.I): + target = target.strip('"') + if target not in known: + unknown.append(f'{path}: {target}') + return sorted(unknown) + + +def plan(files, config, entries, unknown_targets=()): + reasons = [] + force_all = files is None or not config + force_export = files is None or not config + if files is None: + reasons.append('No reliable change base (manual/initial run): run all') + files = [p for p in files or [] if p] + # Positive exemptions, not a blanket "all YAML is harmless" rule. + relevant = [p for p in files if not matches(p, config.get('ignore', []))] + if config and files and not relevant: + return dict(build=False, targets=[], matrix={'include': []}, format=False, + reasons=['Only explicitly ignored documentation/review files changed'], unknown_targets=[]) + if unknown_targets: + force_all = force_export = True + reasons.append('Unregistered CMake targets: full build and all available smoke recipes') + paths_by_model = {} + for name in entries: + dependency = config.get('models', {}).get(name, {}) + paths_by_model[name] = (inputs(dependency.get('export'), config), inputs(dependency.get('runtime'), config)) + owned = [p for pair in paths_by_model.values() for patterns in pair if patterns for p in patterns] + for item in config.get('build_only', {}).values(): + owned.extend(inputs(item, config) or []) + recipe_paths = [f'{RECIPES}/{name}.json' for name in entries] + unknown_files = [p for p in relevant if not matches(p, owned + CONTROL + recipe_paths)] + if unknown_files: + force_all = force_export = True + reasons.append('Unowned changes: ' + ', '.join(unknown_files)) + control_changed = any(matches(p, CONTROL) for p in relevant) + selected, targets = [], set() + full_build = force_all or control_changed + for name, entry in entries.items(): + export_paths, runtime_paths = paths_by_model[name] + undeclared = export_paths is None or runtime_paths is None + changed = any(matches(p, (export_paths or []) + (runtime_paths or []) + [f'{RECIPES}/{name}.json']) for p in relevant) + if force_all or control_changed or undeclared or changed: + reasons.append(f'{name}: ' + ('missing dependency rules; run conservatively' if undeclared else 'affected or full validation')) + targets.update(entry['targets']) + for variant in entry['variants']: + selected.append({'model': name, 'variant': variant['slug'], 'force_export': force_export or export_paths is None}) + for name, item in config.get('build_only', {}).items(): + patterns = inputs(item, config) + if force_all or control_changed or patterns is None or any(matches(p, patterns) for p in relevant): + targets.update(item['targets']) + reasons.append(f'{name}: build only ({item["reason"]})') + return dict(build=bool(targets) or full_build, targets=[] if full_build else sorted(targets), + matrix={'include': selected}, format=files is None or any(Path(p).suffix in NATIVE_SUFFIXES or p == '.clang-format' for p in files), + reasons=reasons, unknown_targets=list(unknown_targets)) + + +def changed_files(event, root=ROOT): + pr = event.get('pull_request') + base = pr['base']['sha'] if pr else event.get('before') + if not base or not base.strip('0'): + return None + try: + git('cat-file', '-e', f'{base}^{{commit}}', root=root) + if pr: + base = git('merge-base', base, 'HEAD', root=root).decode().strip() + return git('diff', '--no-renames', '--name-only', '-z', base, 'HEAD', root=root).decode().split('\0') + except subprocess.CalledProcessError: + return None + + +def export_key(name, variant, config, entries, root=ROOT, force=False): + entry = entries[name] + patterns = inputs(config.get('models', {}).get(name, {}).get('export'), config) + fingerprint = {'recipe': entry['export'], 'variant': variant, 'inputs': patterns, + 'environment': 'ubuntu-24.04-python-3.12-cpu'} + # The runner and export workflow affect all export recipes, unlike runtime code. + patterns = (patterns or []) + ['.github/scripts/model_ci.py', '.github/scripts/run_model_ci.py', '.github/workflows/model-test.yml'] + owned = CONTROL + config.get('ignore', []) + [RECIPES + '/**'] + for dependency in config.get('models', {}).values(): + for stage in ('export', 'runtime'): + owned.extend(inputs(dependency.get(stage), config) or []) + for component in config.get('build_only', {}).values(): + owned.extend(inputs(component, config) or []) + sources = [] + for record in git('ls-files', '--stage', '-z', root=root).decode().split('\0'): + if not record: + continue + metadata, path = record.split('\t', 1) + if matches(path, patterns) or not matches(path, owned): + sources.append([path, metadata]) + fingerprint['sources'] = sources + if force or not inputs(config.get('models', {}).get(name, {}).get('export'), config): + # Unknown export dependencies must not reuse an earlier run's model. + fingerprint['uncertain_run'] = [os.getenv('GITHUB_RUN_ID', 'local'), os.getenv('GITHUB_RUN_ATTEMPT', '1'), git('rev-parse', 'HEAD', root=root).decode().strip()] + return hashlib.sha256(json.dumps(fingerprint, sort_keys=True).encode()).hexdigest()[:24] + + +def write_outputs(values): + with Path(os.environ['GITHUB_OUTPUT']).open('a') as output: + for key, value in values.items(): + output.write(f'{key}={json.dumps(value, separators=(",", ":")) if not isinstance(value, str) else value}\n') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('mode', choices=['plan', 'build']) + parser.add_argument('--preset') + args = parser.parse_args() + if args.mode == 'build': + targets = json.loads(os.environ.get('BUILD_TARGETS', '[]')) + command = ['cmake', '--build', '--preset', args.preset, '--parallel'] + if targets: + command += ['--target', *targets] + subprocess.run(command, check=True) + return + entries = recipes() + config = read_config() + event = json.loads(Path(os.environ['GITHUB_EVENT_PATH']).read_text()) + result = plan(changed_files(event), config, entries, discover_unregistered(config, entries)) + print(json.dumps(result, indent=2)) + write_outputs({'build': result['build'], 'targets': result['targets'], 'matrix': result['matrix'], + 'models': bool(result['matrix']['include']), 'format': result['format']}) + if os.getenv('GITHUB_STEP_SUMMARY'): + with Path(os.environ['GITHUB_STEP_SUMMARY']).open('a') as summary: + summary.write('## CI selection\n\n' + '\n'.join('- ' + reason for reason in result['reasons']) + '\n') + if result['unknown_targets']: + summary.write('\nUnregistered executables are included in the full build. Add an execution recipe to run their inference tests:\n\n') + summary.write('\n'.join('- ' + name for name in result['unknown_targets']) + '\n') + + +if __name__ == '__main__': + main() diff --git a/.github/scripts/run_model_ci.py b/.github/scripts/run_model_ci.py new file mode 100644 index 0000000..ab82270 --- /dev/null +++ b/.github/scripts/run_model_ci.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +"""Execute model recipes without model-specific workflow branches.""" + +import argparse +import base64 +import json +import os +import shutil +import subprocess +import sys +from pathlib import Path + +from model_ci import ROOT, export_key, read_config, recipes, write_outputs + + +def expand(command, values): + return [argument.format_map(values) for argument in command] + + +def run(command, values, env=None): + subprocess.run(expand(command, values), cwd=ROOT, env=env, check=True) + + +def artifact_name(slug, precision, key): + return f'model-{slug}-{precision}-{key}' + + +def find_artifact(name): + command = [sys.executable, str(ROOT / '.github/scripts/find_successful_artifact.py'), + '--repository', os.environ['GITHUB_REPOSITORY'], '--artifact-name', name, + '--default-branch', os.environ['DEFAULT_BRANCH'], '--event-name', os.environ['EVENT_NAME'], + '--source-branch', os.environ['SOURCE_BRANCH'], '--pull-request', os.environ.get('PULL_REQUEST', ''), '--json'] + return json.loads(subprocess.check_output(command, text=True)) + + +def export_model(entry, variant, values, key): + definition = entry['export'] + env = dict(os.environ, **definition.get('env', {})) + for command in definition['install']: + run(command, values, env) + output = Path(values['output']) + precisions = [variant['precision']] + if variant.get('fallback_precision'): + precisions.append(variant['fallback_precision']) + for index, precision in enumerate(precisions): + current = dict(values, precision=precision, dtype={'fp16': 'float16', 'fp32': 'float32'}[precision]) + if output.exists(): + shutil.rmtree(output) + try: + run(definition['command'], current, env) + except subprocess.CalledProcessError: + if index == len(precisions) - 1: + raise + print(f'::warning::{precision} export failed; trying {precisions[index + 1]}', flush=True) + continue + missing = [name for name in definition['required_files'] if not (output / name).is_file()] + if missing: + if index == len(precisions) - 1: + raise RuntimeError(f'Export did not produce required files: {missing}') + print(f'::warning::{precision} export missing required files: {missing}; trying {precisions[index + 1]}', flush=True) + continue + manifest = {'model_id': variant['model_id'], 'precision': precision, 'source_hash': key, + 'commit': os.getenv('GITHUB_SHA'), 'variant': variant} + (output / 'ci-export-manifest.json').write_text(json.dumps(manifest, indent=2) + '\n') + write_outputs({'name': artifact_name(variant['slug'], precision, key), 'run_id': os.environ['GITHUB_RUN_ID']}) + return + + +def smoke_test(entry, values): + definition = entry['smoke'] + for source, destination in definition.get('decode_base64', {}).items(): + (ROOT / destination).write_bytes(base64.b64decode((ROOT / source).read_text())) + for directory in definition.get('mkdir', []): + (ROOT / directory.format_map(values)).mkdir(parents=True, exist_ok=True) + executable = ROOT / 'runtime' / (entry['targets'][0] + ('.exe' if os.name == 'nt' else '')) + if os.name != 'nt': + executable.chmod(executable.stat().st_mode | 0o111) + env = dict(os.environ) + env['LD_LIBRARY_PATH'] = str(ROOT / 'runtime') + os.pathsep + env.get('LD_LIBRARY_PATH', '') + run(definition['command'], dict(values, executable=str(executable)), env) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('mode', choices=['resolve', 'export', 'smoke']) + args = parser.parse_args() + entries = recipes() + entry = entries[os.environ['MODEL']] + variant = next(v for v in entry['variants'] if v['slug'] == os.environ['VARIANT']) + values = dict(variant, python=sys.executable, output=str(ROOT / 'artifacts/ci' / variant['slug'])) + if args.mode == 'smoke': + smoke_test(entry, values) + return + key = export_key(entry['name'], variant, read_config(), entries, force=os.getenv('FORCE_EXPORT') == 'true') + if args.mode == 'export': + export_model(entry, variant, values, key) + return + for precision in dict.fromkeys([variant['precision'], variant.get('fallback_precision')]): + if not precision: + continue + found = find_artifact(artifact_name(variant['slug'], precision, key)) + if found['found']: + write_outputs(found) + return + write_outputs({'found': False, 'name': '', 'run_id': ''}) + + +if __name__ == '__main__': + main() diff --git a/.github/tests/test_model_ci.py b/.github/tests/test_model_ci.py new file mode 100644 index 0000000..e6dd24c --- /dev/null +++ b/.github/tests/test_model_ci.py @@ -0,0 +1,317 @@ +import copy +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +SCRIPTS = Path(__file__).resolve().parents[1] / 'scripts' +sys.path.insert(0, str(SCRIPTS)) +import model_ci as ci +import run_model_ci as runner + + +class SelectionTests(unittest.TestCase): + def setUp(self): + self.config = ci.read_config() + self.entries = {name: value for name, value in ci.recipes().items() if name in {"whisper", "parakeet", "nemotron", "sam2"}} + + def plan(self, *paths, **kwargs): + return ci.plan(list(paths), self.config, self.entries, **kwargs) + + def selected(self, result): + return {row['model'] for row in result['matrix']['include']} + + def test_review_config_and_docs_do_not_build_or_export(self): + for path in ['.coderabbit.yaml', 'README.md', 'asr/rnnt/README.md']: + with self.subTest(path=path): + result = self.plan(path) + self.assertFalse(result['build']) + self.assertEqual(self.selected(result), set()) + + def test_parakeet_runtime_selects_only_parakeet(self): + result = self.plan('asr/rnnt/cpp/parakeet_tdt.cpp') + self.assertEqual(self.selected(result), {'parakeet'}) + self.assertEqual(result['targets'], ['din_asr_parakeet_tdt_cli']) + self.assertFalse(result['matrix']['include'][0]['force_export']) + + def test_nemotron_runtime_selects_only_nemotron(self): + self.assertEqual(self.selected(self.plan('asr/rnnt/cpp/nemotron.cpp')), {'nemotron'}) + + def test_export_change_also_selects_inference(self): + self.assertEqual(self.selected(self.plan('asr/whisper/model_export/export_whisper.py')), {'whisper'}) + + def test_shared_audio_runs_all_models(self): + self.assertEqual(self.selected(self.plan('common/io/audio.cpp')), set(self.entries)) + + def test_shared_export_selects_only_consumers(self): + self.assertEqual(self.selected(self.plan('asr/rnnt/python/tools/nemo_preprocessor_export.py')), {'parakeet', 'nemotron'}) + + def test_test_asset_selects_consumers_without_forcing_export(self): + result = self.plan('assets/sample.wav') + self.assertEqual(self.selected(result), {'whisper', 'parakeet', 'nemotron'}) + self.assertFalse(any(row['force_export'] for row in result['matrix']['include'])) + + def test_unknown_file_runs_everything_and_forces_export(self): + result = self.plan('new_component/kernel.cu') + self.assertEqual(self.selected(result), set(self.entries)) + self.assertEqual(result['targets'], []) + self.assertTrue(all(row['force_export'] for row in result['matrix']['include'])) + + def test_unknown_yaml_is_not_silently_ignored(self): + self.assertEqual(self.selected(self.plan('model-settings.yaml')), set(self.entries)) + + def test_missing_dependency_entry_always_runs_existing_recipe(self): + del self.config['models']['parakeet'] + result = self.plan('vision/sam2/cpp/main.cpp') + self.assertIn('parakeet', self.selected(result)) + self.assertTrue(next(row['force_export'] for row in result['matrix']['include'] if row['model'] == 'parakeet')) + + def test_new_recipe_requires_no_planner_change(self): + self.entries['new-model'] = copy.deepcopy(self.entries['parakeet']) + self.entries['new-model']['targets'] = ['din_new_cli'] + self.entries['new-model']['variants'] = [{'slug': 'new-variant'}] + result = self.plan('assets/sample.wav') + self.assertIn({'model': 'new-model', 'variant': 'new-variant', 'force_export': True}, result['matrix']['include']) + self.assertIn('din_new_cli', result['targets']) + + def test_unknown_group_runs_component_conservatively(self): + self.config['models']['whisper']['export']['groups'] = ['missing'] + result = self.plan('asr/rnnt/cpp/parakeet_tdt.cpp') + self.assertIn('whisper', self.selected(result)) + + def test_unknown_target_forces_full_build(self): + result = self.plan('assets/sample.wav', unknown_targets=['models/new/CMakeLists.txt: din_new']) + self.assertEqual(result['targets'], []) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_manual_or_unavailable_base_runs_all(self): + result = ci.plan(None, self.config, self.entries) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_missing_configuration_runs_all(self): + result = ci.plan(['README.md'], {}, self.entries) + self.assertTrue(result['build']) + self.assertEqual(self.selected(result), set(self.entries)) + + def test_workflow_change_validates_all_models(self): + self.assertEqual(self.selected(self.plan('.github/workflows/ci.yml')), set(self.entries)) + + def test_flux_changes_build_without_unrelated_inference(self): + result = self.plan('image_gen/flux2/cpp/flux.cpp') + self.assertEqual(result['targets'], ['din_flux2_cli']) + self.assertEqual(self.selected(result), set()) + + def test_variant_matrix_preserves_existing_coverage(self): + result = ci.plan(None, self.config, self.entries) + self.assertEqual(len(result['matrix']['include']), 12) + + +class GitTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.git('init', '-q') + self.git('config', 'user.name', 'CI Test') + self.git('config', 'user.email', 'ci@example.invalid') + self.write('export.py', 'initial') + self.write('runtime.cpp', 'initial') + self.git('add', '.') + self.git('commit', '-qm', 'initial') + self.base = self.git('rev-parse', 'HEAD').strip() + self.config = {'models': {'demo': {'export': {'paths': ['export.py']}, 'runtime': {'paths': ['runtime.cpp']}}}} + self.entries = {'demo': {'export': {'command': ['export.py']}}} + self.variant = {'model_id': 'model/revision', 'precision': 'fp16'} + + def git(self, *args): + return ci.git(*args, root=self.root).decode() + + def write(self, path, content): + target = self.root / path + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + + def key(self, **kwargs): + return ci.export_key('demo', self.variant, self.config, self.entries, root=self.root, **kwargs) + + def test_runtime_change_preserves_export_key(self): + before = self.key() + self.write('runtime.cpp', 'changed') + self.git('add', '.') + self.assertEqual(before, self.key()) + + def test_exporter_edit_delete_and_rename_change_key(self): + previous = self.key() + self.write('export.py', 'changed') + self.git('add', '.') + self.assertNotEqual(previous, self.key()) + previous = self.key() + self.git('mv', 'export.py', 'renamed.py') + self.assertNotEqual(previous, self.key()) + previous = self.key() + self.git('rm', '-f', 'renamed.py') + self.assertNotEqual(previous, self.key()) + + def test_unknown_file_stays_in_key_on_future_runs(self): + before = self.key() + self.write('unowned_helper.py', 'new export dependency') + self.git('add', '.') + self.assertNotEqual(before, self.key()) + + def test_export_settings_change_key(self): + before = self.key() + self.entries['demo']['export']['install'] = ['changed dependency'] + self.assertNotEqual(before, self.key()) + before = self.key() + self.variant['model_id'] = 'different/model' + self.assertNotEqual(before, self.key()) + + def test_fallback_cannot_reuse_previous_run(self): + with patch.dict(os.environ, {'GITHUB_RUN_ID': '1'}): + before = self.key(force=True) + with patch.dict(os.environ, {'GITHUB_RUN_ID': '2'}): + self.assertNotEqual(before, self.key(force=True)) + + def test_deletion_and_rename_list_both_paths(self): + self.git('mv', 'runtime.cpp', 'renamed.cpp') + self.git('commit', '-qm', 'rename') + changed = ci.changed_files({'before': self.base}, root=self.root) + self.assertIn('runtime.cpp', changed) + self.assertIn('renamed.cpp', changed) + + def test_pr_uses_merge_base(self): + self.git('checkout', '-qb', 'base-update') + self.write('base-only.cpp', 'base change') + self.git('add', '.') + self.git('commit', '-qm', 'base update') + base_tip = self.git('rev-parse', 'HEAD').strip() + self.git('checkout', '-qb', 'feature', self.base) + self.write('runtime.cpp', 'feature change') + self.git('add', '.') + self.git('commit', '-qm', 'feature') + changed = ci.changed_files({'pull_request': {'base': {'sha': base_tip}}}, root=self.root) + self.assertIn('runtime.cpp', changed) + self.assertNotIn('base-only.cpp', changed) + + def test_unknown_target_discovery_is_independent_of_registry(self): + self.write('new/CMakeLists.txt', 'add_din_executable(din_new_cli main.cpp)') + self.git('add', '.') + found = ci.discover_unregistered({}, {}, root=self.root) + self.assertEqual(found, ['new/CMakeLists.txt: din_new_cli']) + + +class ExecutionTests(unittest.TestCase): + def test_subprocess_failures_propagate(self): + with patch.object(subprocess, 'run', side_effect=subprocess.CalledProcessError(1, ['cli'])): + with self.assertRaises(subprocess.CalledProcessError): + runner.run(['{executable}', '--model-dir', '{output}'], {'executable': 'cli', 'output': 'model'}) + + def test_command_arguments_do_not_go_through_shell(self): + with patch.object(subprocess, 'run') as run: + runner.run(['cli', '{output}'], {'output': 'directory with spaces'}) + self.assertEqual(run.call_args.args[0], ['cli', 'directory with spaces']) + self.assertNotIn('shell', run.call_args.kwargs) + + def test_all_current_recipes_expand(self): + for entry in ci.recipes().values(): + for variant in entry['variants']: + values = dict(variant, python='python', output='model', executable='cli', dtype='float16') + runner.expand(entry['export']['command'], values) + runner.expand(entry['smoke']['command'], values) + for command in entry['export']['install']: + runner.expand(command, values) + + +class ExportExecutionTests(unittest.TestCase): + def test_export_retries_fallback_and_records_precision(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + outputs = Path(directory) / 'outputs' + definition = {'export': {'install': [['install']], 'command': ['export', '{precision}'], 'required_files': ['model.onnx']}} + variant = {'slug': 'example', 'model_id': 'org/model', 'precision': 'fp16', 'fallback_precision': 'fp32'} + values = dict(variant, output=str(output)) + def execute(command, expanded, env): + if command[0] == 'install': + return + output.mkdir(parents=True) + if expanded['precision'] == 'fp16': + (output / 'partial').write_text('partial export') + raise subprocess.CalledProcessError(1, command) + self.assertFalse((output / 'partial').exists()) + (output / 'model.onnx').write_text('valid model') + with patch.object(runner, 'run', side_effect=execute), patch.dict(os.environ, {'GITHUB_OUTPUT': str(outputs), 'GITHUB_RUN_ID': '123'}): + runner.export_model(definition, variant, values, 'fingerprint') + self.assertIn('name=model-example-fp32-fingerprint', outputs.read_text()) + self.assertEqual(json.loads((output / 'ci-export-manifest.json').read_text())['precision'], 'fp32') + + def test_successful_command_with_missing_artifacts_fails(self): + with tempfile.TemporaryDirectory() as directory: + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['missing.onnx']}} + variant = {'precision': 'fp32'} + with patch.object(runner, 'run'): + with self.assertRaisesRegex(RuntimeError, 'required files'): + runner.export_model(entry, variant, {'output': str(Path(directory) / 'model')}, 'key') + + def test_missing_artifacts_retry_fallback_and_clear_partial_output(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + outputs = Path(directory) / 'outputs' + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx', 'metadata.json']}} + variant = {'slug': 'example', 'model_id': 'org/model', 'precision': 'fp16', 'fallback_precision': 'fp32'} + attempted = [] + + def execute(command, values, env): + attempted.append(values['precision']) + output.mkdir(parents=True) + if values['precision'] == 'fp16': + (output / 'partial').write_text('partial export') + (output / 'model.onnx').write_text('stale model') + else: + self.assertFalse((output / 'partial').exists()) + self.assertFalse((output / 'model.onnx').exists()) + (output / 'model.onnx').write_text('valid model') + (output / 'metadata.json').write_text('valid metadata') + + with patch.object(runner, 'run', side_effect=execute), patch.dict(os.environ, {'GITHUB_OUTPUT': str(outputs), 'GITHUB_RUN_ID': '123'}): + runner.export_model(entry, variant, {'output': str(output)}, 'fingerprint') + self.assertEqual(attempted, ['fp16', 'fp32']) + self.assertIn('name=model-example-fp32-fingerprint', outputs.read_text()) + self.assertEqual(json.loads((output / 'ci-export-manifest.json').read_text())['precision'], 'fp32') + + def test_missing_artifacts_after_fallback_fail(self): + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / 'model' + entry = {'export': {'install': [], 'command': ['export'], 'required_files': ['model.onnx']}} + variant = {'precision': 'fp16', 'fallback_precision': 'fp32'} + attempted = [] + + def execute(command, values, env): + attempted.append(values['precision']) + output.mkdir(parents=True) + + with patch.object(runner, 'run', side_effect=execute), patch.object(runner, 'write_outputs') as write_outputs: + with self.assertRaisesRegex(RuntimeError, 'required files.*model.onnx'): + runner.export_model(entry, variant, {'output': str(output)}, 'key') + self.assertEqual(attempted, ['fp16', 'fp32']) + self.assertFalse((output / 'ci-export-manifest.json').exists()) + write_outputs.assert_not_called() + + +class ConfigurationTests(unittest.TestCase): + def test_malformed_optional_configuration_disables_optimization(self): + for content in ['{', '[]', '{"version": 1, "models": {"demo": []}}', '{"version": 1, "groups": {"shared": "not-a-list"}}']: + with self.subTest(content=content), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + (root / '.github').mkdir() + (root / ci.CONFIG).write_text(content) + self.assertEqual(ci.read_config(root), {}) + + +if __name__ == '__main__': + unittest.main() diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5d0965e..a55bb53 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -4,6 +4,7 @@ on: push: branches: [main] pull_request: + types: [opened, synchronize, reopened, ready_for_review, converted_to_draft] workflow_dispatch: permissions: @@ -18,7 +19,31 @@ env: CMAKE_BUILD_PARALLEL_LEVEL: "4" jobs: + plan: + if: github.event_name != 'pull_request' || !github.event.pull_request.draft + runs-on: ubuntu-24.04 + outputs: + build: ${{ steps.plan.outputs.build }} + targets: ${{ steps.plan.outputs.targets }} + models: ${{ steps.plan.outputs.models }} + matrix: ${{ steps.plan.outputs.matrix }} + format: ${{ steps.plan.outputs.format }} + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + - uses: actions/setup-python@v7 + with: + python-version: '3.12' + - name: Validate CI selection and execution recipes + run: python -m unittest discover -s .github/tests -v + - name: Select affected components + id: plan + run: python .github/scripts/model_ci.py plan + clang-format: + needs: plan + if: needs.plan.outputs.format == 'true' name: clang-format runs-on: ubuntu-24.04 @@ -47,6 +72,8 @@ jobs: | xargs -0 -r clang-format-22 --dry-run --Werror -- linux-x64: + needs: plan + if: needs.plan.outputs.build == 'true' name: Ubuntu 24.04 / Clang / x64 runs-on: ubuntu-24.04 container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -76,7 +103,9 @@ jobs: -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset linux-x64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python3 .github/scripts/model_ci.py build --preset linux-x64-release - name: List registered CTest tests run: ctest --test-dir out/build/linux-x64 --build-config Release --show-only=human @@ -90,6 +119,8 @@ jobs: path: out/build/linux-x64/bin/Release/ linux-arm64: + needs: plan + if: needs.plan.outputs.build == 'true' name: Ubuntu 24.04 / Clang / ARM64 runs-on: ubuntu-24.04-arm container: nvidia/cuda:13.2.1-devel-ubuntu24.04 @@ -117,7 +148,9 @@ jobs: run: cmake --preset linux-arm64 -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset linux-arm64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python3 .github/scripts/model_ci.py build --preset linux-arm64-release - name: List registered CTest tests run: ctest --test-dir out/build/linux-arm64 --build-config Release --show-only=human @@ -131,6 +164,8 @@ jobs: path: out/build/linux-arm64/bin/Release/ windows-x64: + needs: plan + if: needs.plan.outputs.build == 'true' name: Windows Server 2025 / MSVC / x64 runs-on: windows-2025 timeout-minutes: 90 @@ -162,7 +197,9 @@ jobs: -DDIN_BUILD_TRT_RTX_EP=ON - name: Build - run: cmake --build --preset windows-x64-release --parallel + env: + BUILD_TARGETS: ${{ needs.plan.outputs.targets }} + run: python .github/scripts/model_ci.py build --preset windows-x64-release - name: List registered CTest tests run: ctest --test-dir out/build/windows-x64 --build-config Release --show-only=human @@ -174,3 +211,18 @@ jobs: name: windows-x64-release if-no-files-found: error path: out/build/windows-x64/bin/Release/ + + model-tests: + needs: [plan, linux-x64, windows-x64] + if: >- + !cancelled() && needs.plan.outputs.models == 'true' && + needs.linux-x64.result == 'success' && needs.windows-x64.result == 'success' + name: ${{ matrix.variant }} + strategy: + fail-fast: false + matrix: ${{ fromJSON(needs.plan.outputs.matrix) }} + uses: ./.github/workflows/model-test.yml + with: + model: ${{ matrix.model }} + variant: ${{ matrix.variant }} + force_export: ${{ matrix.force_export }} diff --git a/.github/workflows/export-nvidia-asr-model.yml b/.github/workflows/export-nvidia-asr-model.yml deleted file mode 100644 index dc098d1..0000000 --- a/.github/workflows/export-nvidia-asr-model.yml +++ /dev/null @@ -1,318 +0,0 @@ -name: Export NVIDIA ASR model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - model_type: - required: true - type: string - precision: - required: true - type: string - fallback_precision: - required: false - default: '' - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.final-artifact.outputs.name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'asr/rnnt/python/*.py' \ - 'asr/rnnt/python/**/*.py' \ - '.github/workflows/export-nvidia-asr-model.yml' \ - '.github/workflows/nvidia-asr.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - - name: Compute artifact names - id: artifact-names - shell: bash - env: - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_SLUG: ${{ inputs.model_slug }} - MODEL_TYPE: ${{ inputs.model_type }} - PRECISION: ${{ inputs.precision }} - run: | - set -euo pipefail - case "$MODEL_TYPE" in - parakeet|nemotron) ;; - *) echo "Unsupported NVIDIA ASR model type: $MODEL_TYPE" >&2; exit 2 ;; - esac - case "$PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported precision: $PRECISION" >&2; exit 2 ;; - esac - if [[ -n "$FALLBACK_PRECISION" ]]; then - case "$FALLBACK_PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; - esac - fi - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}" >> "$GITHUB_OUTPUT" - if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}" >> "$GITHUB_OUTPUT" - fi - - - name: Find successful primary artifact - if: steps.changes.outputs.force != 'true' - id: primary-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" - - - name: Find successful fallback artifact - if: >- - steps.changes.outputs.force != 'true' && - steps.primary-artifact.outputs.found != 'true' && - inputs.fallback_precision != '' - id: fallback-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" - - - name: Select export action - id: selection - shell: bash - env: - FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} - FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} - FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - FORCE_EXPORT: ${{ steps.changes.outputs.force }} - PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} - PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} - PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$FORCE_EXPORT" == "true" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - elif [[ "$PRIMARY_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" - elif [[ "$FALLBACK_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$FALLBACK_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$FALLBACK_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "export=true" >> "$GITHUB_OUTPUT" - fi - - - name: Reclaim runner disk space - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.selection.outputs.export == 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install NeMo export dependencies - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install --index-url https://download.pytorch.org/whl/cpu torch - python -m pip install \ - onnx \ - onnxscript \ - soundfile \ - "nemo_toolkit[asr] @ git+https://github.com/NVIDIA/NeMo.git@main" - - - name: Export NVIDIA ASR ONNX - if: steps.selection.outputs.export == 'true' - id: exported - shell: bash - env: - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - MODEL_SLUG: ${{ inputs.model_slug }} - MODEL_TYPE: ${{ inputs.model_type }} - PRECISION: ${{ inputs.precision }} - PYTHONPATH: asr/rnnt/python - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -uo pipefail - export_dtype() { - case "$1" in - fp16) echo float16 ;; - fp32) echo float32 ;; - *) return 2 ;; - esac - } - run_export() { - local dtype - dtype="$(export_dtype "$1")" - case "$MODEL_TYPE" in - parakeet) - python asr/rnnt/python/tools/export_parakeet_tdt_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$dtype" \ - --external-data - ;; - nemotron) - python asr/rnnt/python/tools/export_nemotron_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$dtype" \ - --external-data - ;; - esac - } - - selected_precision="$PRECISION" - set +e - run_export "$selected_precision" - export_status=$? - set -e - if [[ $export_status -ne 0 ]]; then - if [[ -z "$FALLBACK_PRECISION" ]]; then - exit "$export_status" - fi - echo "::warning::$PRECISION export failed; retrying once with $FALLBACK_PRECISION." - rm -rf "$MODEL_OUTPUT" - selected_precision="$FALLBACK_PRECISION" - run_export "$selected_precision" - fi - - case "$MODEL_TYPE" in - parakeet) - required_files=(preprocessor.onnx encoder.onnx predict_step.onnx metadata.json tokenizer.json) - ;; - nemotron) - required_files=(preprocessor.onnx encoder_step.onnx predict_step.onnx metadata.json vocab.txt decoder_init_hidden.f32 decoder_init_cell.f32) - ;; - esac - for required_file in "${required_files[@]}"; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SELECTED_PRECISION="$selected_precision" SOURCE_REF - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": os.environ["SELECTED_PRECISION"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}" >> "$GITHUB_OUTPUT" - - - name: Upload NVIDIA ASR ONNX - if: steps.selection.outputs.export == 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.exported.outputs.name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - EXISTING_NAME: ${{ steps.selection.outputs.name }} - EXISTING_RUN_ID: ${{ steps.selection.outputs.run_id }} - EXPORTED: ${{ steps.selection.outputs.export }} - EXPORTED_NAME: ${{ steps.exported.outputs.name }} - run: | - set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then - echo "name=$EXPORTED_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "name=$EXISTING_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/export-sam2-model.yml b/.github/workflows/export-sam2-model.yml deleted file mode 100644 index 227015b..0000000 --- a/.github/workflows/export-sam2-model.yml +++ /dev/null @@ -1,210 +0,0 @@ -name: Export SAM2 model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.model-key.outputs.artifact_name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'vision/sam2/python/*.py' \ - 'vision/sam2/python/**/*.py' \ - '.github/workflows/export-sam2-model.yml' \ - '.github/workflows/sam2.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - - name: Compute model artifact key - id: model-key - shell: bash - env: - MODEL_SLUG: ${{ inputs.model_slug }} - run: | - set -euo pipefail - source_hash="$( - git ls-files -z -- 'vision/sam2/python/*.py' 'vision/sam2/python/**/*.py' \ - | xargs -0 sha256sum \ - | sha256sum \ - | cut -d' ' -f1 - )" - echo "source_hash=$source_hash" >> "$GITHUB_OUTPUT" - echo "artifact_name=vision-${MODEL_SLUG}-onnx-v1-${source_hash:0:16}" >> "$GITHUB_OUTPUT" - - - name: Resolve reusable model artifact - id: existing-artifact - shell: bash - env: - FORCE_EXPORT: ${{ steps.changes.outputs.force }} - GH_TOKEN: ${{ github.token }} - MODEL_ARTIFACT: ${{ steps.model-key.outputs.artifact_name }} - run: | - set -euo pipefail - artifact_run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/artifacts?name=${MODEL_ARTIFACT}&per_page=100" \ - --jq '.artifacts | map(select(.expired | not)) | sort_by(.created_at) | last | .workflow_run.id // empty' - )" - if [[ "$FORCE_EXPORT" == "true" || -z "$artifact_run_id" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - else - echo "export=false" >> "$GITHUB_OUTPUT" - fi - echo "run_id=$artifact_run_id" >> "$GITHUB_OUTPUT" - - - name: Reclaim runner disk space - if: steps.existing-artifact.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.existing-artifact.outputs.export == 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install SAM2 export dependencies - if: steps.existing-artifact.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install \ - --index-url https://download.pytorch.org/whl/cpu \ - torch \ - torchvision - python -m pip install \ - huggingface_hub \ - numpy \ - onnx \ - onnxscript \ - "sam-2 @ git+https://github.com/facebookresearch/sam2.git" - - - name: Export SAM2 ONNX - if: steps.existing-artifact.outputs.export == 'true' - shell: bash - env: - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - SOURCE_HASH: ${{ steps.model-key.outputs.source_hash }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euxo pipefail - python vision/sam2/python/export_sam2_onnx.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype float16 \ - --max-points 3 - - required_files=( - image_preprocess.onnx - image_encoder.onnx - sam2_decoder.onnx - sam2_decoder_propagate.onnx - memory_attention.onnx - memory_encoder.onnx - mask_postprocess.onnx - metadata.json - ) - for required_file in "${required_files[@]}"; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SOURCE_HASH - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": "float16", - "source_hash": os.environ["SOURCE_HASH"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - - - name: Upload SAM2 ONNX - if: steps.existing-artifact.outputs.export == 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.model-key.outputs.artifact_name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - EXPORTED: ${{ steps.existing-artifact.outputs.export }} - EXISTING_RUN_ID: ${{ steps.existing-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/export-whisper-model.yml b/.github/workflows/export-whisper-model.yml deleted file mode 100644 index 0f895e8..0000000 --- a/.github/workflows/export-whisper-model.yml +++ /dev/null @@ -1,280 +0,0 @@ -name: Export Whisper model - -on: - workflow_call: - inputs: - model_slug: - required: true - type: string - model_label: - required: true - type: string - model_id: - required: true - type: string - precision: - required: true - type: string - fallback_precision: - required: false - default: '' - type: string - attention: - required: false - default: sdpa - type: string - source_ref: - required: true - type: string - outputs: - artifact_name: - value: ${{ jobs.export.outputs.artifact_name }} - artifact_run_id: - value: ${{ jobs.export.outputs.artifact_run_id }} - -permissions: - actions: read - contents: read - -jobs: - export: - name: Export ${{ inputs.model_label }} ONNX - runs-on: ubuntu-24.04 - timeout-minutes: 240 - outputs: - artifact_name: ${{ steps.final-artifact.outputs.name }} - artifact_run_id: ${{ steps.final-artifact.outputs.run_id }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - fetch-depth: 0 - ref: ${{ inputs.source_ref }} - - - name: Determine export requirement - id: changes - shell: bash - env: - BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }} - EVENT_NAME: ${{ github.event_name }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -euo pipefail - if [[ "$EVENT_NAME" == "workflow_dispatch" ]]; then - force=false - elif [[ -z "$BASE_SHA" ]] || ! git cat-file -e "${BASE_SHA}^{commit}"; then - force=true - elif git diff --quiet "$BASE_SHA" "$SOURCE_REF" -- \ - 'asr/whisper/*.py' \ - 'asr/whisper/**/*.py' \ - '.github/workflows/export-whisper-model.yml' \ - '.github/workflows/whisper-asr.yml'; then - force=false - else - force=true - fi - echo "force=$force" >> "$GITHUB_OUTPUT" - - - name: Compute artifact names - id: artifact-names - shell: bash - env: - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_SLUG: ${{ inputs.model_slug }} - PRECISION: ${{ inputs.precision }} - run: | - set -euo pipefail - case "$PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported precision: $PRECISION" >&2; exit 2 ;; - esac - if [[ -n "$FALLBACK_PRECISION" ]]; then - case "$FALLBACK_PRECISION" in - fp16|fp32) ;; - *) echo "Unsupported fallback precision: $FALLBACK_PRECISION" >&2; exit 2 ;; - esac - fi - echo "primary=asr-${MODEL_SLUG}-onnx-${PRECISION}" >> "$GITHUB_OUTPUT" - if [[ -n "$FALLBACK_PRECISION" ]]; then - echo "fallback=asr-${MODEL_SLUG}-onnx-${FALLBACK_PRECISION}" >> "$GITHUB_OUTPUT" - fi - - - name: Find successful primary artifact - if: steps.changes.outputs.force != 'true' - id: primary-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.primary }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" - - - name: Find successful fallback artifact - if: >- - steps.changes.outputs.force != 'true' && - steps.primary-artifact.outputs.found != 'true' && - inputs.fallback_precision != '' - id: fallback-artifact - shell: bash - env: - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} - run: | - python .github/scripts/find_successful_artifact.py \ - --repository "$GITHUB_REPOSITORY" \ - --artifact-name "${{ steps.artifact-names.outputs.fallback }}" \ - --default-branch "$DEFAULT_BRANCH" \ - --event-name "$EVENT_NAME" \ - --source-branch "$SOURCE_BRANCH" - - - name: Select export action - id: selection - shell: bash - env: - FALLBACK_FOUND: ${{ steps.fallback-artifact.outputs.found }} - FALLBACK_NAME: ${{ steps.fallback-artifact.outputs.name }} - FALLBACK_RUN_ID: ${{ steps.fallback-artifact.outputs.run_id }} - FORCE_EXPORT: ${{ steps.changes.outputs.force }} - PRIMARY_FOUND: ${{ steps.primary-artifact.outputs.found }} - PRIMARY_NAME: ${{ steps.primary-artifact.outputs.name }} - PRIMARY_RUN_ID: ${{ steps.primary-artifact.outputs.run_id }} - run: | - set -euo pipefail - if [[ "$FORCE_EXPORT" == "true" ]]; then - echo "export=true" >> "$GITHUB_OUTPUT" - elif [[ "$PRIMARY_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$PRIMARY_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$PRIMARY_RUN_ID" >> "$GITHUB_OUTPUT" - elif [[ "$FALLBACK_FOUND" == "true" ]]; then - echo "export=false" >> "$GITHUB_OUTPUT" - echo "name=$FALLBACK_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$FALLBACK_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "export=true" >> "$GITHUB_OUTPUT" - fi - - - name: Reclaim runner disk space - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - df -h / - sudo rm -rf \ - /opt/ghc \ - /usr/local/.ghcup \ - /usr/local/lib/android \ - /usr/share/dotnet - df -h / - - - name: Set up Python - if: steps.selection.outputs.export == 'true' - uses: actions/setup-python@v7 - with: - python-version: '3.12' - - - name: Install Whisper export dependencies - if: steps.selection.outputs.export == 'true' - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip - python -m pip install --index-url https://download.pytorch.org/whl/cpu torch - python -m pip install onnx onnxscript safetensors transformers - - - name: Export Whisper ONNX - if: steps.selection.outputs.export == 'true' - id: exported - shell: bash - env: - ATTENTION: ${{ inputs.attention }} - FALLBACK_PRECISION: ${{ inputs.fallback_precision }} - MODEL_ID: ${{ inputs.model_id }} - MODEL_OUTPUT: artifacts/ci/${{ inputs.model_slug }} - MODEL_SLUG: ${{ inputs.model_slug }} - PRECISION: ${{ inputs.precision }} - SOURCE_REF: ${{ inputs.source_ref }} - run: | - set -uo pipefail - run_export() { - python asr/whisper/model_export/export_whisper.py \ - --model "$MODEL_ID" \ - --output "$MODEL_OUTPUT" \ - --device cpu \ - --dtype "$1" \ - --attention "$ATTENTION" - } - - selected_precision="$PRECISION" - set +e - run_export "$selected_precision" - export_status=$? - set -e - if [[ $export_status -ne 0 ]]; then - if [[ -z "$FALLBACK_PRECISION" ]]; then - exit "$export_status" - fi - echo "::warning::$PRECISION export failed; retrying once with $FALLBACK_PRECISION." - rm -rf "$MODEL_OUTPUT" - selected_precision="$FALLBACK_PRECISION" - run_export "$selected_precision" - fi - - for required_file in encoder.onnx decoder.onnx mel.onnx vocab.json; do - test -f "$MODEL_OUTPUT/$required_file" - done - - export MODEL_ID MODEL_OUTPUT SELECTED_PRECISION="$selected_precision" SOURCE_REF - python - <<'PY' - import json - import os - from pathlib import Path - - output = Path(os.environ["MODEL_OUTPUT"]) - manifest = { - "model_id": os.environ["MODEL_ID"], - "precision": os.environ["SELECTED_PRECISION"], - "commit": os.environ["SOURCE_REF"], - } - (output / "ci-export-manifest.json").write_text( - json.dumps(manifest, indent=2) + "\n", encoding="utf-8" - ) - PY - echo "name=asr-${MODEL_SLUG}-onnx-${selected_precision}" >> "$GITHUB_OUTPUT" - - - name: Upload Whisper ONNX - if: steps.selection.outputs.export == 'true' - uses: actions/upload-artifact@v7 - with: - name: ${{ steps.exported.outputs.name }} - path: artifacts/ci/${{ inputs.model_slug }}/ - if-no-files-found: error - compression-level: 0 - - - name: Select model artifact run - id: final-artifact - shell: bash - env: - EXISTING_NAME: ${{ steps.selection.outputs.name }} - EXISTING_RUN_ID: ${{ steps.selection.outputs.run_id }} - EXPORTED: ${{ steps.selection.outputs.export }} - EXPORTED_NAME: ${{ steps.exported.outputs.name }} - run: | - set -euo pipefail - if [[ "$EXPORTED" == "true" ]]; then - echo "name=$EXPORTED_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$GITHUB_RUN_ID" >> "$GITHUB_OUTPUT" - else - echo "name=$EXISTING_NAME" >> "$GITHUB_OUTPUT" - echo "run_id=$EXISTING_RUN_ID" >> "$GITHUB_OUTPUT" - fi diff --git a/.github/workflows/model-test.yml b/.github/workflows/model-test.yml new file mode 100644 index 0000000..85f2847 --- /dev/null +++ b/.github/workflows/model-test.yml @@ -0,0 +1,109 @@ +name: Model export and inference + +on: + workflow_call: + inputs: + model: + required: true + type: string + variant: + required: true + type: string + force_export: + required: false + default: false + type: boolean + +permissions: + contents: read + actions: read + +jobs: + export: + runs-on: ubuntu-24.04 + timeout-minutes: 240 + outputs: + name: ${{ steps.resolve.outputs.name || steps.export.outputs.name }} + run_id: ${{ steps.resolve.outputs.run_id || steps.export.outputs.run_id }} + env: + MODEL: ${{ inputs.model }} + VARIANT: ${{ inputs.variant }} + FORCE_EXPORT: ${{ inputs.force_export }} + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-python@v7 + with: + python-version: '3.12' + - name: Find a successful matching export + id: resolve + env: + GH_TOKEN: ${{ github.token }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} + SOURCE_BRANCH: ${{ github.head_ref || github.ref_name }} + PULL_REQUEST: ${{ github.event.pull_request.number }} + run: python .github/scripts/run_model_ci.py resolve + - name: Reclaim export disk space + if: steps.resolve.outputs.found != 'true' + run: sudo rm -rf /opt/ghc /usr/local/.ghcup /usr/local/lib/android /usr/share/dotnet + - name: Install dependencies and export model + if: steps.resolve.outputs.found != 'true' + id: export + run: python .github/scripts/run_model_ci.py export + - uses: actions/upload-artifact@v7 + if: steps.resolve.outputs.found != 'true' + with: + name: ${{ steps.export.outputs.name }} + path: artifacts/ci/${{ inputs.variant }}/ + if-no-files-found: error + compression-level: 0 + + smoke: + needs: export + name: ${{ matrix.platform }} / CPU + runs-on: ${{ matrix.runner }} + container: ${{ matrix.container }} + timeout-minutes: 120 + strategy: + fail-fast: false + matrix: + include: + - platform: linux-x64 + runner: ubuntu-24.04 + container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 + - platform: windows-x64 + runner: windows-2025 + container: '' + env: + MODEL: ${{ inputs.model }} + VARIANT: ${{ inputs.variant }} + steps: + - name: Install Python in Linux runtime container + if: runner.os == 'Linux' + run: apt-get update && apt-get install -y --no-install-recommends python3 + - uses: actions/checkout@v7 + - uses: actions/setup-python@v7 + if: runner.os == 'Windows' + with: + python-version: '3.12' + - name: Download native build from this run + uses: actions/download-artifact@v8 + with: + name: ${{ matrix.platform }}-release + path: runtime + - name: Download matching model export + uses: actions/download-artifact@v8 + with: + name: ${{ needs.export.outputs.name }} + path: artifacts/ci/${{ inputs.variant }}/ + github-token: ${{ github.token }} + repository: ${{ github.repository }} + run-id: ${{ needs.export.outputs.run_id }} + - name: Run model inference + shell: bash + run: | + if [ "$RUNNER_OS" = Linux ]; then + python3 .github/scripts/run_model_ci.py smoke + else + python .github/scripts/run_model_ci.py smoke + fi diff --git a/.github/workflows/nvidia-asr.yml b/.github/workflows/nvidia-asr.yml deleted file mode 100644 index 780708c..0000000 --- a/.github/workflows/nvidia-asr.yml +++ /dev/null @@ -1,220 +0,0 @@ -name: NVIDIA ASR - -on: - push: - branches: [main] - pull_request: - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: nvidia-asr-${{ github.ref }} - cancel-in-progress: true - -jobs: - requirements: - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export-parakeet: - name: Export Parakeet TDT ONNX - needs: requirements - uses: ./.github/workflows/export-nvidia-asr-model.yml - with: - model_slug: parakeet-tdt - model_label: Parakeet TDT - model_id: nvidia/parakeet-tdt-0.6b-v3 - model_type: parakeet - precision: fp16 - fallback_precision: fp32 - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-nemotron: - name: Export Nemotron 3.5 ASR Streaming ONNX - needs: requirements - uses: ./.github/workflows/export-nvidia-asr-model.yml - with: - model_slug: nemotron-3.5-asr-streaming-0.6b - model_label: Nemotron 3.5 ASR Streaming 0.6B - model_id: nvidia/nemotron-3.5-asr-streaming-0.6b - model_type: nemotron - precision: fp16 - fallback_precision: fp32 - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Collect NVIDIA ASR artifacts - needs: [export-parakeet, export-nemotron] - runs-on: ubuntu-24.04 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - - steps: - - name: Build runtime matrix - id: matrix - shell: bash - env: - NEMOTRON_NAME: ${{ needs.export-nemotron.outputs.artifact_name }} - NEMOTRON_RUN_ID: ${{ needs.export-nemotron.outputs.artifact_run_id }} - PARAKEET_NAME: ${{ needs.export-parakeet.outputs.artifact_name }} - PARAKEET_RUN_ID: ${{ needs.export-parakeet.outputs.artifact_run_id }} - run: | - python - <<'PY' - import json - import os - from pathlib import Path - - matrix = { - "include": [ - { - "slug": "parakeet-tdt", - "label": "Parakeet TDT", - "executable": "din_asr_parakeet_tdt_cli", - "artifact_name": os.environ["PARAKEET_NAME"], - "artifact_run_id": os.environ["PARAKEET_RUN_ID"], - }, - { - "slug": "nemotron-3.5-asr-streaming-0.6b", - "label": "Nemotron 3.5 ASR Streaming 0.6B", - "executable": "din_asr_nemotron_cli", - "artifact_name": os.environ["NEMOTRON_NAME"], - "artifact_run_id": os.environ["NEMOTRON_RUN_ID"], - }, - ] - } - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - output.write(f"matrix={json.dumps(matrix, separators=(',', ':'))}\n") - PY - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - chmod +x "runtime/${{ matrix.executable }}" - "runtime/${{ matrix.executable }}" \ - assets/sample.wav \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - & "runtime/${{ matrix.executable }}.exe" ` - "assets/sample.wav" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/.github/workflows/sam2.yml b/.github/workflows/sam2.yml deleted file mode 100644 index cb12fa8..0000000 --- a/.github/workflows/sam2.yml +++ /dev/null @@ -1,252 +0,0 @@ -name: SAM2 - -on: - push: - branches: [main] - pull_request: - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: sam2-${{ github.ref }} - cancel-in-progress: true - -jobs: - requirements: - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export: - name: Export ${{ matrix.label }} ONNX - needs: requirements - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - model_id: facebook/sam2.1-hiera-tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - model_id: facebook/sam2.1-hiera-small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - model_id: facebook/sam2.1-hiera-base-plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - model_id: facebook/sam2.1-hiera-large - uses: ./.github/workflows/export-sam2-model.yml - with: - model_slug: ${{ matrix.slug }} - model_label: ${{ matrix.label }} - model_id: ${{ matrix.model_id }} - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Resolve SAM2 artifacts - needs: [requirements, export] - runs-on: ubuntu-24.04 - outputs: - models: ${{ steps.resolve.outputs.models }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Resolve model artifacts - id: resolve - shell: bash - env: - GH_TOKEN: ${{ github.token }} - run: | - set -euo pipefail - source_hash="$( - git ls-files -z -- 'vision/sam2/python/*.py' 'vision/sam2/python/**/*.py' \ - | xargs -0 sha256sum \ - | sha256sum \ - | cut -d' ' -f1 - )" - models='{}' - for slug in \ - sam2.1-hiera-tiny \ - sam2.1-hiera-small \ - sam2.1-hiera-base-plus \ - sam2.1-hiera-large; do - artifact_name="vision-${slug}-onnx-v1-${source_hash:0:16}" - artifact_run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/artifacts?name=${artifact_name}&per_page=100" \ - --jq '.artifacts | map(select(.expired | not)) | sort_by(.created_at) | last | .workflow_run.id // empty' - )" - test -n "$artifact_run_id" - models="$(jq \ - --arg slug "$slug" \ - --arg name "$artifact_name" \ - --arg run_id "$artifact_run_id" \ - '. + {($slug): {name: $name, run_id: $run_id}}' \ - <<< "$models")" - done - echo "models=$(jq -c . <<< "$models")" >> "$GITHUB_OUTPUT" - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - base64 --decode assets/sam2-ci.png.base64 > assets/sam2-ci.png - mkdir -p "out/${{ matrix.slug }}" - chmod +x runtime/din_sam2_cli - runtime/din_sam2_cli \ - assets/sam2-ci.png \ - assets/sam2-ci-prompt.json \ - "out/${{ matrix.slug }}" \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: - include: - - slug: sam2.1-hiera-tiny - label: SAM2.1 Hiera Tiny - - slug: sam2.1-hiera-small - label: SAM2.1 Hiera Small - - slug: sam2.1-hiera-base-plus - label: SAM2.1 Hiera Base Plus - - slug: sam2.1-hiera-large - label: SAM2.1 Hiera Large - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ fromJSON(needs.artifacts.outputs.models)[matrix.slug].run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - $imageBytes = [Convert]::FromBase64String( - (Get-Content -LiteralPath "assets/sam2-ci.png.base64" -Raw).Trim() - ) - [IO.File]::WriteAllBytes("assets/sam2-ci.png", $imageBytes) - New-Item -ItemType Directory -Force -Path "out/${{ matrix.slug }}" | Out-Null - & "runtime/din_sam2_cli.exe" ` - "assets/sam2-ci.png" ` - "assets/sam2-ci-prompt.json" ` - "out/${{ matrix.slug }}" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/.github/workflows/whisper-asr.yml b/.github/workflows/whisper-asr.yml deleted file mode 100644 index 6d7d64f..0000000 --- a/.github/workflows/whisper-asr.yml +++ /dev/null @@ -1,285 +0,0 @@ -name: Whisper ASR - -on: - push: - branches: [main] - pull_request: - workflow_dispatch: - inputs: - build_run_id: - description: Successful Build workflow run ID whose artifacts should be tested - required: true - type: string - -permissions: - actions: read - contents: read - -concurrency: - group: whisper-asr-${{ github.ref }} - cancel-in-progress: true - -jobs: - requirements: - name: Wait for Build - runs-on: ubuntu-24.04 - timeout-minutes: 100 - outputs: - build_run_id: ${{ steps.build.outputs.run_id }} - source_ref: ${{ steps.build.outputs.source_ref }} - - steps: - - name: Resolve successful Build run - id: build - shell: bash - env: - EVENT_NAME: ${{ github.event_name }} - GH_TOKEN: ${{ github.token }} - REQUESTED_RUN_ID: ${{ inputs.build_run_id }} - SOURCE_SHA: ${{ github.event.pull_request.head.sha || github.sha }} - run: | - set -euo pipefail - run_id="$REQUESTED_RUN_ID" - while [[ -z "$run_id" ]]; do - run_id="$( - gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/ci.yml/runs?per_page=100" \ - --jq ".workflow_runs | map(select(.head_sha == \"$SOURCE_SHA\" and .event == \"$EVENT_NAME\")) | sort_by(.created_at) | last | .id // empty" - )" - [[ -n "$run_id" ]] || sleep 10 - done - while true; do - status="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .status)" - [[ "$status" == "completed" ]] && break - sleep 15 - done - conclusion="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .conclusion)" - if [[ "$conclusion" != "success" ]]; then - echo "Build run ${run_id} concluded ${conclusion}" >&2 - exit 1 - fi - echo "run_id=$run_id" >> "$GITHUB_OUTPUT" - source_ref="$(gh api "repos/${GITHUB_REPOSITORY}/actions/runs/${run_id}" --jq .head_sha)" - echo "source_ref=$source_ref" >> "$GITHUB_OUTPUT" - - export-tiny: - name: Export Whisper Tiny ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-tiny - model_label: Whisper Tiny - model_id: openai/whisper-tiny - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-base: - name: Export Whisper Base ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-base - model_label: Whisper Base - model_id: openai/whisper-base - precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-small: - name: Export Whisper Small ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-small - model_label: Whisper Small - model_id: openai/whisper-small - precision: fp16 - fallback_precision: fp32 - attention: math - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-medium: - name: Export Whisper Medium ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-medium - model_label: Whisper Medium - model_id: openai/whisper-medium - precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-large-v3: - name: Export Whisper Large v3 ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-large-v3 - model_label: Whisper Large v3 - model_id: openai/whisper-large-v3 - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - export-large-v3-turbo: - name: Export Whisper Large v3 Turbo ONNX - needs: requirements - uses: ./.github/workflows/export-whisper-model.yml - with: - model_slug: whisper-large-v3-turbo - model_label: Whisper Large v3 Turbo - model_id: openai/whisper-large-v3-turbo - precision: fp16 - fallback_precision: fp32 - attention: sdpa - source_ref: ${{ needs.requirements.outputs.source_ref }} - - artifacts: - name: Collect Whisper artifacts - needs: - - export-tiny - - export-base - - export-small - - export-medium - - export-large-v3 - - export-large-v3-turbo - runs-on: ubuntu-24.04 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - - steps: - - name: Build runtime matrix - id: matrix - shell: bash - env: - BASE_NAME: ${{ needs.export-base.outputs.artifact_name }} - BASE_RUN_ID: ${{ needs.export-base.outputs.artifact_run_id }} - LARGE_NAME: ${{ needs.export-large-v3.outputs.artifact_name }} - LARGE_RUN_ID: ${{ needs.export-large-v3.outputs.artifact_run_id }} - MEDIUM_NAME: ${{ needs.export-medium.outputs.artifact_name }} - MEDIUM_RUN_ID: ${{ needs.export-medium.outputs.artifact_run_id }} - SMALL_NAME: ${{ needs.export-small.outputs.artifact_name }} - SMALL_RUN_ID: ${{ needs.export-small.outputs.artifact_run_id }} - TINY_NAME: ${{ needs.export-tiny.outputs.artifact_name }} - TINY_RUN_ID: ${{ needs.export-tiny.outputs.artifact_run_id }} - TURBO_NAME: ${{ needs.export-large-v3-turbo.outputs.artifact_name }} - TURBO_RUN_ID: ${{ needs.export-large-v3-turbo.outputs.artifact_run_id }} - run: | - python - <<'PY' - import json - import os - from pathlib import Path - - models = [ - ("whisper-tiny", "Whisper Tiny", "TINY"), - ("whisper-base", "Whisper Base", "BASE"), - ("whisper-small", "Whisper Small", "SMALL"), - ("whisper-medium", "Whisper Medium", "MEDIUM"), - ("whisper-large-v3", "Whisper Large v3", "LARGE"), - ("whisper-large-v3-turbo", "Whisper Large v3 Turbo", "TURBO"), - ] - matrix = { - "include": [ - { - "slug": slug, - "label": label, - "artifact_name": os.environ[f"{key}_NAME"], - "artifact_run_id": os.environ[f"{key}_RUN_ID"], - } - for slug, label, key in models - ] - } - with Path(os.environ["GITHUB_OUTPUT"]).open("a", encoding="utf-8") as output: - output.write(f"matrix={json.dumps(matrix, separators=(',', ':'))}\n") - PY - - linux-runtime: - name: Linux x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: ubuntu-24.04 - container: nvidia/cuda:13.2.1-runtime-ubuntu24.04 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Linux build artifact - uses: actions/download-artifact@v8 - with: - name: linux-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: bash - env: - LD_LIBRARY_PATH: ${{ github.workspace }}/runtime - run: | - set -euxo pipefail - chmod +x runtime/din_asr_whisper_cli - runtime/din_asr_whisper_cli \ - assets/sample.wav \ - --model-dir "models/${{ matrix.slug }}" \ - --provider cpu - - windows-runtime: - name: Windows x64 / CPU / ${{ matrix.label }} - needs: [requirements, artifacts] - runs-on: windows-2025 - timeout-minutes: 120 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.artifacts.outputs.matrix) }} - - steps: - - name: Check out source - uses: actions/checkout@v7 - with: - ref: ${{ needs.requirements.outputs.source_ref }} - - - name: Download Windows build artifact - uses: actions/download-artifact@v8 - with: - name: windows-x64-release - path: runtime - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ needs.requirements.outputs.build_run_id }} - - - name: Download model artifact - uses: actions/download-artifact@v8 - with: - name: ${{ matrix.artifact_name }} - path: models/${{ matrix.slug }} - github-token: ${{ github.token }} - repository: ${{ github.repository }} - run-id: ${{ matrix.artifact_run_id }} - - - name: Run C++ CPU smoke test - shell: pwsh - run: | - $ErrorActionPreference = "Stop" - & "runtime/din_asr_whisper_cli.exe" ` - "assets/sample.wav" ` - --model-dir "models/${{ matrix.slug }}" ` - --provider cpu diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..b1f650b --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,83 @@ +# DIN Deploy: repository guidance for coding agents + +## Goal + +Build practical, high-performance local inference samples that others can understand, +reuse, and integrate into commercial applications. Keep this a sample repository, +not a framework: as small as possible and as complex as necessary, without sacrificing +correctness, functionality, or performance. + +Apply these rules to new and changed code; do not rewrite unrelated samples to enforce them. + +## Portable models and execution providers + +- Export standard ONNX operators. Do not introduce ONNX Runtime contrib operators + (for example `com.microsoft::*`), ATen fallbacks, custom operator libraries, or + vendor-specific graph operators. Use standard ONNX decompositions instead. +- Inspect exported graphs, including subgraphs and local functions, for nonstandard + operators; ONNX checker success alone does not demonstrate EP compatibility. +- Keep graphs and core algorithms EP-independent; EP-internal fusion is allowed. + Maintain a usable CPU path and isolate optional backend-specific optimizations. +- Target Windows and Linux on x64 and ARM64. Hardware acceleration depends on the + device and driver, not just the OS/CPU architecture; identify those requirements. +- Validate supported EPs and precisions explicitly; standard ONNX does not guarantee + kernel support. Do not silently move expensive operations to CPU or claim that + accelerator-only precisions work there. Provide a compatible configuration or + explain the limitation. + +## Performance and memory ownership + +- Optimize measured bottlenecks in latency, throughput, memory, and startup/compilation. +- Keep intermediate tensors on their device through preprocessing and inference. + Use device-buffer binding and reusable allocations where supported. Avoid host + round trips and per-inference allocation in hot paths. +- Make memory location, ownership, layout/stride, and lifetime explicit at API + boundaries. Keep buffers alive until asynchronous consumers finish. Bound queues + and caches. +- Reuse the application's CUDA context for cooperating components where supported. + Do not create extra contexts or processes casually; document any required boundary. +- Prefer stream/event dependencies over CPU waits and device-wide synchronization. + Preserve producer/consumer dependencies when removing waits. +- GPU-resident is not synonymous with zero-copy or wait-free. Account for device + copies and preprocessing when evaluating performance. + +## Minimal code and useful reuse + +- Reuse existing export, runtime, and I/O helpers. Extract small shared functions + for concrete repetition; keep model-specific behavior near its model. Avoid + speculative abstractions, registries, and frameworks for hypothetical reuse. +- Simplify control flow and remove obsolete code. Do not shorten code at the expense + of readability, buffer reuse, or asynchronous execution. +- Follow surrounding conventions and repository formatting, including `.clang-format`. +- Preserve error causes; do not present failures or unsupported settings as success. + +## Licensing and commercial reuse + +- Keep code contributions compatible with this repository's Apache-2.0 license and + commercial reuse. Prefer permissive dependencies with clear redistribution terms. +- Review licenses for new code, transitive dependencies, model weights, and bundled + assets. Do not assume that open source or public availability permits every use. +- Discuss downstream impact with the user before introducing non-commercial + restrictions or copyleft obligations. Flag unclear terms; do not claim unverified + commercial compatibility or assume model weights share the code's license. +- Preserve required attribution and license notices, and document redistribution + obligations. Consider relevant patent and SDK terms separately from code licenses. + +## Validation and working practices + +- Read the relevant README, build configuration, and CONTRIBUTING.md. Preserve + unrelated user/agent changes. Use the existing project environment and build paths. +- Run focused validation: export parity and graph checks for model changes, + numerical/task-quality checks for inference changes, and regression checks for + state, timing, cancellation, and ownership bugs. +- Benchmark performance changes before/after with matching hardware, EP, precision, + input, and warmup. Measure completed GPU work, not just submission time. Inspect + copies/waits when claiming to reduce them; a smoke test is not performance evidence. +- Separate cold startup from steady-state timings. End-to-end timings include all + enabled steps. Label model-only timings separately and state the RTF convention. +- Build/test the affected available targets. Report untested platform/EP combinations + and unavailable prerequisites honestly. Do not add redundant tests or expand + testing without a concrete reason. +- Update nearby documentation when behavior, requirements, or commands change. + Report what changed, what was verified, and material limitations concisely. +- Commit/push only when requested; follow CONTRIBUTING.md's sign-off requirement. diff --git a/docs/model-ci.md b/docs/model-ci.md new file mode 100644 index 0000000..057eaa3 --- /dev/null +++ b/docs/model-ci.md @@ -0,0 +1,41 @@ +# Model CI + +Model recipes in `.github/model-tests/*.json` define exports, CMake targets, and +CPU smoke tests. Optional dependency rules in `.github/model-ci.json` determine +which models need to run. Adding a model requires no workflow changes. + +## What runs + +| Change | CI behavior | +| --- | --- | +| Exporter or export dependencies | Build and test affected models; reuse only matching exports | +| Runtime or test inputs | Build and test affected models using cached exports | +| Shared code | Run all affected consumers | +| Documentation or `.coderabbit.yaml` only | Lightweight checks only | +| Unknown files or CI control logic | Full build and all model recipes | + +Draft PRs skip CI. Missing dependency rules make a model run conservatively; +missing export rules prevent reuse across runs. Missing or expired artifacts are +re-exported. Unknown files also force fresh exports. + +Unregistered CMake executables trigger a full build and all available recipes; +new inference tests still need an execution recipe. Linux and Windows x64 run +CPU smoke tests; ARM64, FLUX, and the base ONNX sample retain build-only coverage. + +## Adding a model + +1. Add its CMake target and exporter. +2. Copy an existing `.github/model-tests/.json` recipe and adapt its + variants, commands, and required outputs. The first target is the inference CLI. +3. Optionally declare export and runtime/test inputs in `.github/model-ci.json`, + including shared dependencies. Without these rules, the recipe always runs on + non-exempt changes. Patterns use `fnmatch`; `*` spans directories. +4. Run `python -m unittest discover -s .github/tests -v`, then verify the CI + selection summary and model results on both runtime platforms. + +Exports are fingerprinted from their declared inputs, recipe, and model options. +Bump `export.cache_epoch` to refresh floating upstream dependencies or weights; +pin versions when reproducibility is required. + +When migrating, update required-check rules that reference the old standalone +model workflows. The first run creates exports under the new cache keys.