diff --git a/.github/workflows/onnx-release.yml b/.github/workflows/onnx-release.yml index 4ba0e70..fd5252b 100644 --- a/.github/workflows/onnx-release.yml +++ b/.github/workflows/onnx-release.yml @@ -12,11 +12,21 @@ on: required: false default: true type: boolean + evidence_run_id: + description: "Workflow run ID containing vision-v8-quantized-release-inputs" + required: false + type: string + runner: + description: "Runner label with CUDA support for FP16 ONNX gates" + required: false + default: "self-hosted" + type: string push: tags: - "onnx-v8-*" permissions: + actions: read contents: write concurrency: @@ -25,7 +35,7 @@ concurrency: jobs: publish: - runs-on: ubuntu-latest + runs-on: ${{ inputs.runner || 'self-hosted' }} # Two 640px exports plus 50-seed parity gates on both branches. timeout-minutes: 60 steps: @@ -55,10 +65,26 @@ jobs: --index-url https://download.pytorch.org/whl/cpu python -m pip install \ "onnx==${{ steps.pins.outputs.onnx }}" \ - "onnxruntime==${{ steps.pins.outputs.onnxruntime }}" + "onnxruntime-gpu==${{ steps.pins.outputs.onnxruntime }}" \ + psutil + + - name: Download quantized release evidence + env: + GH_TOKEN: ${{ github.token }} + EVIDENCE_RUN_ID: ${{ github.event.inputs.evidence_run_id || vars.ONNX_RELEASE_EVIDENCE_RUN_ID }} + run: | + test -n "$EVIDENCE_RUN_ID" + mkdir -p artifacts/vision_v8_quantized_eval + gh run download "$EVIDENCE_RUN_ID" \ + --name vision-v8-quantized-release-inputs \ + --dir artifacts/vision_v8_quantized_eval + test -f artifacts/vision_v8_quantized_eval/calibration.json + test -d artifacts/vision_v8_quantized_eval/calibration_images + test -f artifacts/vision_v8_quantized_eval/accuracy.json + test -f artifacts/vision_v8_quantized_eval/accuracy.md - name: Test the release tooling - run: pytest -q tests/test_onnx_release.py + run: pytest -q tests/test_onnx_release.py tests/test_onnx_quantization_benchmark.py - name: Build, gate and verify the release env: @@ -104,9 +130,5 @@ jobs: --notes-file dist/onnx/RELEASE_NOTES.md \ $DRAFT gh release upload "$TAG" \ - dist/onnx/manifest.json \ - dist/onnx/tr_hash_v8_o2m.onnx \ - dist/onnx/tr_hash_v8_o2m.json \ - dist/onnx/tr_hash_v8_nms_free.onnx \ - dist/onnx/tr_hash_v8_nms_free.json \ + dist/onnx/* \ --clobber diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 6adbb62..d44324d 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -47,6 +47,33 @@ on: onnx_nms_free_metadata: description: "Local path to the NMS-free Vision v8 ONNX metadata sidecar on the runner" required: false + onnx_o2m_fp16_model: + description: "Local path to the FP16 O2M Vision v8 ONNX model on the runner" + required: false + onnx_o2m_fp16_metadata: + description: "Local path to the FP16 O2M Vision v8 ONNX metadata sidecar" + required: false + onnx_nms_free_fp16_model: + description: "Local path to the FP16 NMS-free Vision v8 ONNX model" + required: false + onnx_nms_free_fp16_metadata: + description: "Local path to the FP16 NMS-free Vision v8 ONNX metadata sidecar" + required: false + onnx_o2m_int8_model: + description: "Local path to the INT8 O2M Vision v8 ONNX model on the runner" + required: false + onnx_o2m_int8_metadata: + description: "Local path to the INT8 O2M Vision v8 ONNX metadata sidecar" + required: false + onnx_nms_free_int8_model: + description: "Local path to the INT8 NMS-free Vision v8 ONNX model" + required: false + onnx_nms_free_int8_metadata: + description: "Local path to the INT8 NMS-free Vision v8 ONNX metadata sidecar" + required: false + calibration_manifest: + description: "Local path to the quantization calibration manifest" + required: false annotations: description: "Local path to instances_val2017.json on the runner" required: true @@ -66,7 +93,19 @@ on: - pytorch - onnx provider: - description: "ONNX Runtime provider alias for backend=onnx" + description: "Legacy ONNX provider alias; precision-specific provider inputs are used for release evidence" + required: false + default: "cpu" + fp32_provider: + description: "ONNX Runtime provider alias for FP32 evaluation" + required: true + default: "cpu" + fp16_provider: + description: "ONNX Runtime provider alias for FP16 evaluation" + required: true + default: "cuda" + int8_provider: + description: "ONNX Runtime provider alias for INT8 evaluation" required: true default: "cpu" release_tag: @@ -136,6 +175,8 @@ jobs: run: | python -m pip install --upgrade pip python -m pip install -e ".[detection,export]" + ONNXRUNTIME_VERSION=$(python -c "import json; print(json.load(open('docs/onnx/release.json'))['toolchain']['onnxruntime'])") + python -m pip install "onnxruntime-gpu==$ONNXRUNTIME_VERSION" - name: Run deterministic COCO evaluation twice run: | if [ "${{ inputs.backend }}" = "pytorch" ]; then @@ -147,11 +188,11 @@ jobs: test -n "${{ inputs.onnx_o2m_metadata }}" test -n "${{ inputs.onnx_nms_free_model }}" test -n "${{ inputs.onnx_nms_free_metadata }}" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b fi - name: Gate full COCO report @@ -160,6 +201,35 @@ jobs: artifacts/vision_v8_coco_eval/run_a/evaluation.json --repeat-report artifacts/vision_v8_coco_eval/run_b/evaluation.json --config configs/vision_v8_coco_accuracy_gate.json + - name: Build quantized release evidence artifact + if: inputs.backend == 'onnx' && inputs.calibration_manifest != '' + env: + CALIBRATION_MANIFEST: ${{ inputs.calibration_manifest }} + run: | + test -n "${{ inputs.onnx_o2m_fp16_model }}" + test -n "${{ inputs.onnx_o2m_fp16_metadata }}" + test -n "${{ inputs.onnx_nms_free_fp16_model }}" + test -n "${{ inputs.onnx_nms_free_fp16_metadata }}" + test -n "${{ inputs.onnx_o2m_int8_model }}" + test -n "${{ inputs.onnx_o2m_int8_metadata }}" + test -n "${{ inputs.onnx_nms_free_int8_model }}" + test -n "${{ inputs.onnx_nms_free_int8_metadata }}" + mkdir -p artifacts/vision_v8_quantized_eval + python -c "import os; from pathlib import Path; from scripts.check_onnx_quantized_artifacts import materialize_calibration_manifest_images; materialize_calibration_manifest_images(Path(os.environ['CALIBRATION_MANIFEST']), artifact_root=Path('artifacts/vision_v8_quantized_eval'), output_manifest_path=Path('artifacts/vision_v8_quantized_eval/calibration.json'))" + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_fp16_model }}" --metadata "${{ inputs.onnx_o2m_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_fp16 --branch o2m-nms --provider "${{ inputs.fp16_provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_fp16_model }}" --metadata "${{ inputs.onnx_nms_free_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_fp16 --branch nms-free --provider "${{ inputs.fp16_provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_int8_model }}" --metadata "${{ inputs.onnx_o2m_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_int8 --branch o2m-nms --provider "${{ inputs.int8_provider }}" --precision int8 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_int8_model }}" --metadata "${{ inputs.onnx_nms_free_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_int8 --branch nms-free --provider "${{ inputs.int8_provider }}" --precision int8 --seed 0 + python scripts/merge_vision_v8_coco_reports.py \ + artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json \ + artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_int8/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_int8/evaluation.json \ + --output artifacts/vision_v8_quantized_eval + cp artifacts/vision_v8_quantized_eval/evaluation.json artifacts/vision_v8_quantized_eval/accuracy.json + cp artifacts/vision_v8_quantized_eval/evaluation.md artifacts/vision_v8_quantized_eval/accuracy.md - uses: actions/upload-artifact@v4 with: name: vision-v8-coco-accuracy-report @@ -168,6 +238,15 @@ jobs: artifacts/vision_v8_coco_eval/run_a/evaluation.md artifacts/vision_v8_coco_eval/run_b/evaluation.json artifacts/vision_v8_coco_eval/run_b/evaluation.md + - uses: actions/upload-artifact@v4 + if: inputs.backend == 'onnx' && inputs.calibration_manifest != '' + with: + name: vision-v8-quantized-release-inputs + path: | + artifacts/vision_v8_quantized_eval/calibration.json + artifacts/vision_v8_quantized_eval/calibration_images/** + artifacts/vision_v8_quantized_eval/accuracy.json + artifacts/vision_v8_quantized_eval/accuracy.md - name: Upload gated reports to GitHub Release if: inputs.release_tag != '' env: diff --git a/complexity/deploy/onnx_detector/pipeline.py b/complexity/deploy/onnx_detector/pipeline.py index ac84484..037e531 100644 --- a/complexity/deploy/onnx_detector/pipeline.py +++ b/complexity/deploy/onnx_detector/pipeline.py @@ -39,9 +39,19 @@ def from_files( model_path: str | Path, metadata_path: str | Path, providers: Sequence[str] = ("CPUExecutionProvider",), + *, + warmup_runs: int = 1, + intra_op_num_threads: int | None = None, + inter_op_num_threads: int | None = None, ) -> "OnnxDetectorPipeline": metadata = load_metadata(metadata_path) - session = cls.create_session(model_path, providers).open() + session = cls.create_session( + model_path, + providers, + warmup_runs=warmup_runs, + intra_op_num_threads=intra_op_num_threads, + inter_op_num_threads=inter_op_num_threads, + ).open() validate_output_shape(metadata, session._require_session().get_outputs()[0].shape) if session.config.warmup_runs: session.warmup((1, 3, metadata.image_size, metadata.image_size)) @@ -74,13 +84,11 @@ def postprocess_single_image( scores: np.ndarray, classes: np.ndarray, ) -> tuple[np.ndarray, np.ndarray, np.ndarray]: - filtered_boxes, filtered_scores, filtered_classes, _ = ( - postprocess.filter_by_confidence( - boxes, - scores, - classes, - self.metadata.confidence_threshold, - ) + filtered_boxes, filtered_scores, filtered_classes, _ = postprocess.filter_by_confidence( + boxes, + scores, + classes, + self.metadata.confidence_threshold, ) if self.metadata.branch == "o2m": keep = postprocess.class_aware_nms( diff --git a/configs/vision_v8_quantization_calibration.example.json b/configs/vision_v8_quantization_calibration.example.json new file mode 100644 index 0000000..e6fb9c0 --- /dev/null +++ b/configs/vision_v8_quantization_calibration.example.json @@ -0,0 +1,19 @@ +{ + "schema_version": 1, + "dataset": { + "name": "coco-2017-train-calibration", + "split": "train2017-calibration", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "replace-with-annotations-sha256", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 1 + } +} diff --git a/configs/vision_v8_quantization_thresholds.json b/configs/vision_v8_quantization_thresholds.json new file mode 100644 index 0000000..5944cf6 --- /dev/null +++ b/configs/vision_v8_quantization_thresholds.json @@ -0,0 +1,49 @@ +{ + "schema_version": 1, + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "require_artifact_bindings": true, + "provider_by_precision": { + "fp32": "CPUExecutionProvider", + "fp16": "CUDAExecutionProvider", + "int8": "CPUExecutionProvider" + }, + "partial_release": "block", + "optional_provider_precisions": [] + }, + "precisions": { + "fp16": { + "max_raw_logit_abs_error": 0.01, + "max_decoded_box_px_error": 1.0, + "max_score_abs_error": 0.01, + "max_map50_95_drop": 0.005, + "max_map50_drop": 0.01, + "unexpected_fp32_nodes": "fail" + }, + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + "max_map50_95_drop": 0.02, + "max_map50_drop": 0.03, + "unexpected_fp32_nodes": "allow" + } + }, + "providers": { + "CPUExecutionProvider": ["fp32", "int8"], + "CUDAExecutionProvider": ["fp32", "fp16"], + "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] + }, + "benchmark": { + "warmup_iterations": 25, + "measured_iterations": 100, + "report": [ + "median_ms", + "mean_ms", + "stddev_ms", + "p95_ms", + "throughput_images_per_second", + "peak_memory_mb" + ] + } +} diff --git a/docs/index.md b/docs/index.md index 133d9db..c5ecf1c 100644 --- a/docs/index.md +++ b/docs/index.md @@ -92,6 +92,7 @@ GQA + TR-MoE architecture. - [TR-Hash object detection and serving](tr-hash-object-detection.md) - [TR-Hash Vision ONNX deployment](onnx_deploy.md) - [Vision v8 COCO accuracy gates](vision-v8-coco-accuracy-gates.md) +- [Vision v8 ONNX quantization](vision-v8-onnx-quantization.md) - [Detector specialization and ablations](TR_HASH_DETECTOR_SPECIALIZATION.md) - [TR-Hash sensor fusion](tr_hash_sensor_fusion.md) - [Vision dependency stack](vision-dependency-stack.md) diff --git a/docs/onnx/release.json b/docs/onnx/release.json index 77f6c2d..e64308f 100644 --- a/docs/onnx/release.json +++ b/docs/onnx/release.json @@ -3,10 +3,23 @@ "checkpoint_revision": "f3b3e659612e543ca9ff91892c0662d38dc1a1d6", "opset": 17, "parity_num_tests": 50, + "quantization": { + "enabled_precisions": ["fp16", "int8"], + "calibration_manifest": "artifacts/vision_v8_quantized_eval/calibration.json", + "thresholds": "configs/vision_v8_quantization_thresholds.json", + "accuracy_report": "artifacts/vision_v8_quantized_eval/accuracy.json", + "accuracy_markdown": "artifacts/vision_v8_quantized_eval/accuracy.md", + "fp32_op_allowlist": [], + "provider_gates": [ + {"provider": "CPUExecutionProvider", "precision": "fp32"}, + {"provider": "CUDAExecutionProvider", "precision": "fp16"}, + {"provider": "CPUExecutionProvider", "precision": "int8"} + ] + }, "toolchain": { "torch": "2.13.0", "onnx": "1.21.0", - "onnxruntime": "1.24.4" + "onnxruntime": "1.23.2" }, "branches": [ { diff --git a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md new file mode 100644 index 0000000..f8fe0f3 --- /dev/null +++ b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md @@ -0,0 +1,818 @@ +# Vision v8 ONNX Quantization Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Build reproducible FP16 and INT8 ONNX artifacts for both Vision v8 detector branches and gate them against FP32 accuracy, decoded detections, performance, and release metadata. + +**Architecture:** Quantization is implemented as a post-export pipeline layered on top of the existing Vision v8 ONNX export, ONNX detector runtime, COCO evaluation, and release workflow. FP32 remains the reference; FP16 and INT8 artifacts are generated deterministically, validated for provider support and node dtypes, evaluated through the shared ONNX COCO evaluator, and published only after release gates pass. + +**Tech Stack:** Python, ONNX, ONNX Runtime quantization tools, ONNX Runtime providers, PyTorch, COCO evaluation helpers, GitHub Actions. + +**Spec:** User issue: “The published Vision v8 ONNX artifacts are FP32 only. We need smaller and faster deployment artifacts without trading away decoded detections or COCO accuracy silently.” + +## Global Constraints + +- Quantization commands must be deterministic from a pinned checkpoint, FP32 ONNX model, framework commit, calibration manifest, and quantization settings. +- Calibration inputs must be pinned and disjoint from the COCO evaluation inputs used by the accuracy gate. +- Unsupported provider/precision combinations must fail clearly instead of falling back silently. +- FP16 conversion must report remaining FP32 nodes and fail unless every remaining FP32 node is covered by a checked-in allowlist. +- INT8 settings must pin calibration method, per-channel/per-tensor mode, symmetric/asymmetric mode, activation type, weight type, and calibration batch size. +- FP32, FP16, and INT8 reports must compare artifact size, latency, throughput, peak memory, raw-logit parity, decoded-output parity, and COCO metrics. +- Tolerances and floors must live in checked-in config, not inside scripts. +- If any required quantized artifact fails its gate, the release blocks; do not publish a partial FP32+FP16-only release unless the release config explicitly marks INT8 optional for that provider. +- ONNX binaries must remain excluded from normal source commits and be published as release assets. + +--- + +## File Structure + +- Create `configs/vision_v8_quantization_thresholds.json` for FP16/INT8 raw-logit, decoded-output, COCO, performance, dtype, and provider thresholds. +- Create `configs/vision_v8_quantization_calibration.example.json` for pinned calibration input metadata and quantization settings. +- Create `scripts/quantize_onnx.py` to generate FP16 and INT8 ONNX files plus JSON sidecars. +- Create `scripts/check_onnx_quantized_artifacts.py` to validate metadata, checksums, node dtypes, provider support, calibration/eval disjointness, parity reports, COCO reports, and performance reports. +- Create `scripts/benchmark_onnx_artifacts.py` to measure artifact latency, throughput, and peak memory with stable warm-up and measured iteration counts. +- Modify `scripts/check_onnx_parity.py` only if needed to expose reusable raw-logit and decoded-output comparison helpers. +- Reuse `scripts/evaluate_onnx_coco.py` from the COCO accuracy gate work for all ONNX COCO AP metrics. +- Modify `.github/workflows/detector-export.yml` or add `.github/workflows/vision-v8-onnx-quantization-release.yml` to generate, gate, and upload FP32/FP16/INT8 release assets. +- Create `docs/vision-v8-onnx-quantization.md` for reproduction commands, supported providers, precision limitations, and release policy. +- Add tests under `tests/test_onnx_quantization_*.py`. + +--- + +### Task 1: Quantization Config Contract + +**Files:** +- Create: `configs/vision_v8_quantization_thresholds.json` +- Create: `configs/vision_v8_quantization_calibration.example.json` +- Test: `tests/test_onnx_quantization_config.py` + +**Interfaces:** +- Produces: `load_quantization_thresholds(path: Path) -> dict[str, Any]` +- Produces: `load_calibration_manifest(path: Path) -> dict[str, Any]` + +- [ ] **Step 1: Write config validation tests** + +```python +from pathlib import Path + +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + load_calibration_manifest, + load_quantization_thresholds, +) + + +def test_threshold_config_requires_explicit_precision_policy(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text('{"schema_version": 1, "precisions": {"fp16": {}, "int8": {}}}') + + with pytest.raises(ValueError, match="release_policy"): + load_quantization_thresholds(path) + + +def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": {"name": "coco-2017-train", "image_ids_sha256": "abc"}, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + ) + + manifest = load_calibration_manifest(path) + + assert manifest["quantization"]["calibration_method"] == "minmax" + assert manifest["quantization"]["batch_size"] == 8 +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Expected: imports fail because `scripts/check_onnx_quantized_artifacts.py` does not exist yet. + +- [ ] **Step 3: Add config files** + +Create `configs/vision_v8_quantization_thresholds.json`: + +```json +{ + "schema_version": 1, + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "partial_release": "block", + "optional_provider_precisions": [] + }, + "precisions": { + "fp16": { + "max_raw_logit_abs_error": 0.01, + "max_decoded_box_px_error": 1.0, + "max_score_abs_error": 0.01, + "max_map50_95_drop": 0.005, + "max_map50_drop": 0.01, + "unexpected_fp32_nodes": "fail" + }, + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + "max_map50_95_drop": 0.02, + "max_map50_drop": 0.03, + "unexpected_fp32_nodes": "allow" + } + }, + "providers": { + "CPUExecutionProvider": ["fp32", "int8"], + "CUDAExecutionProvider": ["fp32", "fp16"], + "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] + }, + "benchmark": { + "warmup_iterations": 25, + "measured_iterations": 100, + "report": ["median_ms", "mean_ms", "stddev_ms", "p95_ms", "throughput_images_per_second", "peak_memory_mb"] + } +} +``` + +Create `configs/vision_v8_quantization_calibration.example.json`: + +```json +{ + "schema_version": 1, + "dataset": { + "name": "coco-2017-train-calibration", + "split": "train2017-calibration", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "replace-with-annotations-sha256", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } +} +``` + +- [ ] **Step 4: Add minimal config loaders** + +Create `scripts/check_onnx_quantized_artifacts.py` with: + +```python +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + + +def _load_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"JSON root must be an object: {path}") + return payload + + +def load_quantization_thresholds(path: Path) -> dict[str, Any]: + config = _load_json(path) + if "release_policy" not in config: + raise ValueError("threshold config missing release_policy") + precisions = config.get("precisions") + if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: + raise ValueError("threshold config must define fp16 and int8 precisions") + return config + + +def load_calibration_manifest(path: Path) -> dict[str, Any]: + manifest = _load_json(path) + quantization = manifest.get("quantization") + if not isinstance(quantization, dict): + raise ValueError("calibration manifest missing quantization settings") + required = { + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + } + missing = sorted(required - set(quantization)) + if missing: + raise ValueError(f"calibration manifest missing settings: {missing}") + return manifest +``` + +- [ ] **Step 5: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Commit: + +```bash +git add configs/vision_v8_quantization_thresholds.json configs/vision_v8_quantization_calibration.example.json scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py +git commit -m "Add Vision v8 quantization gate config" +``` + +--- + +### Task 2: Calibration and Evaluation Disjointness + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Test: `tests/test_onnx_quantization_config.py` + +**Interfaces:** +- Produces: `assert_disjoint_image_ids(calibration_ids: set[int], evaluation_ids: set[int]) -> None` +- Produces: `image_id_manifest_sha256(image_ids: Sequence[int]) -> str` + +- [ ] **Step 1: Write failing overlap tests** + +```python +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, + image_id_manifest_sha256, +) + + +def test_calibration_and_eval_image_ids_must_be_disjoint() -> None: + with pytest.raises(ValueError, match="overlap"): + assert_disjoint_image_ids({1, 2, 3}, {3, 4, 5}) + + +def test_image_id_manifest_hash_is_order_stable() -> None: + assert image_id_manifest_sha256([3, 1, 2]) == image_id_manifest_sha256([1, 2, 3]) +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +- [ ] **Step 3: Implement helpers** + +Add: + +```python +import hashlib +from collections.abc import Sequence + + +def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: + digest = hashlib.sha256() + for image_id in sorted(map(int, image_ids)): + digest.update(f"{image_id}\n".encode("utf-8")) + return digest.hexdigest() + + +def assert_disjoint_image_ids(calibration_ids: set[int], evaluation_ids: set[int]) -> None: + overlap = calibration_ids & evaluation_ids + if overlap: + preview = sorted(overlap)[:10] + raise ValueError(f"calibration/evaluation image ID overlap: {preview}") +``` + +- [ ] **Step 4: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py +git commit -m "Reject quantization calibration eval leakage" +``` + +--- + +### Task 3: FP16 and INT8 Quantization CLI + +**Files:** +- Create: `scripts/quantize_onnx.py` +- Test: `tests/test_onnx_quantization_cli.py` + +**Interfaces:** +- Produces: `quantize_fp16(input_model: Path, output_model: Path, keep_fp32_op_types: Sequence[str]) -> None` +- Produces: `quantize_int8(input_model: Path, output_model: Path, calibration_manifest: Mapping[str, Any]) -> None` +- Produces: sidecar JSON with precision, source SHA-256, output SHA-256, toolchain versions, and quantization settings. + +- [ ] **Step 1: Write failing CLI metadata tests** + +```python +import json +from pathlib import Path + +from scripts.quantize_onnx import write_quantization_sidecar + + +def test_quantization_sidecar_binds_artifact_to_source_and_settings(tmp_path: Path) -> None: + sidecar = tmp_path / "model.fp16.json" + + write_quantization_sidecar( + sidecar, + precision="fp16", + source_model_sha256="source", + output_model_sha256="output", + framework_commit="commit", + checkpoint_revision="checkpoint", + settings={"keep_fp32_op_types": ["ReduceSum"]}, + ) + + data = json.loads(sidecar.read_text()) + assert data["precision"] == "fp16" + assert data["source_model_sha256"] == "source" + assert data["output_model_sha256"] == "output" + assert data["framework_commit"] == "commit" +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_cli.py` + +- [ ] **Step 3: Implement CLI skeleton and sidecar writer** + +Use `onnxconverter-common` or ONNX Runtime quantization APIs where available. If the dependency is unavailable, fail with: + +```text +FP16 quantization requires onnxconverter-common; install the export/quantization extra. +``` + +For INT8, fail with: + +```text +INT8 quantization requires onnxruntime.quantization and a calibration manifest. +``` + +- [ ] **Step 4: Implement FP16 conversion** + +Call the FP16 converter with explicit settings: + +```python +keep_fp32_op_types = tuple(settings.get("keep_fp32_op_types", ())) +disable_shape_infer = bool(settings.get("disable_shape_infer", False)) +``` + +Write output ONNX and sidecar. + +- [ ] **Step 5: Implement INT8 static quantization** + +Use pinned calibration settings: + +```python +calibration_method = manifest["quantization"]["calibration_method"] +per_channel = manifest["quantization"]["per_channel"] +activation_type = manifest["quantization"]["activation_type"] +weight_type = manifest["quantization"]["weight_type"] +``` + +Set calibration reader iteration order from the sorted calibration manifest. + +- [ ] **Step 6: Run quantization twice and assert same hash** + +Add CLI option: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model tr_hash_v8_o2m.onnx \ + --metadata tr_hash_v8_o2m.json \ + --precision fp16 \ + --output artifacts/a.onnx \ + --repeat-output artifacts/b.onnx \ + --require-identical-hash +``` + +If hashes differ, fail with: + +```text +quantization is not deterministic: first SHA-256 ... differs from repeat SHA-256 ... +``` + +- [ ] **Step 7: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_cli.py` + +Commit: + +```bash +git add scripts/quantize_onnx.py tests/test_onnx_quantization_cli.py +git commit -m "Add deterministic Vision v8 ONNX quantization CLI" +``` + +--- + +### Task 4: Node Dtype and Provider Support Validation + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Test: `tests/test_onnx_quantization_artifact_checks.py` + +**Interfaces:** +- Produces: `inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]` +- Produces: `check_provider_precision_supported(provider: str, precision: str, thresholds: Mapping[str, Any]) -> None` +- Produces: `check_unexpected_fp32_nodes(dtype_report: Mapping[str, Any], allowlist: Sequence[str]) -> list[str]` + +- [ ] **Step 1: Write provider support tests** + +```python +import pytest + +from scripts.check_onnx_quantized_artifacts import check_provider_precision_supported + + +def test_unsupported_provider_precision_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="does not support fp16"): + check_provider_precision_supported("CPUExecutionProvider", "fp16", thresholds) +``` + +- [ ] **Step 2: Write FP32 node allowlist tests** + +```python +from scripts.check_onnx_quantized_artifacts import check_unexpected_fp32_nodes + + +def test_unexpected_fp32_nodes_are_reported() -> None: + report = {"fp32_nodes": [{"name": "Conv_1", "op_type": "Conv"}, {"name": "ReduceSum_1", "op_type": "ReduceSum"}]} + + unexpected = check_unexpected_fp32_nodes(report, allowlist=["ReduceSum"]) + + assert unexpected == ["Conv_1:Conv"] +``` + +- [ ] **Step 3: Implement provider support check** + +```python +def check_provider_precision_supported(provider: str, precision: str, thresholds: Mapping[str, Any]) -> None: + supported = thresholds.get("providers", {}).get(provider) + if precision not in supported: + raise ValueError(f"{provider} does not support {precision} in quantization release config") +``` + +- [ ] **Step 4: Implement dtype allowlist check** + +```python +def check_unexpected_fp32_nodes(dtype_report: Mapping[str, Any], allowlist: Sequence[str]) -> list[str]: + allowed = set(allowlist) + unexpected = [] + for node in dtype_report.get("fp32_nodes", []): + if node["op_type"] not in allowed: + unexpected.append(f"{node['name']}:{node['op_type']}") + return unexpected +``` + +- [ ] **Step 5: Implement ONNX dtype inspection** + +Use ONNX graph initializers and value info to identify FP32 tensors that remain after FP16 conversion. Report: + +```json +{ + "fp32_nodes": [{"name": "node_name", "op_type": "Conv"}], + "fp16_nodes": 123, + "int8_nodes": 45 +} +``` + +- [ ] **Step 6: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_artifact_checks.py` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_artifact_checks.py +git commit -m "Validate quantized ONNX providers and node dtypes" +``` + +--- + +### Task 5: Quantized Parity and COCO Accuracy Gate + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Modify: `scripts/check_onnx_parity.py` if helper reuse is needed +- Reuse: `scripts/evaluate_onnx_coco.py` +- Test: `tests/test_onnx_quantization_accuracy_gate.py` + +**Interfaces:** +- Produces: `check_quantized_accuracy_report(report: Mapping[str, Any], thresholds: Mapping[str, Any]) -> list[str]` +- Consumes: COCO report schema emitted by `scripts/evaluate_onnx_coco.py` + +- [ ] **Step 1: Write FP32 comparison tests** + +```python +from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report + + +def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> None: + report = { + "reference": {"precision": "fp32", "metrics": {"map50_95": 0.2, "map50": 0.32}}, + "candidate": {"precision": "int8", "metrics": {"map50_95": 0.17, "map50": 0.31}}, + } + thresholds = {"precisions": {"int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert any("map50_95" in failure for failure in failures) +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_accuracy_gate.py` + +- [ ] **Step 3: Implement accuracy gate helper** + +Compare candidate metrics against FP32 reference metrics for each branch: + +```python +drop = float(reference_metric) - float(candidate_metric) +if drop > allowed_drop: + failures.append(f"{precision} {branch} {metric} dropped by {drop:.6f}") +``` + +- [ ] **Step 4: Wire to existing ONNX COCO evaluator** + +Do not create a new COCO eval path. The release workflow must call: + +```bash +python scripts/evaluate_onnx_coco.py \ + --model "$MODEL" \ + --metadata "$METADATA" \ + --annotations "$COCO_ANNOTATIONS" \ + --images "$COCO_IMAGES" \ + --output "$REPORT_DIR" \ + --provider "$PROVIDER" +``` + +- [ ] **Step 5: Add decoded-output parity checks** + +Reuse the ONNX detector pipeline to compare: + +```json +{ + "max_box_pixel_error": 0.7, + "max_score_abs_error": 0.006, + "class_id_mismatches": 0 +} +``` + +Fail if the precision-specific threshold is exceeded. + +- [ ] **Step 6: Run tests and commit** + +Run: + +```bash +python -m pytest -q tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_detector_core.py tests/test_vision_v8_coco_accuracy_gate.py +``` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_accuracy_gate.py +git commit -m "Gate quantized ONNX accuracy against FP32" +``` + +--- + +### Task 6: Benchmark Methodology and Reports + +**Files:** +- Create: `scripts/benchmark_onnx_artifacts.py` +- Test: `tests/test_onnx_quantization_benchmark.py` + +**Interfaces:** +- Produces: `summarize_latency_ms(values: Sequence[float]) -> dict[str, float]` +- Produces: benchmark JSON with median, mean, stddev, p95, throughput, peak memory, provider requested, provider used, warmup count, measured count. + +- [ ] **Step 1: Write benchmark summary tests** + +```python +from scripts.benchmark_onnx_artifacts import summarize_latency_ms + + +def test_benchmark_report_uses_distribution_not_single_shot() -> None: + summary = summarize_latency_ms([10.0, 12.0, 14.0]) + + assert summary["median_ms"] == 12.0 + assert summary["mean_ms"] == 12.0 + assert summary["stddev_ms"] > 0.0 +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_benchmark.py` + +- [ ] **Step 3: Implement benchmark script** + +Use fixed methodology: + +```text +warmup_iterations = 25 +measured_iterations = 100 +batch_size = 1 unless explicitly configured +report median + mean + stddev + p95 +benchmark tolerance is separate from accuracy tolerance +``` + +- [ ] **Step 4: Add peak memory capture** + +For CUDA providers, use CUDA memory APIs where available. For CPU, record process RSS before/after and document it as approximate. + +- [ ] **Step 5: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_benchmark.py` + +Commit: + +```bash +git add scripts/benchmark_onnx_artifacts.py tests/test_onnx_quantization_benchmark.py +git commit -m "Add quantized ONNX benchmark reports" +``` + +--- + +### Task 7: Release Workflow Integration + +**Files:** +- Create or modify: `.github/workflows/vision-v8-onnx-quantization-release.yml` +- Modify: `.github/workflows/detector-export.yml` only if the repo prefers one detector workflow +- Test: workflow YAML lint through `ruff` only for Python files and local shell dry-run where possible + +**Interfaces:** +- Consumes: `scripts/quantize_onnx.py` +- Consumes: `scripts/check_onnx_quantized_artifacts.py` +- Consumes: `scripts/evaluate_onnx_coco.py` +- Consumes: `scripts/benchmark_onnx_artifacts.py` + +- [ ] **Step 1: Add manual/tag-triggered workflow** + +Workflow triggers: + +```yaml +on: + workflow_dispatch: + push: + tags: + - "vision-v8-onnx-*" +``` + +- [ ] **Step 2: Export FP32 artifacts** + +Call the existing ONNX export path for: + +```text +o2m fp32 +nms-free fp32 +``` + +- [ ] **Step 3: Quantize all required artifacts** + +Generate: + +```text +o2m fp16 +o2m int8 +nms-free fp16 +nms-free int8 +``` + +Run each quantization command twice with `--require-identical-hash`. + +- [ ] **Step 4: Validate provider and node dtype policy** + +Fail if: + +```text +requested provider != actual provider +precision unsupported by provider config +FP16 output contains unexpected FP32 nodes outside allowlist +INT8 metadata does not match pinned calibration settings +``` + +- [ ] **Step 5: Run parity, COCO, and benchmark gates** + +Call: + +```bash +python scripts/evaluate_onnx_coco.py ... +python scripts/check_onnx_quantized_artifacts.py ... +python scripts/benchmark_onnx_artifacts.py ... +``` + +- [ ] **Step 6: Enforce partial-failure policy** + +If any required artifact fails, stop the workflow before release upload: + +```text +required quantized artifact failed gate; release blocked by release_policy.partial_release=block +``` + +- [ ] **Step 7: Publish release assets** + +Upload: + +```text +FP32 ONNX files +FP16 ONNX files +INT8 ONNX files +JSON sidecars +manifest JSON +accuracy reports JSON/Markdown +benchmark reports JSON/Markdown +``` + +- [ ] **Step 8: Commit** + +```bash +git add .github/workflows/vision-v8-onnx-quantization-release.yml +git commit -m "Publish gated quantized Vision v8 ONNX releases" +``` + +--- + +### Task 8: Documentation and Final Verification + +**Files:** +- Create: `docs/vision-v8-onnx-quantization.md` +- Modify: `docs/index.md` + +**Interfaces:** +- Documents exact commands for FP16, INT8, parity, COCO evaluation, benchmark, and release upload. + +- [ ] **Step 1: Document beginner-friendly concepts** + +Explain: + +```text +FP32 = bigger, safest reference +FP16 = smaller/faster float model, may leave some nodes FP32 for stability +INT8 = smallest/fastest candidate, requires calibration data +Calibration = sample inputs used to estimate activation ranges +Provider fallback = ORT did not use requested runtime +Partial FP16 fallback = some graph nodes stayed FP32 +``` + +- [ ] **Step 2: Document reproduction commands** + +Include: + +```bash +python scripts/quantize_onnx.py ... +python scripts/evaluate_onnx_coco.py ... +python scripts/check_onnx_quantized_artifacts.py ... +python scripts/benchmark_onnx_artifacts.py ... +``` + +- [ ] **Step 3: Run final local checks** + +Run: + +```bash +python -m ruff check scripts/quantize_onnx.py scripts/check_onnx_quantized_artifacts.py scripts/benchmark_onnx_artifacts.py tests/test_onnx_quantization_config.py tests/test_onnx_quantization_cli.py tests/test_onnx_quantization_artifact_checks.py tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_quantization_benchmark.py +python -m pytest -q tests/test_onnx_quantization_config.py tests/test_onnx_quantization_cli.py tests/test_onnx_quantization_artifact_checks.py tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_quantization_benchmark.py +``` + +- [ ] **Step 4: Record known untested release requirements** + +Before opening PR, state clearly whether these have run: + +```text +Full COCO FP32 vs FP16 vs INT8 +CUDA provider execution +TensorRT provider execution +GitHub Release upload +Quantize-twice identical artifact hash +``` + +- [ ] **Step 5: Commit** + +```bash +git add docs/vision-v8-onnx-quantization.md docs/index.md +git commit -m "Document Vision v8 ONNX quantization release process" +``` + +--- + +## Self-Review + +- Spec coverage: FP16 and INT8 quantization, calibration pinning, toolchain/settings metadata, size/latency/throughput/memory reports, raw-logit/decoded/COCO gates, release artifact publishing, checksums, framework commit, checkpoint revision, provider failure policy, and no source-committed ONNX binaries are all mapped to tasks. +- Reviewer gaps covered: quantize-twice hash determinism is Task 3; FP16 partial dtype fallback is Task 4; calibration/eval leakage is Task 2; shared COCO evaluator reuse is Task 5; benchmark noise methodology is Task 6; checked-in thresholds are Task 1; partial release policy is Task 7. +- Type consistency: helper names used by later tasks are introduced before they are consumed. diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md new file mode 100644 index 0000000..34076f6 --- /dev/null +++ b/docs/vision-v8-onnx-quantization.md @@ -0,0 +1,173 @@ +# Vision v8 ONNX Quantization + +Vision v8 ONNX quantization produces smaller deployment artifacts while keeping +FP32 as the accuracy and behavior reference. Quantized artifacts are not +accepted only because they run; they must pass raw-output, decoded-detection, +COCO accuracy, provider, dtype, benchmark, checksum, and release metadata gates. + +## Precision Contract + +- `fp32`: the exported reference model and source of truth. +- `fp16`: smaller floating-point model for CUDA/TensorRT-style deployment. +- `int8`: post-training quantized model calibrated from pinned inputs. + +Each branch is quantized separately: + +- `o2m`: decode, confidence filtering, and class-aware NMS. +- `nms-free`: decode and confidence filtering only; NMS must not run. + +Unsupported provider and precision pairs fail explicitly using +`configs/vision_v8_quantization_thresholds.json`. Provider fallback is separate +from FP16 partial fallback: the release check must verify both that ONNX Runtime +used the requested provider and that unexpected graph nodes did not remain FP32. +The default release contract runs FP32 and INT8 on CPU, but FP16 on CUDA because +the CPU ONNX Runtime build does not support executing FP16 detector graphs. + +## Calibration Contract + +INT8 calibration is pinned by +the release evidence artifact. `configs/vision_v8_quantization_calibration.example.json` +documents the expected shape. The manifest records the +dataset, image-ID manifest hash, annotation hash, calibration method, +per-channel/per-tensor mode, symmetric/asymmetric choices, activation type, +weight type, and batch size. The ORT symmetry options and batch size are passed +into quantization directly; unsupported settings are not recorded in release +metadata. +The default release export uses a fixed batch of 1, so the default calibration +manifest must also use `batch_size: 1`. If the export is changed to dynamic +batching, the calibration batch size can be raised in the same release contract. +Placeholder hashes are rejected by `load_calibration_manifest`; before a release +run, replace them with the approved calibration subset and real SHA-256 values. + +Calibration images must be disjoint from the COCO evaluation images used by the +accuracy gate. This prevents tuning the INT8 ranges on the same images used to +claim final AP. + +## Reproduction Commands + +Create an FP16 artifact: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model artifacts/onnx/tr_hash_v8_o2m.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m.json \ + --precision fp16 \ + --output artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --repeat-output artifacts/onnx/tr_hash_v8_o2m_fp16_repeat.onnx \ + --require-identical-hash \ + --checkpoint-revision AETHORIA-AI/TR-HASH-Vision-v8-2M-COCO-SFT@REVISION +``` + +By default the CLI writes two JSON files for a quantized artifact: +`tr_hash_v8_o2m_fp16.json` is a detector metadata copy used by the inference +pipeline, while `tr_hash_v8_o2m_fp16.quantization.json` is the quantization +provenance sidecar used by release verification. + +Create an INT8 artifact: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model artifacts/onnx/tr_hash_v8_o2m.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m.json \ + --precision int8 \ + --calibration-manifest artifacts/vision_v8_quantized_eval/calibration.json \ + --output artifacts/onnx/tr_hash_v8_o2m_int8.onnx \ + --repeat-output artifacts/onnx/tr_hash_v8_o2m_int8_repeat.onnx \ + --require-identical-hash \ + --checkpoint-revision AETHORIA-AI/TR-HASH-Vision-v8-2M-COCO-SFT@REVISION +``` + +Evaluate a quantized ONNX artifact on COCO: + +```bash +python scripts/evaluate_onnx_coco.py \ + --model artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m_fp16.json \ + --annotations artifacts/COCO/annotations/instances_val2017.json \ + --images artifacts/COCO/images/val2017 \ + --output artifacts/vision_v8_quantized_eval/o2m_fp16 \ + --branch o2m-nms \ + --provider cuda +``` + +Before publishing, merge the FP32 reference and quantized branch reports into +`artifacts/vision_v8_quantized_eval/accuracy.json` and +`artifacts/vision_v8_quantized_eval/accuracy.md`. The release builder validates +that JSON against `configs/vision_v8_quantization_thresholds.json`; missing or +regressed reports block publication. The report must include the evaluated image +IDs so the release gate can prove the calibration set and evaluation set are +actually disjoint. + +The merge report is precision-nested by branch. For example, each branch must +contain `fp32`, `fp16`, and `int8` entries so the gate can compare every +quantized candidate against the FP32 reference: + +```bash +python scripts/merge_vision_v8_coco_reports.py \ + artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json \ + artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_int8/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_int8/evaluation.json \ + --output artifacts/vision_v8_quantized_eval +cp artifacts/vision_v8_quantized_eval/evaluation.json \ + artifacts/vision_v8_quantized_eval/accuracy.json +cp artifacts/vision_v8_quantized_eval/evaluation.md \ + artifacts/vision_v8_quantized_eval/accuracy.md +``` + +Benchmark an artifact: + +```bash +python scripts/benchmark_onnx_artifacts.py \ + --model artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m_fp16.json \ + --output artifacts/vision_v8_quantized_eval/o2m_fp16/benchmark.json \ + --provider cuda \ + --warmup-iterations 25 \ + --measured-iterations 100 +``` + +The release build also generates `quantized_benchmarks.json` and +`quantized_benchmarks.md` for every FP32, FP16, and INT8 branch artifact using +the benchmark methodology from the checked-in threshold config. + +The GitHub release workflow downloads a prior Actions artifact named +`vision-v8-quantized-release-inputs`. The manual Vision v8 COCO accuracy +workflow produces that artifact when `backend=onnx`, `calibration_manifest` is +set, and FP32/FP16/INT8 model plus metadata paths are provided for both +branches. Dispatch it with `fp32_provider=cpu`, `fp16_provider=cuda`, and +`int8_provider=cpu` so the evidence matches the checked-in release provider +policy. The merged accuracy report preserves `requested_provider` and +`actual_provider` inside each precision entry, and the release gate rejects +provider fallback or stale evidence generated from a different framework +commit. The artifact must provide `calibration.json`, `calibration_images/`, +`accuracy.json`, and `accuracy.md` under +`artifacts/vision_v8_quantized_eval/` before `build_onnx_release.py` runs. The +COCO workflow copies the pinned calibration images into `calibration_images/` +and rewrites the manifest paths so the fresh release runner can open them. +Pass the source run as workflow input `evidence_run_id`, or set repository +variable `ONNX_RELEASE_EVIDENCE_RUN_ID` for tag-triggered releases. +Run the ONNX release workflow on a CUDA-capable self-hosted runner so the FP16 +parity and benchmark gates use `onnxruntime-gpu` instead of CPU fallback. + +## Release Policy + +The quantized release workflow blocks the release when any required artifact +fails. Do not publish a partial release, such as FP32 plus FP16 only, unless the +checked-in release policy explicitly marks the failed precision/provider as +optional. + +Release assets must include: + +- FP32, FP16, and INT8 ONNX files for both O2M and NMS-free branches. +- Detector metadata JSON sidecars and quantization provenance JSON sidecars. +- FP32-vs-quantized parity reports for raw logits, decoded boxes, scores, and + class/count stability. +- Accuracy reports in JSON and Markdown. +- Benchmark reports in JSON and Markdown. +- A manifest binding every artifact to framework commit, checkpoint revision, + SHA-256, file size, provider, precision, and quantization settings. + +ONNX binaries remain excluded from normal source commits. diff --git a/scripts/benchmark_onnx_artifacts.py b/scripts/benchmark_onnx_artifacts.py new file mode 100644 index 0000000..445ef23 --- /dev/null +++ b/scripts/benchmark_onnx_artifacts.py @@ -0,0 +1,189 @@ +"""Benchmark Vision v8 ONNX artifacts with stable release-report methodology.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import platform +import statistics +import sys +import time +from collections.abc import Sequence +from pathlib import Path +from typing import Any + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--provider", action="append", default=None) + parser.add_argument("--batch-size", type=int, default=1) + parser.add_argument("--warmup-iterations", type=int, default=25) + parser.add_argument("--measured-iterations", type=int, default=100) + parser.add_argument("--ort-intra-op-threads", type=int, default=1) + parser.add_argument("--ort-inter-op-threads", type=int, default=1) + return parser.parse_args() + + +def _percentile(values: Sequence[float], fraction: float) -> float: + if not values: + return 0.0 + ordered = sorted(values) + index = min(round((len(ordered) - 1) * fraction), len(ordered) - 1) + return float(ordered[index]) + + +def summarize_latency_ms(values: Sequence[float]) -> dict[str, float]: + """Summarize latency as a distribution to avoid single-shot noise.""" + + if not values: + return { + "median_ms": 0.0, + "mean_ms": 0.0, + "stddev_ms": 0.0, + "p95_ms": 0.0, + "p99_ms": 0.0, + } + return { + "median_ms": float(statistics.median(values)), + "mean_ms": float(statistics.fmean(values)), + "stddev_ms": float(statistics.stdev(values)) if len(values) > 1 else 0.0, + "p95_ms": _percentile(values, 0.95), + "p99_ms": _percentile(values, 0.99), + } + + +def _current_memory_mb() -> float | None: + try: + import psutil + except ImportError: + return None + return psutil.Process(os.getpid()).memory_info().rss / (1024 * 1024) + + +def _benchmark_session( + pipeline: Any, + *, + batch_size: int, + warmup_iterations: int, + measured_iterations: int, +) -> tuple[list[float], float | None]: + input_shape = ( + batch_size, + 3, + pipeline.metadata.image_size, + pipeline.metadata.image_size, + ) + dummy = np.zeros(input_shape, dtype=np.float32) + peak_memory_mb = _current_memory_mb() + for _ in range(warmup_iterations): + pipeline.session.run(dummy) + peak_memory_mb = _max_optional(peak_memory_mb, _current_memory_mb()) + + latencies: list[float] = [] + for _ in range(measured_iterations): + started = time.perf_counter() + pipeline.session.run(dummy) + latencies.append((time.perf_counter() - started) * 1000.0) + peak_memory_mb = _max_optional(peak_memory_mb, _current_memory_mb()) + return latencies, peak_memory_mb + + +def _max_optional(first: float | None, second: float | None) -> float | None: + if first is None: + return second + if second is None: + return first + return max(first, second) + + +def benchmark_onnx_artifact( + *, + model_path: Path, + metadata_path: Path, + providers: Sequence[str], + batch_size: int, + warmup_iterations: int, + measured_iterations: int, + ort_intra_op_threads: int, + ort_inter_op_threads: int, +) -> dict[str, Any]: + from complexity.deploy.onnx_detector import OnnxDetectorPipeline + from scripts.quantize_onnx import package_version, sha256_file + + if batch_size <= 0 or warmup_iterations < 0 or measured_iterations <= 0: + raise ValueError("batch size and measured iterations must be positive") + pipeline = OnnxDetectorPipeline.from_files( + model_path, + metadata_path, + providers=providers, + intra_op_num_threads=ort_intra_op_threads, + inter_op_num_threads=ort_inter_op_threads, + ) + latencies, peak_memory_mb = _benchmark_session( + pipeline, + batch_size=batch_size, + warmup_iterations=warmup_iterations, + measured_iterations=measured_iterations, + ) + summary = summarize_latency_ms(latencies) + mean_ms = summary["mean_ms"] + throughput = (batch_size * 1000.0 / mean_ms) if mean_ms else 0.0 + return { + "schema_version": 1, + "model": str(model_path), + "metadata": str(metadata_path), + "model_sha256": sha256_file(model_path), + "model_size_bytes": model_path.stat().st_size, + "requested_provider": list(providers), + "actual_provider": pipeline.session.provider_used, + "batch_size": batch_size, + "warmup_iterations": warmup_iterations, + "measured_iterations": measured_iterations, + "latency": summary, + "throughput_images_per_second": throughput, + "peak_memory_mb": peak_memory_mb, + "benchmark_methodology": ("fixed warmup, fixed measured iterations, latency distribution"), + "environment": { + "python": sys.version.split()[0], + "os": platform.platform(), + "onnxruntime": package_version("onnxruntime"), + }, + "ort_intra_op_threads": ort_intra_op_threads, + "ort_inter_op_threads": ort_inter_op_threads, + } + + +def main() -> None: + from scripts.onnx_detect import provider_names + + args = parse_args() + if not math.isfinite(float(args.batch_size)): + raise ValueError("batch size must be finite") + report = benchmark_onnx_artifact( + model_path=args.model, + metadata_path=args.metadata, + providers=provider_names(args.provider), + batch_size=args.batch_size, + warmup_iterations=args.warmup_iterations, + measured_iterations=args.measured_iterations, + ort_intra_op_threads=args.ort_intra_op_threads, + ort_inter_op_threads=args.ort_inter_op_threads, + ) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/build_onnx_release.py b/scripts/build_onnx_release.py index 7c31101..c839912 100644 --- a/scripts/build_onnx_release.py +++ b/scripts/build_onnx_release.py @@ -21,10 +21,15 @@ import json import os import subprocess +import sys from dataclasses import dataclass from pathlib import Path from typing import Any, Mapping, Sequence +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + os.environ.setdefault("COMPLEXITY_DISABLE_KERNELS", "1") MANIFEST_NAME = "manifest.json" @@ -61,6 +66,20 @@ class ReleaseConfig: parity_num_tests: int toolchain: Mapping[str, str] branches: tuple[BranchSpec, ...] + quantization: QuantizationConfig | None = None + + +@dataclass(frozen=True) +class QuantizationConfig: + """Pinned inputs for optional quantized release artifacts.""" + + enabled_precisions: tuple[str, ...] + calibration_manifest: Path + thresholds: Path + accuracy_report: Path + accuracy_markdown: Path + fp32_op_allowlist: tuple[str, ...] + provider_gates: tuple[tuple[str, str], ...] class ReleaseError(RuntimeError): @@ -78,8 +97,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: BranchSpec( branch=str(entry["branch"]), checkpoint_subdir=( - None if entry.get("checkpoint_subdir") is None - else str(entry["checkpoint_subdir"]) + None if entry.get("checkpoint_subdir") is None else str(entry["checkpoint_subdir"]) ), stem=str(entry["stem"]), post_processing=str(entry["post_processing"]), @@ -102,6 +120,31 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: if not isinstance(toolchain, Mapping) or not toolchain: raise ReleaseError("release config must pin a toolchain") + quantization = None + if isinstance(values.get("quantization"), Mapping): + raw_quantization = values["quantization"] + enabled_precisions = tuple( + str(precision) for precision in raw_quantization.get("enabled_precisions", ()) + ) + unsupported = sorted(set(enabled_precisions) - {"fp16", "int8"}) + if unsupported: + raise ReleaseError(f"unsupported quantized precisions: {unsupported}") + provider_gates = tuple( + (str(gate["provider"]), str(gate["precision"])) + for gate in raw_quantization.get("provider_gates", ()) + ) + quantization = QuantizationConfig( + enabled_precisions=enabled_precisions, + calibration_manifest=Path(str(raw_quantization["calibration_manifest"])), + thresholds=Path(str(raw_quantization["thresholds"])), + accuracy_report=Path(str(raw_quantization["accuracy_report"])), + accuracy_markdown=Path(str(raw_quantization["accuracy_markdown"])), + fp32_op_allowlist=tuple( + str(op_type) for op_type in raw_quantization.get("fp32_op_allowlist", ()) + ), + provider_gates=provider_gates, + ) + return ReleaseConfig( checkpoint_repo=str(values["checkpoint_repo"]), checkpoint_revision=revision, @@ -109,6 +152,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: parity_num_tests=int(values.get("parity_num_tests", 5)), toolchain={str(k): str(v) for k, v in toolchain.items()}, branches=branches, + quantization=quantization, ) @@ -175,6 +219,22 @@ def artifact_entry(path: Path, **fields: Any) -> dict[str, Any]: } +def provider_chain(provider: str) -> tuple[str, ...]: + if provider == "TensorrtExecutionProvider": + return ( + "TensorrtExecutionProvider", + "CUDAExecutionProvider", + "CPUExecutionProvider", + ) + if provider == "CUDAExecutionProvider": + return ("CUDAExecutionProvider", "CPUExecutionProvider") + return (provider,) + + +def coco_report_branch(branch: str) -> str: + return "o2m-nms" if branch == "o2m" else branch + + def output_contract(sidecar: Mapping[str, Any]) -> dict[str, Any]: """Describe the ONNX input/output contract from an export sidecar.""" @@ -231,9 +291,7 @@ def verify_manifest( if expect_commit is not None: recorded = str(manifest.get("framework_commit", "")) if recorded != expect_commit: - problems.append( - f"framework_commit {recorded or 'missing'}, expected {expect_commit}" - ) + problems.append(f"framework_commit {recorded or 'missing'}, expected {expect_commit}") for artifact in manifest["artifacts"]: path = Path(directory) / str(artifact["name"]) if not path.is_file(): @@ -242,14 +300,12 @@ def verify_manifest( actual_size = path.stat().st_size if actual_size != int(artifact["size_bytes"]): problems.append( - f"{artifact['name']}: size {actual_size}, " - f"manifest {artifact['size_bytes']}" + f"{artifact['name']}: size {actual_size}, manifest {artifact['size_bytes']}" ) actual_digest = sha256_file(path) if actual_digest != str(artifact["sha256"]): problems.append( - f"{artifact['name']}: sha256 {actual_digest}, " - f"manifest {artifact['sha256']}" + f"{artifact['name']}: sha256 {actual_digest}, manifest {artifact['sha256']}" ) return problems @@ -267,14 +323,15 @@ def render_release_notes(manifest: Mapping[str, Any]) -> str: "Both models expose raw detector logits only; decode and post-processing " "run outside the graph.", "", - "| Branch | Model | Post-processing | Size | SHA-256 |", - "|---|---|---|---:|---|", + "| Branch | Precision | Model | Post-processing | Size | SHA-256 |", + "|---|---|---|---|---:|---|", ] for artifact in manifest["artifacts"]: if artifact.get("kind") != "model": continue lines.append( - f"| {artifact['branch']} | `{artifact['name']}` | " + f"| {artifact['branch']} | {artifact.get('precision', 'unknown')} | " + f"`{artifact['name']}` | " f"{artifact['post_processing']} | {artifact['size_bytes']:,} bytes | " f"`{artifact['sha256']}` |" ) @@ -307,6 +364,34 @@ def render_release_notes(manifest: Mapping[str, Any]) -> str: return "\n".join(lines) +def render_benchmark_report(report: Mapping[str, Any]) -> str: + """Render benchmark evidence into a release-friendly Markdown table.""" + + lines = [ + "# Vision v8 ONNX Quantization Benchmarks", + "", + "| Branch | Precision | Provider | Median ms | P95 ms | Throughput | Peak MB |", + "|---|---|---|---:|---:|---:|---:|", + ] + for branch, branch_report in _mapping(report.get("branches")).items(): + for precision, precision_report in _mapping(branch_report).items(): + data = _mapping(precision_report) + latency = _mapping(data.get("latency")) + lines.append( + f"| {branch} | {precision} | {data.get('actual_provider', '')} | " + f"{float(latency.get('median_ms', 0.0)):.3f} | " + f"{float(latency.get('p95_ms', 0.0)):.3f} | " + f"{float(data.get('throughput_images_per_second', 0.0)):.3f} | " + f"{float(data.get('peak_memory_mb', 0.0)):.3f} |" + ) + lines.append("") + return "\n".join(lines) + + +def _mapping(value: object) -> Mapping[str, Any]: + return value if isinstance(value, Mapping) else {} + + def build_release( config: ReleaseConfig, output_dir: Path, @@ -317,8 +402,32 @@ def build_release( from huggingface_hub import snapshot_download + from scripts.benchmark_onnx_artifacts import benchmark_onnx_artifact from scripts.check_onnx_parity import check_parity + from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, + check_provider_precision_supported, + check_quantized_accuracy_report, + check_quantized_benchmark_report, + check_quantized_parity_report, + check_unexpected_fp32_nodes, + evaluation_image_ids_from_report, + inspect_onnx_node_dtypes, + load_calibration_manifest, + load_quantization_thresholds, + ) + from scripts.check_onnx_quantized_parity import build_parity_report from scripts.export_onnx import export_onnx + from scripts.quantize_onnx import ( + assert_identical_artifact_hashes, + copy_detector_metadata, + default_quantization_sidecar, + quantize_once, + write_quantization_sidecar, + ) + from scripts.quantize_onnx import ( + toolchain_versions as quantization_toolchain_versions, + ) installed = installed_toolchain() mismatches = toolchain_mismatches(config.toolchain, installed) @@ -335,6 +444,57 @@ def build_release( destination = Path(output_dir) destination.mkdir(parents=True, exist_ok=True) + commit = framework_commit() + artifacts: list[dict[str, Any]] = [] + + quantization_thresholds: Mapping[str, Any] | None = None + calibration_manifest: Mapping[str, Any] | None = None + accuracy_report: Mapping[str, Any] | None = None + provider_by_precision: dict[str, str] = {} + benchmark_report: dict[str, Any] | None = None + benchmark_settings: Mapping[str, Any] = {} + expected_accuracy_artifacts: dict[str, dict[str, dict[str, str]]] = {} + if config.quantization is not None: + quantization_thresholds = load_quantization_thresholds(config.quantization.thresholds) + for provider, precision in config.quantization.provider_gates: + check_provider_precision_supported( + provider, + precision, + quantization_thresholds, + ) + provider_by_precision[precision] = provider + if config.quantization.enabled_precisions: + calibration_manifest = load_calibration_manifest( + config.quantization.calibration_manifest + ) + batch_size = int(calibration_manifest["quantization"]["batch_size"]) + if batch_size != 1: + raise ReleaseError( + "default ONNX release exports fixed batch size 1, " + f"but calibration batch_size is {batch_size}" + ) + if config.quantization.enabled_precisions and calibration_manifest is None: + raise ReleaseError("quantized release parity requires calibration images") + parity_image = Path(str(calibration_manifest["images"][0])) + if not config.quantization.accuracy_report.is_file(): + raise ReleaseError( + "quantized release requires an accuracy comparison report: " + f"{config.quantization.accuracy_report}" + ) + if not config.quantization.accuracy_markdown.is_file(): + raise ReleaseError( + "quantized release requires a Markdown accuracy report: " + f"{config.quantization.accuracy_markdown}" + ) + accuracy_report = json.loads( + config.quantization.accuracy_report.read_text(encoding="utf-8") + ) + assert_disjoint_image_ids( + {int(image_id) for image_id in calibration_manifest["image_ids"]}, + evaluation_image_ids_from_report(accuracy_report), + ) + benchmark_report = {"schema_version": 1, "branches": {}} + benchmark_settings = _mapping(quantization_thresholds.get("benchmark")) print(f"Downloading {config.checkpoint_repo}@{config.checkpoint_revision}") checkpoint_root = Path( @@ -344,7 +504,6 @@ def build_release( ) ) - artifacts: list[dict[str, Any]] = [] for spec in config.branches: checkpoint = ( checkpoint_root @@ -377,19 +536,275 @@ def build_release( model_path, kind="model", branch=spec.branch, + precision="fp32", requires_nms=bool(sidecar["requires_nms"]), post_processing=spec.post_processing, contract=output_contract(sidecar), ) ) artifacts.append( - artifact_entry(sidecar_path, kind="metadata", branch=spec.branch) + artifact_entry( + sidecar_path, + kind="metadata", + branch=spec.branch, + precision="fp32", + ) + ) + if config.quantization is not None and quantization_thresholds is not None: + expected_accuracy_artifacts.setdefault(coco_report_branch(spec.branch), {})["fp32"] = { + "checkpoint_sha256": sha256_file(model_path), + "metadata_sha256": sha256_file(sidecar_path), + } + fp32_benchmark = benchmark_onnx_artifact( + model_path=model_path, + metadata_path=sidecar_path, + providers=provider_chain(provider_by_precision.get("fp32", "CPUExecutionProvider")), + batch_size=1, + warmup_iterations=int(benchmark_settings["warmup_iterations"]), + measured_iterations=int(benchmark_settings["measured_iterations"]), + ort_intra_op_threads=1, + ort_inter_op_threads=1, + ) + check_provider_precision_supported( + str(fp32_benchmark["actual_provider"]), + "fp32", + quantization_thresholds, + ) + benchmark_report["branches"].setdefault(spec.branch, {})["fp32"] = fp32_benchmark + if config.quantization is not None and quantization_thresholds is not None: + for precision in config.quantization.enabled_precisions: + quantized_model_path = destination / f"{spec.stem}_{precision}.onnx" + repeat_model_path = destination / f"{spec.stem}_{precision}_repeat.onnx" + print(f"Quantizing {spec.branch} to {precision}") + settings = quantize_once( + fp32_model=model_path, + metadata_path=sidecar_path, + precision=precision, + output_model=quantized_model_path, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=config.quantization.fp32_op_allowlist, + disable_shape_infer=False, + ) + quantize_once( + fp32_model=model_path, + metadata_path=sidecar_path, + precision=precision, + output_model=repeat_model_path, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=config.quantization.fp32_op_allowlist, + disable_shape_infer=False, + ) + assert_identical_artifact_hashes( + sha256_file(quantized_model_path), + sha256_file(repeat_model_path), + ) + repeat_model_path.unlink() + + precision_thresholds = quantization_thresholds["precisions"][precision] + if precision_thresholds.get("unexpected_fp32_nodes") == "fail": + dtype_report = inspect_onnx_node_dtypes(quantized_model_path) + unexpected = check_unexpected_fp32_nodes( + dtype_report, + allowlist=config.quantization.fp32_op_allowlist, + ) + if unexpected: + raise ReleaseError( + f"{quantized_model_path.name} retained unexpected " + "FP32 nodes: " + ", ".join(unexpected) + ) + + quantized_metadata_path = quantized_model_path.with_suffix(".json") + copy_detector_metadata(sidecar_path, quantized_metadata_path) + quantized_sidecar_path = default_quantization_sidecar(quantized_model_path) + write_quantization_sidecar( + quantized_sidecar_path, + precision=precision, + source_model_sha256=sha256_file(model_path), + output_model_sha256=sha256_file(quantized_model_path), + framework_commit=commit, + checkpoint_revision=config.checkpoint_revision, + settings=settings, + toolchain=quantization_toolchain_versions(), + ) + parity_report = build_parity_report( + reference_model=model_path, + candidate_model=quantized_model_path, + metadata_path=sidecar_path, + image_path=parity_image, + precision=precision, + providers=provider_chain( + provider_by_precision.get(precision, "CPUExecutionProvider") + ), + ) + provider_used = str(parity_report["provider_used"]) + check_provider_precision_supported( + provider_used, + precision, + quantization_thresholds, + ) + parity_failures = check_quantized_parity_report( + parity_report, + quantization_thresholds, + ) + if int(parity_report["class_mismatch_count"]) > 0: + parity_failures.append( + f"{spec.branch} {precision} decoded class/count mismatch: " + f"{parity_report['class_mismatch_count']}" + ) + if parity_failures: + raise ReleaseError( + f"{quantized_model_path.name} failed quantized parity:\n " + + "\n ".join(parity_failures) + ) + parity_report_path = destination / f"{spec.stem}_{precision}_parity.json" + parity_report_path.write_text( + json.dumps(parity_report, indent=2) + "\n", + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + quantized_model_path, + kind="model", + branch=spec.branch, + precision=precision, + source_model=Path(model_path).name, + requires_nms=bool(sidecar["requires_nms"]), + post_processing=spec.post_processing, + contract=output_contract(sidecar), + ) + ) + artifacts.append( + artifact_entry( + quantized_metadata_path, + kind="metadata", + branch=spec.branch, + precision=precision, + ) + ) + artifacts.append( + artifact_entry( + quantized_sidecar_path, + kind="quantization_metadata", + branch=spec.branch, + precision=precision, + source_model=Path(model_path).name, + ) + ) + artifacts.append( + artifact_entry( + parity_report_path, + kind="parity_report", + branch=spec.branch, + precision=precision, + ) + ) + expected_accuracy_artifacts.setdefault( + coco_report_branch(spec.branch), + {}, + )[precision] = { + "checkpoint_sha256": sha256_file(quantized_model_path), + "metadata_sha256": sha256_file(quantized_metadata_path), + } + quantized_benchmark = benchmark_onnx_artifact( + model_path=quantized_model_path, + metadata_path=quantized_metadata_path, + providers=provider_chain( + provider_by_precision.get(precision, "CPUExecutionProvider") + ), + batch_size=1, + warmup_iterations=int(benchmark_settings["warmup_iterations"]), + measured_iterations=int(benchmark_settings["measured_iterations"]), + ort_intra_op_threads=1, + ort_inter_op_threads=1, + ) + check_provider_precision_supported( + str(quantized_benchmark["actual_provider"]), + precision, + quantization_thresholds, + ) + benchmark_report["branches"].setdefault(spec.branch, {})[precision] = ( + quantized_benchmark + ) + + if ( + config.quantization is not None + and quantization_thresholds is not None + and benchmark_report is not None + ): + benchmark_failures = check_quantized_benchmark_report( + benchmark_report, + quantization_thresholds, + required_branches=[spec.branch for spec in config.branches], + ) + if benchmark_failures: + raise ReleaseError( + "quantized benchmark gate failed:\n " + "\n ".join(benchmark_failures) + ) + assert accuracy_report is not None + accuracy_failures = check_quantized_accuracy_report( + accuracy_report, + quantization_thresholds, + required_branches=[coco_report_branch(spec.branch) for spec in config.branches], + expected_artifacts=expected_accuracy_artifacts, + expected_framework_commit=commit, + ) + if accuracy_failures: + raise ReleaseError( + "quantized COCO accuracy gate failed:\n " + "\n ".join(accuracy_failures) + ) + accuracy_report_path = destination / "quantized_accuracy.json" + accuracy_markdown_path = destination / "quantized_accuracy.md" + accuracy_report_path.write_text( + json.dumps(accuracy_report, indent=2) + "\n", + encoding="utf-8", + ) + accuracy_markdown_path.write_text( + config.quantization.accuracy_markdown.read_text(encoding="utf-8"), + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + accuracy_report_path, + kind="accuracy_report", + precision="mixed", + ) + ) + artifacts.append( + artifact_entry( + accuracy_markdown_path, + kind="accuracy_report_markdown", + precision="mixed", + ) + ) + benchmark_json_path = destination / "quantized_benchmarks.json" + benchmark_markdown_path = destination / "quantized_benchmarks.md" + benchmark_json_path.write_text( + json.dumps(benchmark_report, indent=2) + "\n", + encoding="utf-8", + ) + benchmark_markdown_path.write_text( + render_benchmark_report(benchmark_report), + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + benchmark_json_path, + kind="benchmark_report", + precision="mixed", + ) + ) + artifacts.append( + artifact_entry( + benchmark_markdown_path, + kind="benchmark_report_markdown", + precision="mixed", + ) ) manifest = build_manifest( config, artifacts, - commit=framework_commit(), + commit=commit, toolchain=installed, ) manifest_path = destination / MANIFEST_NAME diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py new file mode 100644 index 0000000..f76e747 --- /dev/null +++ b/scripts/check_onnx_quantized_artifacts.py @@ -0,0 +1,570 @@ +"""Validate Vision v8 quantized ONNX artifact metadata and reports.""" + +from __future__ import annotations + +import hashlib +import json +import math +import shutil +import string +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + + +def _load_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"JSON root must be an object: {path}") + return payload + + +def load_quantization_thresholds(path: Path) -> dict[str, Any]: + """Load and validate the checked-in quantized artifact gate thresholds.""" + + config = _load_json(path) + if "release_policy" not in config: + raise ValueError("threshold config missing release_policy") + precisions = config.get("precisions") + if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: + raise ValueError("threshold config must define fp16 and int8 precisions") + return config + + +def load_calibration_manifest(path: Path) -> dict[str, Any]: + """Load and validate the pinned INT8 calibration manifest contract.""" + + manifest = _load_json(path) + dataset = manifest.get("dataset") + if not isinstance(dataset, dict): + raise ValueError("calibration manifest missing dataset") + for key in ("image_ids_sha256", "annotations_sha256"): + if not _is_sha256(dataset.get(key)): + raise ValueError(f"calibration manifest dataset.{key} must be a SHA-256") + if "disjoint_from" not in dataset: + raise ValueError("calibration manifest dataset.disjoint_from must be declared") + + quantization = manifest.get("quantization") + if not isinstance(quantization, dict): + raise ValueError("calibration manifest missing quantization settings") + required = { + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + } + missing = sorted(required - set(quantization)) + if missing: + raise ValueError(f"calibration manifest missing settings: {missing}") + image_ids = manifest.get("image_ids") + if not isinstance(image_ids, Sequence) or isinstance(image_ids, (str, bytes)): + raise ValueError("calibration manifest image_ids must be a sequence") + if not image_ids: + raise ValueError("calibration manifest image_ids must not be empty") + actual_digest = image_id_manifest_sha256([int(image_id) for image_id in image_ids]) + expected_digest = str(dataset["image_ids_sha256"]) + if actual_digest != expected_digest: + raise ValueError("calibration manifest dataset.image_ids_sha256 does not match image_ids") + + images = manifest.get("images") + if not isinstance(images, Sequence) or isinstance(images, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence") + if not images: + raise ValueError("calibration manifest images must not be empty") + manifest["images"] = [ + str(_resolve_manifest_path(path.parent, Path(str(image)))) for image in images + ] + return manifest + + +def materialize_calibration_manifest_images( + manifest_path: Path, + *, + artifact_root: Path, + output_manifest_path: Path | None = None, + image_directory_name: str = "calibration_images", +) -> None: + """Copy calibration images beside an evidence manifest and rewrite paths.""" + + manifest = _load_json(manifest_path) + images = manifest.get("images") + if not isinstance(images, Sequence) or isinstance(images, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence") + + destination_dir = artifact_root / image_directory_name + destination_dir.mkdir(parents=True, exist_ok=True) + rewritten: list[str] = [] + for index, image in enumerate(images): + source = _resolve_manifest_path(manifest_path.parent, Path(str(image))) + if not source.is_file(): + raise ValueError(f"calibration image does not exist: {source}") + destination = destination_dir / f"{index:06d}_{source.name}" + shutil.copyfile(source, destination) + rewritten.append(destination.relative_to(artifact_root).as_posix()) + + manifest["images"] = rewritten + output_path = output_manifest_path or manifest_path + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") + + +def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: + """Hash a stable sorted image-ID manifest for calibration/eval pinning.""" + + digest = hashlib.sha256() + for image_id in sorted(map(int, image_ids)): + digest.update(f"{image_id}\n".encode("utf-8")) + return digest.hexdigest() + + +def _resolve_manifest_path(base: Path, path: Path) -> Path: + return path if path.is_absolute() else base / path + + +def assert_disjoint_image_ids( + calibration_ids: set[int], + evaluation_ids: set[int], +) -> None: + """Fail when INT8 calibration inputs overlap accuracy-gate inputs.""" + + overlap = calibration_ids & evaluation_ids + if overlap: + preview = sorted(overlap)[:10] + raise ValueError(f"calibration/evaluation image ID overlap: {preview}") + + +def evaluation_image_ids_from_report(report: Mapping[str, Any]) -> set[int]: + """Extract the actual evaluated image IDs from an accuracy report.""" + + raw_ids = report.get("evaluation_image_ids") + if raw_ids is None: + raw_ids = _mapping(report.get("dataset")).get("image_ids") + if not isinstance(raw_ids, Sequence) or isinstance(raw_ids, (str, bytes)): + raise ValueError("accuracy report must include evaluation image IDs") + if not raw_ids: + raise ValueError("accuracy report evaluation image IDs must not be empty") + return {int(image_id) for image_id in raw_ids} + + +def check_accuracy_artifact_bindings( + report: Mapping[str, Any], + generated_artifacts: Mapping[str, Mapping[str, Mapping[str, str]]], +) -> list[str]: + """Verify accuracy evidence hashes match the artifacts being published.""" + + branches = _mapping(report.get("branches")) + failures: list[str] = [] + for branch, precision_hashes in generated_artifacts.items(): + branch_report = _mapping(branches.get(branch)) + if not branch_report: + failures.append(f"accuracy report missing branch {branch}") + continue + for precision, expected_hashes in precision_hashes.items(): + precision_report = _mapping(branch_report.get(precision)) + if not precision_report: + failures.append(f"accuracy report missing {branch} {precision}") + continue + for field, expected in expected_hashes.items(): + actual = precision_report.get(field) + if actual != expected: + failures.append( + f"{branch} {precision} {field} {actual or 'missing'} " + f"does not match generated {expected}" + ) + return failures + + +def check_provider_precision_supported( + provider: str, + precision: str, + thresholds: Mapping[str, Any], +) -> None: + """Fail clearly when a provider/precision pair is not release-supported.""" + + providers = thresholds.get("providers", {}) + if not isinstance(providers, Mapping) or provider not in providers: + raise ValueError(f"{provider} is not configured in quantization release config") + supported = providers[provider] + if not isinstance(supported, Sequence) or isinstance(supported, (str, bytes)): + raise ValueError(f"{provider} precision policy must be a sequence") + if precision not in supported: + raise ValueError(f"{provider} does not support {precision} in quantization release config") + + +def check_unexpected_fp32_nodes( + dtype_report: Mapping[str, Any], + allowlist: Sequence[str], +) -> list[str]: + """Return FP32 nodes not covered by an explicit op-type allowlist.""" + + allowed = set(allowlist) + unexpected: list[str] = [] + fp32_nodes = dtype_report.get("fp32_nodes", []) + if not isinstance(fp32_nodes, Sequence) or isinstance(fp32_nodes, (str, bytes)): + raise ValueError("dtype report fp32_nodes must be a sequence") + for node in fp32_nodes: + if not isinstance(node, Mapping): + raise ValueError("dtype report fp32_nodes entries must be objects") + name = str(node.get("name", "")) + op_type = str(node.get("op_type", "")) + if op_type not in allowed: + unexpected.append(f"{name}:{op_type}") + return unexpected + + +def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: + """Inventory node-level dtype hints in an ONNX graph. + + ONNX does not assign one dtype directly to every node, so this inspector + maps typed graph values back to producer nodes and reports nodes producing + FP32, FP16, or INT8/UINT8 tensors. + """ + + try: + import onnx + from onnx import TensorProto + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError("ONNX dtype inspection requires the onnx package") from error + + model = onnx.load(str(model_path)) + value_dtypes: dict[str, int] = {} + for value_info in [ + *model.graph.input, + *model.graph.output, + *model.graph.value_info, + *model.graph.initializer, + ]: + name = getattr(value_info, "name", "") + data_type = None + if hasattr(value_info, "data_type"): + data_type = value_info.data_type + elif getattr(value_info, "type", None) is not None: + tensor_type = value_info.type.tensor_type + data_type = tensor_type.elem_type + if name and data_type: + value_dtypes[name] = int(data_type) + + fp32_nodes: list[dict[str, str]] = [] + fp16_nodes = 0 + int8_nodes = 0 + for node in model.graph.node: + output_types = {value_dtypes[name] for name in node.output if name in value_dtypes} + if TensorProto.FLOAT in output_types: + fp32_nodes.append({"name": node.name or node.output[0], "op_type": node.op_type}) + if TensorProto.FLOAT16 in output_types: + fp16_nodes += 1 + if TensorProto.INT8 in output_types or TensorProto.UINT8 in output_types: + int8_nodes += 1 + + return { + "fp32_nodes": fp32_nodes, + "fp16_nodes": fp16_nodes, + "int8_nodes": int8_nodes, + } + + +def check_quantized_accuracy_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str] = (), + expected_artifacts: Mapping[str, Mapping[str, Mapping[str, str]]] | None = None, + expected_framework_commit: str | None = None, +) -> list[str]: + """Compare a quantized candidate COCO report against its FP32 reference.""" + + if "branches" in report: + failures = _check_framework_commit(report, expected_framework_commit) + failures.extend( + _check_branch_accuracy_report( + report, + thresholds, + required_branches=required_branches, + ) + ) + failures.extend(_check_provider_contract(report, thresholds)) + release_policy = _mapping(thresholds.get("release_policy")) + if release_policy.get("require_artifact_bindings") is True: + if expected_artifacts is None: + failures.append("generated artifact bindings are required by release policy") + else: + failures.extend(check_accuracy_artifact_bindings(report, expected_artifacts)) + return failures + + reference = _mapping(report.get("reference")) + candidate = _mapping(report.get("candidate")) + failures = _check_framework_commit(report, expected_framework_commit) + failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) + return failures + + +def check_quantized_parity_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + """Gate raw-logit and decoded-output drift against checked-in thresholds.""" + + precision = str(report.get("precision", "")) + branch = str(report.get("branch", "")) + precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) + if not precision_thresholds: + return [f"candidate precision {precision} has no quantization thresholds"] + + failures: list[str] = [] + for metric, threshold_name in ( + ("max_raw_logit_abs_error", "max_raw_logit_abs_error"), + ("max_decoded_box_px_error", "max_decoded_box_px_error"), + ("max_score_abs_error", "max_score_abs_error"), + ): + if metric not in report: + failures.append(f"missing parity metric {metric}") + continue + value = _finite_float(report[metric]) + if value is None: + failures.append(f"non-finite parity metric {metric}") + continue + allowed = float(precision_thresholds[threshold_name]) + if value > allowed: + failures.append(f"{precision} {branch} {metric} {value:.6f} exceeds {allowed:.6f}") + return failures + + +def check_quantized_benchmark_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str], +) -> list[str]: + """Validate benchmark evidence covers every release branch and precision.""" + + branches = _mapping(report.get("branches")) + if not branches: + return ["benchmark report must contain branches"] + required_precisions = _required_precisions(thresholds) + required_fields = tuple(_mapping(thresholds.get("benchmark")).get("report", ())) + failures: list[str] = [] + for branch in required_branches: + branch_report = _mapping(branches.get(branch)) + if not branch_report: + failures.append(f"benchmark report missing branch {branch}") + continue + for precision in required_precisions: + precision_report = _mapping(branch_report.get(precision)) + if not precision_report: + failures.append(f"benchmark report missing {branch} {precision}") + continue + for field in required_fields: + value = _benchmark_field(precision_report, str(field)) + if _finite_float(value) is None: + failures.append(f"benchmark report has invalid {branch} {precision} {field}") + return failures + + +def _check_branch_accuracy_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str], +) -> list[str]: + branches = _mapping(report.get("branches")) + if not branches: + return ["quantized COCO report must contain branch comparisons"] + reference_precision = str(report.get("reference_precision", "fp32")) + threshold_precisions = set(_mapping(thresholds.get("precisions"))) + candidate_precisions = _candidate_precisions(report, branches, threshold_precisions) + required_candidates = tuple( + precision + for precision in _required_precisions(thresholds) + if precision != reference_precision and precision in threshold_precisions + ) + if required_candidates: + candidate_precisions = required_candidates + + failures: list[str] = [] + if not candidate_precisions: + return ["candidate precision missing from quantized COCO report"] + for branch in required_branches: + if branch not in branches: + failures.append(f"quantized COCO report missing branch {branch}") + for branch, branch_report in branches.items(): + branch_data = _mapping(branch_report) + for candidate_precision in candidate_precisions: + reference = _mapping(branch_data.get(reference_precision, branch_data.get("reference"))) + candidate = _mapping(branch_data.get(candidate_precision, branch_data.get("candidate"))) + if not reference or not candidate: + failures.append( + f"{branch} missing {reference_precision} or {candidate_precision} metrics" + ) + continue + failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) + return failures + + +def _check_framework_commit( + report: Mapping[str, Any], + expected_framework_commit: str | None, +) -> list[str]: + if expected_framework_commit is None: + return [] + actual = str(report.get("framework_commit", "")) + if actual != expected_framework_commit: + return [ + f"accuracy report framework_commit {actual or 'missing'}, " + f"expected {expected_framework_commit}" + ] + return [] + + +def _check_provider_contract( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + release_policy = _mapping(thresholds.get("release_policy")) + provider_by_precision = _mapping(release_policy.get("provider_by_precision")) + if not provider_by_precision: + return [] + + branches = _mapping(report.get("branches")) + failures: list[str] = [] + for branch, branch_report in branches.items(): + branch_data = _mapping(branch_report) + for precision, expected_provider_value in provider_by_precision.items(): + expected_provider = str(expected_provider_value) + precision_report = _mapping(branch_data.get(str(precision))) + if not precision_report: + continue + environment = _mapping(precision_report.get("environment")) + if not environment: + failures.append(f"{branch} {precision} missing environment provider metadata") + continue + + requested = _provider_values(environment.get("requested_provider")) + actual = str(environment.get("actual_provider", "")) + if not requested: + failures.append(f"{branch} {precision} missing requested_provider") + elif requested[0] != expected_provider: + failures.append( + f"{branch} {precision} requested_provider {requested[0]}, " + f"expected {expected_provider}" + ) + if actual != expected_provider: + failures.append( + f"{branch} {precision} actual_provider {actual or 'missing'}, " + f"expected {expected_provider}" + ) + return failures + + +def _required_precisions(thresholds: Mapping[str, Any]) -> tuple[str, ...]: + release_policy = _mapping(thresholds.get("release_policy")) + raw_precisions = release_policy.get("required_precisions", ()) + if not isinstance(raw_precisions, Sequence) or isinstance( + raw_precisions, + (str, bytes), + ): + return () + return tuple(str(precision) for precision in raw_precisions) + + +def _benchmark_field(report: Mapping[str, Any], field: str) -> object: + if field in report: + return report[field] + latency = _mapping(report.get("latency")) + return latency.get(field) + + +def _candidate_precisions( + report: Mapping[str, Any], + branches: Mapping[str, Any], + threshold_precisions: set[str], +) -> tuple[str, ...]: + raw_candidate_precisions = report.get("candidate_precisions") + if isinstance(raw_candidate_precisions, Sequence) and not isinstance( + raw_candidate_precisions, + (str, bytes), + ): + candidates = {str(precision) for precision in raw_candidate_precisions} + else: + candidates = set() + raw_candidate_precision = report.get("candidate_precision", report.get("precision")) + if raw_candidate_precision is not None: + candidates.add(str(raw_candidate_precision)) + for branch_report in branches.values(): + branch_data = _mapping(branch_report) + candidates.update(str(key) for key in branch_data if str(key) in threshold_precisions) + return tuple(sorted(candidates)) + + +def _provider_values(value: object) -> tuple[str, ...]: + if isinstance(value, Sequence) and not isinstance(value, (str, bytes)): + return tuple(str(provider) for provider in value) + if value is None: + return () + return (str(value),) + + +def _check_accuracy_pair( + reference: Mapping[str, Any], + candidate: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + reference_branch = str(reference.get("branch", "")) + candidate_branch = str(candidate.get("branch", "")) + if reference_branch != candidate_branch: + return [ + "candidate branch " + f"{candidate_branch} does not match FP32 reference branch " + f"{reference_branch}" + ] + + precision = str(candidate.get("precision", "")) + precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) + if not precision_thresholds: + return [f"candidate precision {precision} has no quantization thresholds"] + + reference_metrics = _mapping(reference.get("metrics")) + candidate_metrics = _mapping(candidate.get("metrics")) + failures: list[str] = [] + for metric, threshold_name in ( + ("map50_95", "max_map50_95_drop"), + ("map50", "max_map50_drop"), + ): + if metric not in reference_metrics or metric not in candidate_metrics: + failures.append(f"missing metric {metric} in FP32 or candidate report") + continue + reference_value = _finite_float(reference_metrics[metric]) + candidate_value = _finite_float(candidate_metrics[metric]) + if reference_value is None or candidate_value is None: + failures.append(f"non-finite metric {metric} in FP32 or candidate report") + continue + allowed_drop = float(precision_thresholds[threshold_name]) + drop = reference_value - candidate_value + if drop > allowed_drop: + failures.append( + f"{precision} {candidate_branch} {metric} dropped by {drop:.6f}; " + f"allowed drop {allowed_drop:.6f}" + ) + return failures + + +def _mapping(value: object) -> Mapping[str, Any]: + return value if isinstance(value, Mapping) else {} + + +def _finite_float(value: object) -> float | None: + try: + converted = float(value) + except (TypeError, ValueError): + return None + if not math.isfinite(converted): + return None + return converted + + +def _is_sha256(value: object) -> bool: + if not isinstance(value, str) or len(value) != 64: + return False + return all(character in string.hexdigits for character in value) diff --git a/scripts/check_onnx_quantized_parity.py b/scripts/check_onnx_quantized_parity.py new file mode 100644 index 0000000..bf988c7 --- /dev/null +++ b/scripts/check_onnx_quantized_parity.py @@ -0,0 +1,177 @@ +"""Gate FP32-vs-quantized Vision v8 ONNX raw and decoded parity.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Sequence + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--reference-model", type=Path, required=True) + parser.add_argument("--candidate-model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--image", type=Path, required=True) + parser.add_argument("--precision", choices=("fp16", "int8"), required=True) + parser.add_argument("--thresholds", type=Path, required=True) + parser.add_argument( + "--provider", + action="append", + default=None, + help="Provider alias or ORT provider name. May be repeated or comma-separated.", + ) + parser.add_argument("--output", type=Path) + return parser.parse_args() + + +def decoded_parity_metrics( + reference_detections: Sequence[object], + candidate_detections: Sequence[object], +) -> dict[str, float | int]: + """Compare decoded detections in stable score order.""" + + count = min(len(reference_detections), len(candidate_detections)) + if count == 0: + return { + "reference_detection_count": len(reference_detections), + "candidate_detection_count": len(candidate_detections), + "class_mismatch_count": int(len(reference_detections) != len(candidate_detections)), + "max_decoded_box_px_error": 0.0, + "max_score_abs_error": 0.0, + } + + box_errors = [] + score_errors = [] + class_mismatches = abs(len(reference_detections) - len(candidate_detections)) + for reference, candidate in zip( + reference_detections[:count], + candidate_detections[:count], + ): + box_errors.append( + float( + np.max( + np.abs( + np.asarray(reference.box_pixel, dtype=np.float32) + - np.asarray(candidate.box_pixel, dtype=np.float32) + ) + ) + ) + ) + score_errors.append(abs(float(reference.score) - float(candidate.score))) + class_mismatches += int(reference.class_id != candidate.class_id) + + return { + "reference_detection_count": len(reference_detections), + "candidate_detection_count": len(candidate_detections), + "class_mismatch_count": class_mismatches, + "max_decoded_box_px_error": max(box_errors), + "max_score_abs_error": max(score_errors), + } + + +def build_parity_report( + *, + reference_model: Path, + candidate_model: Path, + metadata_path: Path, + image_path: Path, + precision: str, + providers: Sequence[str], +) -> dict[str, object]: + from complexity.deploy.onnx_detector import OnnxDetectorPipeline + from complexity.deploy.onnx_detector.metadata import load_metadata + from complexity.deploy.onnx_detector.preprocess import preprocess_image + + metadata = load_metadata(metadata_path) + preprocessed = preprocess_image(image_path, metadata.image_size) + + reference_session = OnnxDetectorPipeline.create_session( + reference_model, + providers=providers, + warmup_runs=0, + ).open() + candidate_session = OnnxDetectorPipeline.create_session( + candidate_model, + providers=providers, + warmup_runs=0, + ).open() + + reference_raw = reference_session.run(preprocessed.pixel_values) + candidate_raw = candidate_session.run(preprocessed.pixel_values) + reference_pipeline = OnnxDetectorPipeline( + metadata=metadata, + session=reference_session, + ) + candidate_pipeline = OnnxDetectorPipeline( + metadata=metadata, + session=candidate_session, + ) + reference_result = reference_pipeline.predict(image_path) + candidate_result = candidate_pipeline.predict(image_path) + + return { + "schema_version": 1, + "branch": metadata.branch, + "precision": precision, + "provider_used": candidate_session.provider_used, + "max_raw_logit_abs_error": float(np.max(np.abs(reference_raw - candidate_raw))), + **decoded_parity_metrics( + reference_result.detections, + candidate_result.detections, + ), + } + + +def main() -> None: + from scripts.check_onnx_quantized_artifacts import ( + check_provider_precision_supported, + check_quantized_parity_report, + load_quantization_thresholds, + ) + from scripts.onnx_detect import provider_names + + args = parse_args() + thresholds = load_quantization_thresholds(args.thresholds) + providers = provider_names(args.provider) + report = build_parity_report( + reference_model=args.reference_model, + candidate_model=args.candidate_model, + metadata_path=args.metadata, + image_path=args.image, + precision=args.precision, + providers=providers, + ) + check_provider_precision_supported( + str(report["provider_used"]), + args.precision, + thresholds, + ) + failures = check_quantized_parity_report(report, thresholds) + if int(report["class_mismatch_count"]) > 0: + failures.append( + f"{report['branch']} {args.precision} decoded class/count mismatch: " + f"{report['class_mismatch_count']}" + ) + + if args.output is not None: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + if failures: + print("Quantized parity FAILED:", file=sys.stderr) + for failure in failures: + print(f" {failure}", file=sys.stderr) + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/scripts/evaluate_onnx_coco.py b/scripts/evaluate_onnx_coco.py index 7e709ee..9d6c059 100644 --- a/scripts/evaluate_onnx_coco.py +++ b/scripts/evaluate_onnx_coco.py @@ -56,6 +56,12 @@ def parse_args() -> argparse.Namespace: help="Provider alias or ORT provider name. May be repeated or comma-separated.", ) parser.add_argument("--seed", type=int, default=0) + parser.add_argument( + "--precision", + choices=("fp32", "fp16", "int8"), + default="fp32", + help="artifact precision label recorded for quantized release gates", + ) parser.add_argument("--confidence", type=float, default=None) parser.add_argument("--nms-iou", type=float, default=None) parser.add_argument("--max-detections", type=int, default=None) @@ -198,6 +204,7 @@ def main() -> None: "schema_version": 1, "format_version": 1, "backend": "onnx", + "precision": args.precision, "framework_commit": _framework_commit(), "checkpoint": str(args.model), "checkpoint_sha256": _checkpoint_sha256(args.model), @@ -209,6 +216,7 @@ def main() -> None: "split": "val2017", "images": len(image_ids), "evaluated_images": len(image_ids), + "image_ids": image_ids, "annotations": str(args.annotations), "annotations_sha256": _sha256_file(args.annotations), "image_list_sha256": _image_list_sha256(coco, image_ids), @@ -225,6 +233,7 @@ def main() -> None: "branches": { branch: { "branch": branch, + "precision": args.precision, "contract": _branch_contract(branch), "predictions": str(prediction_path), "detections": len(predictions), diff --git a/scripts/evaluate_tr_hash_coco.py b/scripts/evaluate_tr_hash_coco.py index a490441..2255628 100644 --- a/scripts/evaluate_tr_hash_coco.py +++ b/scripts/evaluate_tr_hash_coco.py @@ -18,7 +18,7 @@ import time from contextlib import nullcontext from pathlib import Path -from typing import Any, Iterable +from typing import Any, Iterable, Mapping import numpy as np import torch @@ -338,23 +338,23 @@ def _write_markdown_report(report: dict[str, Any], path: Path) -> None: "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |", ] for branch, branch_report in report["branches"].items(): - metrics = branch_report["metrics"] - lines.append( - "| " - + " | ".join( - [ - branch, - f"{float(metrics['map50_95']):.6f}", - f"{float(metrics['map50']):.6f}", - f"{float(metrics['map75']):.6f}", - f"{float(metrics['ap_small']):.6f}", - f"{float(metrics['ap_medium']):.6f}", - f"{float(metrics['ap_large']):.6f}", - f"{float(metrics['ar_100']):.6f}", - ] + for label, metrics in _branch_metric_rows(branch, branch_report): + lines.append( + "| " + + " | ".join( + [ + label, + f"{float(metrics['map50_95']):.6f}", + f"{float(metrics['map50']):.6f}", + f"{float(metrics['map75']):.6f}", + f"{float(metrics['ap_small']):.6f}", + f"{float(metrics['ap_medium']):.6f}", + f"{float(metrics['ap_large']):.6f}", + f"{float(metrics['ar_100']):.6f}", + ] + ) + + " |" ) - + " |" - ) lines.extend( [ "", @@ -373,6 +373,21 @@ def _write_markdown_report(report: dict[str, Any], path: Path) -> None: path.write_text("\n".join(lines), encoding="utf-8") +def _branch_metric_rows( + branch: str, + branch_report: Mapping[str, Any], +) -> list[tuple[str, Mapping[str, Any]]]: + if "metrics" in branch_report: + return [(branch, branch_report["metrics"])] + + rows: list[tuple[str, Mapping[str, Any]]] = [] + for precision, precision_report in branch_report.items(): + if not isinstance(precision_report, Mapping) or "metrics" not in precision_report: + continue + rows.append((f"{branch} {precision}", precision_report["metrics"])) + return rows + + def main() -> None: args = parse_args() if args.batch_size <= 0 or args.warmup < 0 or args.max_detections <= 0: @@ -415,6 +430,7 @@ def main() -> None: "split": "val2017", "images": len(image_ids), "evaluated_images": len(image_ids), + "image_ids": image_ids, "annotations": str(args.annotations), "annotations_sha256": _sha256_file(args.annotations), "image_list_sha256": _image_list_sha256(coco, image_ids), diff --git a/scripts/merge_vision_v8_coco_reports.py b/scripts/merge_vision_v8_coco_reports.py index 90a6a47..d0715d2 100644 --- a/scripts/merge_vision_v8_coco_reports.py +++ b/scripts/merge_vision_v8_coco_reports.py @@ -35,6 +35,10 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: merged["metadata"] = "multiple ONNX branch sidecars" merged["branches"] = {} shared_keys = ("backend", "framework_commit", "dataset", "protocol") + precision_reports = [_report_precision(report) for report in reports] + nest_by_precision = any(precision is not None for precision in precision_reports) + if nest_by_precision and not all(precision_reports): + raise ValueError("cannot merge mixed precision and non-precision reports") for report in reports: for key in shared_keys: if report.get(key) != reports[0].get(key): @@ -44,24 +48,53 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: if not isinstance(branches, Mapping): raise ValueError("each report must contain a branches object") for branch, branch_report in branches.items(): - if branch in merged["branches"]: - raise ValueError(f"duplicate branch in merged reports: {branch}") if not isinstance(branch_report, Mapping): raise ValueError(f"branch report must be an object: {branch}") merged_branch = deepcopy(dict(branch_report)) for key in ( "checkpoint", "checkpoint_sha256", + "environment", "model", "metadata", "metadata_sha256", ): merged_branch.setdefault(key, report.get(key)) - merged["branches"][branch] = merged_branch + precision = _report_precision(report) + if nest_by_precision: + assert precision is not None + merged_branch.setdefault("precision", precision) + branch_precisions = merged["branches"].setdefault(branch, {}) + if precision in branch_precisions: + raise ValueError(f"duplicate precision in merged reports: {branch} {precision}") + branch_precisions[precision] = merged_branch + else: + if branch in merged["branches"]: + raise ValueError(f"duplicate branch in merged reports: {branch}") + merged["branches"][branch] = merged_branch + + if nest_by_precision: + precisions = tuple(dict.fromkeys(str(precision) for precision in precision_reports)) + if "fp32" in precisions: + merged["reference_precision"] = "fp32" + merged["candidate_precisions"] = [ + precision for precision in precisions if precision != "fp32" + ] + else: + merged["candidate_precisions"] = list(precisions) return merged +def _report_precision(report: Mapping[str, Any]) -> str | None: + precision = report.get("precision") + if precision is None: + protocol = report.get("protocol") + if isinstance(protocol, Mapping): + precision = protocol.get("precision") + return str(precision) if precision is not None else None + + def main() -> None: args = parse_args() report = merge_reports([load_json(path) for path in args.reports]) diff --git a/scripts/quantize_onnx.py b/scripts/quantize_onnx.py new file mode 100644 index 0000000..3fcb1a5 --- /dev/null +++ b/scripts/quantize_onnx.py @@ -0,0 +1,376 @@ +"""Create reproducible FP16 and INT8 Vision v8 ONNX artifacts.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import platform +import shutil +import subprocess +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any, Literal + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +Precision = Literal["fp32", "fp16", "int8"] + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--fp32-model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--precision", choices=("fp16", "int8"), required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument( + "--sidecar", + type=Path, + help="Quantization provenance sidecar. Defaults to .quantization.json.", + ) + parser.add_argument( + "--detector-metadata-output", + type=Path, + help=("Detector metadata sidecar copied from --metadata. Defaults to .json."), + ) + parser.add_argument("--calibration-manifest", type=Path) + parser.add_argument("--checkpoint-revision", default="unknown") + parser.add_argument( + "--keep-fp32-op-type", + action="append", + default=[], + help="FP16 conversion allowlist; may be repeated", + ) + parser.add_argument("--disable-shape-infer", action="store_true") + parser.add_argument("--repeat-output", type=Path) + parser.add_argument("--require-identical-hash", action="store_true") + return parser.parse_args() + + +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def framework_commit() -> str: + try: + return subprocess.check_output( + ["git", "rev-parse", "HEAD"], + stderr=subprocess.DEVNULL, + text=True, + ).strip() + except (OSError, subprocess.CalledProcessError): + return "unknown" + + +def package_version(module_name: str) -> str | None: + try: + from importlib import metadata + + return metadata.version(module_name) + except metadata.PackageNotFoundError: + return None + + +def toolchain_versions() -> dict[str, Any]: + return { + "python": sys.version.split()[0], + "os": platform.platform(), + "onnx": package_version("onnx"), + "onnxruntime": package_version("onnxruntime"), + "onnxconverter-common": package_version("onnxconverter-common"), + "numpy": np.__version__, + } + + +def assert_identical_artifact_hashes(first_sha256: str, second_sha256: str) -> None: + if first_sha256 != second_sha256: + raise ValueError( + "quantization is not deterministic: " + f"first SHA-256 {first_sha256} differs from repeat SHA-256 {second_sha256}" + ) + + +def write_quantization_sidecar( + path: Path, + *, + precision: Precision, + source_model_sha256: str, + output_model_sha256: str, + framework_commit: str, + checkpoint_revision: str, + settings: Mapping[str, Any], + toolchain: Mapping[str, Any], +) -> None: + payload = { + "schema_version": 1, + "artifact_type": "vision_v8_quantized_onnx", + "precision": precision, + "framework_commit": framework_commit, + "checkpoint_revision": checkpoint_revision, + "source_model_sha256": source_model_sha256, + "output_model_sha256": output_model_sha256, + "settings": dict(settings), + "toolchain": dict(toolchain), + } + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") + + +def default_quantization_sidecar(output_model: Path) -> Path: + return output_model.with_name(f"{output_model.stem}.quantization.json") + + +def default_detector_metadata_output(output_model: Path) -> Path: + return output_model.with_suffix(".json") + + +def copy_detector_metadata(source: Path, destination: Path) -> None: + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, destination) + + +def quantize_fp32(input_model: Path, output_model: Path) -> None: + output_model.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(input_model, output_model) + + +def quantize_fp16( + input_model: Path, + output_model: Path, + *, + keep_fp32_op_types: Sequence[str] = (), + disable_shape_infer: bool = False, +) -> None: + try: + import onnx + from onnxruntime.transformers import float16 + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError( + "FP16 quantization requires ONNX Runtime transformer tools; " + "install the export/quantization extra." + ) from error + + output_model.parent.mkdir(parents=True, exist_ok=True) + model = onnx.load(str(input_model)) + converted = float16.convert_float_to_float16( + model, + keep_io_types=True, + disable_shape_infer=disable_shape_infer, + op_block_list=list(keep_fp32_op_types), + ) + onnx.save(converted, str(output_model)) + + +class _CalibrationReader: + def __init__( + self, + *, + image_paths: Sequence[Path], + metadata_path: Path, + batch_size: int = 1, + input_name: str = "pixel_values", + ) -> None: + from complexity.deploy.onnx_detector.metadata import load_metadata + from complexity.deploy.onnx_detector.preprocess import preprocess_image + + metadata = load_metadata(metadata_path) + self._input_name = input_name + if batch_size <= 0: + raise ValueError("calibration batch_size must be positive") + self._batch_size = batch_size + self._items = [ + preprocess_image(path, metadata.image_size).pixel_values for path in image_paths + ] + self._index = 0 + + def get_next(self) -> dict[str, np.ndarray] | None: + if self._index >= len(self._items): + return None + batch = self._items[self._index : self._index + self._batch_size] + self._index += self._batch_size + return {self._input_name: np.concatenate(batch, axis=0)} + + +def _calibration_method(name: str) -> Any: + from onnxruntime.quantization import CalibrationMethod + + normalized = name.lower() + if normalized == "minmax": + return CalibrationMethod.MinMax + if normalized in {"entropy", "kl"}: + return CalibrationMethod.Entropy + if normalized == "percentile": + return CalibrationMethod.Percentile + raise ValueError(f"unsupported INT8 calibration method: {name}") + + +def _quant_type(name: str) -> Any: + from onnxruntime.quantization import QuantType + + normalized = name.lower() + if normalized == "qint8": + return QuantType.QInt8 + if normalized == "quint8": + return QuantType.QUInt8 + raise ValueError(f"unsupported ONNX Runtime quant type: {name}") + + +def _calibration_paths(manifest: Mapping[str, Any]) -> list[Path]: + raw_paths = manifest.get("images", []) + if not isinstance(raw_paths, Sequence) or isinstance(raw_paths, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence of paths") + return [Path(str(path)) for path in sorted(raw_paths, key=str)] + + +def quantize_int8( + input_model: Path, + output_model: Path, + *, + metadata_path: Path, + calibration_manifest: Mapping[str, Any], +) -> None: + try: + from onnxruntime.quantization import QuantFormat, quantize_static + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError( + "INT8 quantization requires onnxruntime.quantization and a calibration manifest." + ) from error + + settings = calibration_manifest["quantization"] + image_paths = _calibration_paths(calibration_manifest) + if not image_paths: + raise ValueError("INT8 quantization requires calibration manifest images") + reader = _CalibrationReader( + image_paths=image_paths, + metadata_path=metadata_path, + batch_size=int(settings["batch_size"]), + ) + output_model.parent.mkdir(parents=True, exist_ok=True) + quantize_static( + str(input_model), + str(output_model), + reader, + quant_format=QuantFormat.QDQ, + calibrate_method=_calibration_method(str(settings["calibration_method"])), + per_channel=bool(settings["per_channel"]), + activation_type=_quant_type(str(settings["activation_type"])), + weight_type=_quant_type(str(settings["weight_type"])), + extra_options={ + "ActivationSymmetric": bool(settings["symmetric_activations"]), + "WeightSymmetric": bool(settings["symmetric_weights"]), + }, + ) + + +def quantize_once( + *, + fp32_model: Path, + metadata_path: Path, + precision: Precision, + output_model: Path, + calibration_manifest: Mapping[str, Any] | None, + keep_fp32_op_types: Sequence[str], + disable_shape_infer: bool, +) -> dict[str, Any]: + if precision == "fp32": + quantize_fp32(fp32_model, output_model) + settings: dict[str, Any] = {} + elif precision == "fp16": + settings = { + "keep_fp32_op_types": list(keep_fp32_op_types), + "disable_shape_infer": disable_shape_infer, + } + quantize_fp16( + fp32_model, + output_model, + keep_fp32_op_types=keep_fp32_op_types, + disable_shape_infer=disable_shape_infer, + ) + else: + if calibration_manifest is None: + raise ValueError("INT8 quantization requires --calibration-manifest") + settings = dict(calibration_manifest["quantization"]) + settings = { + key: settings[key] + for key in ( + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + ) + } + quantize_int8( + fp32_model, + output_model, + metadata_path=metadata_path, + calibration_manifest=calibration_manifest, + ) + return settings + + +def main() -> None: + from scripts.check_onnx_quantized_artifacts import load_calibration_manifest + + args = parse_args() + calibration_manifest = ( + load_calibration_manifest(args.calibration_manifest) + if args.calibration_manifest is not None + else None + ) + settings = quantize_once( + fp32_model=args.fp32_model, + metadata_path=args.metadata, + precision=args.precision, + output_model=args.output, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=tuple(args.keep_fp32_op_type), + disable_shape_infer=args.disable_shape_infer, + ) + output_sha256 = sha256_file(args.output) + if args.repeat_output is not None: + quantize_once( + fp32_model=args.fp32_model, + metadata_path=args.metadata, + precision=args.precision, + output_model=args.repeat_output, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=tuple(args.keep_fp32_op_type), + disable_shape_infer=args.disable_shape_infer, + ) + repeat_sha256 = sha256_file(args.repeat_output) + if args.require_identical_hash: + assert_identical_artifact_hashes(output_sha256, repeat_sha256) + + detector_metadata_output = args.detector_metadata_output or default_detector_metadata_output( + args.output + ) + copy_detector_metadata(args.metadata, detector_metadata_output) + + sidecar = args.sidecar or default_quantization_sidecar(args.output) + write_quantization_sidecar( + sidecar, + precision=args.precision, + source_model_sha256=sha256_file(args.fp32_model), + output_model_sha256=output_sha256, + framework_commit=framework_commit(), + checkpoint_revision=args.checkpoint_revision, + settings=settings, + toolchain=toolchain_versions(), + ) + print(json.dumps({"output": str(args.output), "sha256": output_sha256}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index 64647fb..c773734 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -2,6 +2,7 @@ import pytest +from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report from scripts.evaluate_tr_hash_coco import ( BRANCHES, _branch_contract, @@ -69,9 +70,7 @@ def test_checkpoint_sha256_hashes_loaded_directory_weights(tmp_path): ema_weights.unlink() - assert _checkpoint_sha256(checkpoint) == hashlib.sha256( - b"model weights" - ).hexdigest() + assert _checkpoint_sha256(checkpoint) == hashlib.sha256(b"model weights").hexdigest() def test_branch_contract_distinguishes_nms_requirements(): @@ -156,6 +155,104 @@ def test_merge_onnx_reports_keeps_both_branch_artifact_hashes(): assert merged["branches"]["nms-free"]["metadata_sha256"] == "d" * 64 +def test_merge_onnx_reports_can_build_precision_nested_quantized_report(): + fp32 = { + "backend": "onnx", + "precision": "fp32", + "framework_commit": "abc123", + "checkpoint": "o2m_fp32.onnx", + "checkpoint_sha256": "a" * 64, + "model": "o2m_fp32.onnx", + "metadata": "o2m.json", + "metadata_sha256": "b" * 64, + "dataset": {"name": "coco-2017", "image_ids": [1, 2]}, + "environment": {}, + "protocol": {"seed": 0}, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.3}, + } + }, + } + fp16 = { + **fp32, + "precision": "fp16", + "checkpoint": "o2m_fp16.onnx", + "checkpoint_sha256": "c" * 64, + "model": "o2m_fp16.onnx", + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.295}, + } + }, + } + + merged = merge_reports([fp32, fp16]) + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16"]}, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + assert set(merged["branches"]["o2m-nms"]) == {"fp32", "fp16"} + assert merged["branches"]["o2m-nms"]["fp16"]["checkpoint_sha256"] == "c" * 64 + assert ( + check_quantized_accuracy_report( + merged, + thresholds, + required_branches=["o2m-nms"], + ) + == [] + ) + + +def test_merge_onnx_reports_preserves_environment_per_precision(): + fp32 = { + "backend": "onnx", + "precision": "fp32", + "framework_commit": "abc123", + "checkpoint": "o2m_fp32.onnx", + "checkpoint_sha256": "a" * 64, + "model": "o2m_fp32.onnx", + "metadata": "o2m.json", + "metadata_sha256": "b" * 64, + "dataset": {"name": "coco-2017", "image_ids": [1, 2]}, + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "protocol": {"seed": 0}, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.3}, + } + }, + } + fp16 = { + **fp32, + "precision": "fp16", + "checkpoint": "o2m_fp16.onnx", + "checkpoint_sha256": "c" * 64, + "environment": { + "requested_provider": ["CUDAExecutionProvider"], + "actual_provider": "CUDAExecutionProvider", + }, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.295}, + } + }, + } + + merged = merge_reports([fp32, fp16]) + + assert merged["branches"]["o2m-nms"]["fp32"]["environment"] == fp32["environment"] + assert merged["branches"]["o2m-nms"]["fp16"]["environment"] == fp16["environment"] + + def test_merge_onnx_reports_rejects_mismatched_protocol(): first = { "backend": "onnx", diff --git a/tests/test_onnx_quantization_accuracy_gate.py b/tests/test_onnx_quantization_accuracy_gate.py new file mode 100644 index 0000000..3aefeb5 --- /dev/null +++ b/tests/test_onnx_quantization_accuracy_gate.py @@ -0,0 +1,473 @@ +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + check_accuracy_artifact_bindings, + check_quantized_accuracy_report, + check_quantized_parity_report, + evaluation_image_ids_from_report, +) + + +def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "int8", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.17, "map50": 0.31}, + }, + } + thresholds = {"precisions": {"int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert any("map50_95" in failure for failure in failures) + assert not any("map50=" in failure for failure in failures) + + +def test_quantized_accuracy_accepts_candidate_within_threshold() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "nms-free", + "metrics": {"map50_95": 0.1, "map50": 0.14}, + }, + "candidate": { + "precision": "fp16", + "branch": "nms-free", + "metrics": {"map50_95": 0.098, "map50": 0.135}, + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + assert check_quantized_accuracy_report(report, thresholds) == [] + + +def test_quantized_accuracy_rejects_branch_mismatch() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "fp16", + "branch": "nms-free", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["candidate branch nms-free does not match FP32 reference branch o2m-nms"] + + +def test_quantized_accuracy_accepts_evaluator_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.315}, + }, + } + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + assert check_quantized_accuracy_report(report, thresholds) == [] + + +def test_quantized_accuracy_checks_every_precision_in_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + "int8": { + "precision": "int8", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.01, "map50": 0.02}, + }, + } + }, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, + } + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert any("int8 o2m-nms map50_95" in failure for failure in failures) + + +def test_quantized_accuracy_requires_release_policy_precisions() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16", "int8"]}, + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, + }, + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["o2m-nms missing fp32 or int8 metrics"] + + +def test_quantized_accuracy_rejects_empty_evaluator_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": {}, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["quantized COCO report must contain branch comparisons"] + + +def test_quantized_accuracy_requires_configured_branches() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.315}, + }, + } + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report( + report, + thresholds, + required_branches=["o2m-nms", "nms-free"], + ) + + assert "quantized COCO report missing branch nms-free" in failures + + +def test_quantized_accuracy_rejects_non_finite_metrics() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": float("nan"), "map50": 0.32}, + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["non-finite metric map50_95 in FP32 or candidate report"] + + +def test_quantized_parity_consumes_raw_and_decoded_thresholds() -> None: + report = { + "precision": "int8", + "branch": "o2m-nms", + "max_raw_logit_abs_error": 0.13, + "max_decoded_box_px_error": 3.0, + "max_score_abs_error": 0.04, + } + thresholds = { + "precisions": { + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + } + } + } + + failures = check_quantized_parity_report(report, thresholds) + + assert failures == ["int8 o2m-nms max_raw_logit_abs_error 0.130000 exceeds 0.120000"] + + +def test_accuracy_report_must_include_actual_evaluation_ids() -> None: + assert evaluation_image_ids_from_report({"dataset": {"image_ids": [3, 2, 2]}}) == {2, 3} + + +def test_accuracy_report_rejects_missing_evaluation_ids() -> None: + with pytest.raises(ValueError, match="evaluation image IDs"): + evaluation_image_ids_from_report({"dataset": {"disjoint_from": "train2017"}}) + + +def test_accuracy_report_artifact_hashes_must_match_generated_release_artifacts() -> None: + report = { + "branches": { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + } + generated = { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "e" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + + failures = check_accuracy_artifact_bindings(report, generated) + + assert failures == [ + "o2m-nms fp16 checkpoint_sha256 " + "cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc " + "does not match generated " + "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee" + ] + + +def test_release_accuracy_gate_requires_generated_artifact_bindings() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16"], + "require_artifact_bindings": True, + }, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["generated artifact bindings are required by release policy"] + + +def test_release_accuracy_gate_rejects_bogus_hashes_against_generated_artifacts() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16"], + "require_artifact_bindings": True, + }, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + generated = { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "e" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + + failures = check_quantized_accuracy_report( + report, + thresholds, + expected_artifacts=generated, + ) + + assert any("o2m-nms fp16 checkpoint_sha256" in failure for failure in failures) + + +def test_release_accuracy_gate_validates_provider_per_precision() -> None: + report = { + "reference_precision": "fp32", + "candidate_precisions": ["fp16", "int8"], + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["WrongExecutionProvider"], + "actual_provider": "WrongExecutionProvider", + }, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + "int8": { + "precision": "int8", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "metrics": {"map50_95": 0.19, "map50": 0.31}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "provider_by_precision": { + "fp32": "CPUExecutionProvider", + "fp16": "CUDAExecutionProvider", + "int8": "CPUExecutionProvider", + }, + }, + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, + }, + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == [ + "o2m-nms fp16 requested_provider WrongExecutionProvider, expected CUDAExecutionProvider", + "o2m-nms fp16 actual_provider WrongExecutionProvider, expected CUDAExecutionProvider", + ] + + +def test_release_accuracy_gate_binds_evidence_to_framework_commit() -> None: + report = { + "framework_commit": "old", + "reference_precision": "fp32", + "candidate_precisions": ["fp16"], + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16"]}, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + failures = check_quantized_accuracy_report( + report, + thresholds, + expected_framework_commit="new", + ) + + assert failures == ["accuracy report framework_commit old, expected new"] diff --git a/tests/test_onnx_quantization_artifact_checks.py b/tests/test_onnx_quantization_artifact_checks.py new file mode 100644 index 0000000..b7e86f0 --- /dev/null +++ b/tests/test_onnx_quantization_artifact_checks.py @@ -0,0 +1,33 @@ +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + check_provider_precision_supported, + check_unexpected_fp32_nodes, +) + + +def test_unsupported_provider_precision_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="does not support fp16"): + check_provider_precision_supported("CPUExecutionProvider", "fp16", thresholds) + + +def test_unknown_provider_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="not configured"): + check_provider_precision_supported("MagicExecutionProvider", "fp16", thresholds) + + +def test_unexpected_fp32_nodes_are_reported() -> None: + report = { + "fp32_nodes": [ + {"name": "Conv_1", "op_type": "Conv"}, + {"name": "ReduceSum_1", "op_type": "ReduceSum"}, + ] + } + + unexpected = check_unexpected_fp32_nodes(report, allowlist=["ReduceSum"]) + + assert unexpected == ["Conv_1:Conv"] diff --git a/tests/test_onnx_quantization_benchmark.py b/tests/test_onnx_quantization_benchmark.py new file mode 100644 index 0000000..ff66134 --- /dev/null +++ b/tests/test_onnx_quantization_benchmark.py @@ -0,0 +1,88 @@ +import numpy as np + +from scripts.benchmark_onnx_artifacts import _benchmark_session, summarize_latency_ms +from scripts.check_onnx_quantized_artifacts import check_quantized_benchmark_report + + +def test_benchmark_report_uses_distribution_not_single_shot() -> None: + summary = summarize_latency_ms([10.0, 12.0, 14.0]) + + assert summary["median_ms"] == 12.0 + assert summary["mean_ms"] == 12.0 + assert summary["stddev_ms"] > 0.0 + assert summary["p95_ms"] == 14.0 + + +def test_benchmark_report_handles_empty_measurements() -> None: + summary = summarize_latency_ms([]) + + assert summary["median_ms"] == 0.0 + assert summary["mean_ms"] == 0.0 + assert summary["stddev_ms"] == 0.0 + assert summary["p95_ms"] == 0.0 + + +def test_benchmark_report_requires_every_branch_and_precision() -> None: + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16", "int8"]}, + "benchmark": { + "report": [ + "median_ms", + "throughput_images_per_second", + "peak_memory_mb", + ] + }, + } + report = { + "branches": { + "o2m": { + "fp32": { + "latency": {"median_ms": 1.0}, + "throughput_images_per_second": 10.0, + "peak_memory_mb": 100.0, + } + } + } + } + + failures = check_quantized_benchmark_report( + report, + thresholds, + required_branches=["o2m", "nms-free"], + ) + + assert "benchmark report missing o2m fp16" in failures + assert "benchmark report missing branch nms-free" in failures + + +def test_benchmark_session_reports_observed_peak_memory( + monkeypatch, +) -> None: + memory_samples = iter([100.0, 110.0, 105.0, 125.0]) + + class Session: + def run(self, values): + assert values.shape == (1, 3, 4, 4) + return np.zeros((1, 1), dtype=np.float32) + + class Pipeline: + class Metadata: + image_size = 4 + + metadata = Metadata() + session = Session() + + monkeypatch.setattr( + "scripts.benchmark_onnx_artifacts._current_memory_mb", + lambda: next(memory_samples), + ) + + latencies, peak_memory_mb = _benchmark_session( + Pipeline(), + batch_size=1, + warmup_iterations=1, + measured_iterations=2, + ) + + assert len(latencies) == 2 + assert peak_memory_mb == 125.0 diff --git a/tests/test_onnx_quantization_cli.py b/tests/test_onnx_quantization_cli.py new file mode 100644 index 0000000..64831bd --- /dev/null +++ b/tests/test_onnx_quantization_cli.py @@ -0,0 +1,176 @@ +import json +import sys +import types +from pathlib import Path + +import pytest + +from scripts import quantize_onnx +from scripts.quantize_onnx import ( + assert_identical_artifact_hashes, + copy_detector_metadata, + default_detector_metadata_output, + default_quantization_sidecar, + quantize_int8, + write_quantization_sidecar, +) + + +def test_quantization_sidecar_binds_artifact_to_source_and_settings( + tmp_path: Path, +) -> None: + sidecar = tmp_path / "model.fp16.json" + + write_quantization_sidecar( + sidecar, + precision="fp16", + source_model_sha256="source", + output_model_sha256="output", + framework_commit="commit", + checkpoint_revision="checkpoint", + settings={"keep_fp32_op_types": ["ReduceSum"]}, + toolchain={"onnx": "1.17.0"}, + ) + + data = json.loads(sidecar.read_text(encoding="utf-8")) + assert data["precision"] == "fp16" + assert data["source_model_sha256"] == "source" + assert data["output_model_sha256"] == "output" + assert data["framework_commit"] == "commit" + assert data["checkpoint_revision"] == "checkpoint" + assert data["settings"]["keep_fp32_op_types"] == ["ReduceSum"] + + +def test_repeat_quantization_hash_mismatch_fails_loudly() -> None: + with pytest.raises(ValueError, match="quantization is not deterministic"): + assert_identical_artifact_hashes("first", "second") + + +def test_default_sidecars_keep_detector_metadata_separate(tmp_path: Path) -> None: + output = tmp_path / "tr_hash_v8_o2m_fp16.onnx" + source_metadata = tmp_path / "tr_hash_v8_o2m.json" + source_metadata.write_text('{"branch": "o2m", "image_size": 640}', encoding="utf-8") + + detector_sidecar = default_detector_metadata_output(output) + copy_detector_metadata(source_metadata, detector_sidecar) + + assert detector_sidecar == tmp_path / "tr_hash_v8_o2m_fp16.json" + assert default_quantization_sidecar(output) == ( + tmp_path / "tr_hash_v8_o2m_fp16.quantization.json" + ) + assert json.loads(detector_sidecar.read_text(encoding="utf-8"))["branch"] == "o2m" + + +def test_int8_quantization_passes_claimed_settings_to_ort( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: dict[str, object] = {} + + class QuantFormat: + QDQ = "QDQ" + + class CalibrationMethod: + MinMax = "MinMax" + Entropy = "Entropy" + Percentile = "Percentile" + + class QuantType: + QInt8 = "QInt8" + QUInt8 = "QUInt8" + + def quantize_static(*args: object, **kwargs: object) -> None: + calls["args"] = args + calls["kwargs"] = kwargs + Path(args[1]).write_bytes(b"int8") + + fake_quantization = types.SimpleNamespace( + CalibrationMethod=CalibrationMethod, + QuantFormat=QuantFormat, + QuantType=QuantType, + quantize_static=quantize_static, + ) + monkeypatch.setitem(sys.modules, "onnxruntime.quantization", fake_quantization) + monkeypatch.setattr( + quantize_onnx, + "_calibration_paths", + lambda _manifest: [tmp_path / "a.jpg", tmp_path / "b.jpg"], + ) + + class Reader: + def __init__(self, **kwargs: object) -> None: + calls["reader_kwargs"] = kwargs + + monkeypatch.setattr(quantize_onnx, "_CalibrationReader", Reader) + + manifest = { + "images": ["a.jpg", "b.jpg"], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + }, + } + + quantize_int8( + tmp_path / "fp32.onnx", + tmp_path / "int8.onnx", + metadata_path=tmp_path / "metadata.json", + calibration_manifest=manifest, + ) + + assert calls["reader_kwargs"] == { + "image_paths": [tmp_path / "a.jpg", tmp_path / "b.jpg"], + "metadata_path": tmp_path / "metadata.json", + "batch_size": 2, + } + assert calls["kwargs"]["per_channel"] is True + assert calls["kwargs"]["activation_type"] == "QUInt8" + assert calls["kwargs"]["weight_type"] == "QInt8" + assert calls["kwargs"]["extra_options"] == { + "ActivationSymmetric": False, + "WeightSymmetric": True, + } + + +def test_int8_sidecar_settings_only_include_applied_contract( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(quantize_onnx, "quantize_int8", lambda *_, **__: None) + manifest = { + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + "num_threads": 1, + } + } + + settings = quantize_onnx.quantize_once( + fp32_model=tmp_path / "fp32.onnx", + metadata_path=tmp_path / "metadata.json", + precision="int8", + output_model=tmp_path / "int8.onnx", + calibration_manifest=manifest, + keep_fp32_op_types=(), + disable_shape_infer=False, + ) + + assert settings == { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + } diff --git a/tests/test_onnx_quantization_config.py b/tests/test_onnx_quantization_config.py new file mode 100644 index 0000000..6073a7f --- /dev/null +++ b/tests/test_onnx_quantization_config.py @@ -0,0 +1,324 @@ +import json +from pathlib import Path + +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, + image_id_manifest_sha256, + load_calibration_manifest, + load_quantization_thresholds, + materialize_calibration_manifest_images, +) + +HASH = "a" * 64 +IMAGE_IDS = [9, 25, 30] +IMAGE_IDS_HASH = image_id_manifest_sha256(IMAGE_IDS) + + +def test_threshold_config_requires_explicit_release_policy(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text( + '{"schema_version": 1, "precisions": {"fp16": {}, "int8": {}}}', + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="release_policy"): + load_quantization_thresholds(path) + + +def test_threshold_config_requires_fp16_and_int8_policies(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text( + '{"schema_version": 1, "release_policy": {}, "precisions": {"fp16": {}}}', + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="fp16 and int8"): + load_quantization_thresholds(path) + + +def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + manifest = load_calibration_manifest(path) + + assert manifest["quantization"]["calibration_method"] == "minmax" + assert manifest["quantization"]["batch_size"] == 8 + + +def test_calibration_manifest_rejects_missing_quantization_setting( + tmp_path: Path, +) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8" + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="batch_size"): + load_calibration_manifest(path) + + +def test_calibration_manifest_rejects_placeholder_dataset_hash(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + % HASH, + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="image_ids_sha256"): + load_calibration_manifest(path) + + +def test_calibration_manifest_requires_pinned_image_identity(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + % (HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="image_ids"): + load_calibration_manifest(path) + + +def test_calibration_manifest_requires_calibration_image_paths(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="images"): + load_calibration_manifest(path) + + +def test_calibration_and_eval_image_ids_must_be_disjoint() -> None: + with pytest.raises(ValueError, match="overlap"): + assert_disjoint_image_ids({1, 2, 3}, {3, 4, 5}) + + +def test_image_id_manifest_hash_is_order_stable() -> None: + assert image_id_manifest_sha256([3, 1, 2]) == image_id_manifest_sha256([1, 2, 3]) + + +def test_materializes_calibration_images_and_rewrites_manifest_paths(tmp_path: Path) -> None: + source_dir = tmp_path / "source" + source_dir.mkdir() + first = source_dir / "000000000009.jpg" + second = source_dir / "000000000025.jpg" + first.write_bytes(b"first") + second.write_bytes(b"second") + manifest_path = tmp_path / "calibration.json" + manifest_path.write_text( + json.dumps( + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": image_id_manifest_sha256([9, 25]), + "annotations_sha256": HASH, + "disjoint_from": "coco-2017-val2017", + }, + "image_ids": [9, 25], + "images": [str(first), str(second)], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 1, + }, + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + materialize_calibration_manifest_images( + manifest_path, + artifact_root=tmp_path, + ) + manifest = load_calibration_manifest(manifest_path) + expected_first = tmp_path / "calibration_images" / "000000_000000000009.jpg" + expected_second = tmp_path / "calibration_images" / "000001_000000000025.jpg" + + assert manifest["images"] == [str(expected_first), str(expected_second)] + assert expected_first.read_bytes() == b"first" + assert expected_second.read_bytes() == b"second" + + +def test_materializes_calibration_manifest_to_separate_artifact_path( + tmp_path: Path, +) -> None: + source_dir = tmp_path / "source" + source_dir.mkdir() + (source_dir / "000000000009.jpg").write_bytes(b"first") + source_manifest = source_dir / "calibration.json" + artifact_root = tmp_path / "artifact" + output_manifest = artifact_root / "calibration.json" + source_manifest.write_text( + json.dumps( + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": image_id_manifest_sha256([9]), + "annotations_sha256": HASH, + "disjoint_from": "coco-2017-val2017", + }, + "image_ids": [9], + "images": ["000000000009.jpg"], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 1, + }, + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + materialize_calibration_manifest_images( + source_manifest, + artifact_root=artifact_root, + output_manifest_path=output_manifest, + ) + + assert source_manifest.read_text(encoding="utf-8") != output_manifest.read_text( + encoding="utf-8" + ) + assert load_calibration_manifest(output_manifest)["images"] == [ + str(artifact_root / "calibration_images" / "000000_000000000009.jpg") + ] diff --git a/tests/test_onnx_release.py b/tests/test_onnx_release.py index ca60084..a31a9d7 100644 --- a/tests/test_onnx_release.py +++ b/tests/test_onnx_release.py @@ -1,6 +1,8 @@ from __future__ import annotations import json +import subprocess +import sys from pathlib import Path from typing import Any @@ -43,7 +45,7 @@ def config_mapping(**overrides: Any) -> dict[str, Any]: "checkpoint_revision": "f3b3e659612e543ca9ff91892c0662d38dc1a1d6", "opset": 17, "parity_num_tests": 5, - "toolchain": {"torch": "2.13.0", "onnx": "1.21.0", "onnxruntime": "1.24.4"}, + "toolchain": {"torch": "2.13.0", "onnx": "1.21.0", "onnxruntime": "1.23.2"}, "branches": [ { "branch": "o2m", @@ -100,11 +102,88 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: assert {spec.branch for spec in config.branches} == {"o2m", "nms-free"} assert config.opset == 17 assert set(config.toolchain) == {"torch", "onnx", "onnxruntime"} + assert config.toolchain["onnxruntime"] == "1.23.2" + assert config.quantization is not None + assert config.quantization.enabled_precisions == ("fp16", "int8") + assert config.quantization.calibration_manifest == Path( + "artifacts/vision_v8_quantized_eval/calibration.json" + ) + assert config.quantization.thresholds == Path("configs/vision_v8_quantization_thresholds.json") + assert config.quantization.accuracy_report == Path( + "artifacts/vision_v8_quantized_eval/accuracy.json" + ) + assert config.quantization.accuracy_markdown == Path( + "artifacts/vision_v8_quantized_eval/accuracy.md" + ) + assert config.quantization.provider_gates == ( + ("CPUExecutionProvider", "fp32"), + ("CUDAExecutionProvider", "fp16"), + ("CPUExecutionProvider", "int8"), + ) # Five seeds underestimate the observed maxima (see the validation report), # so a release gate has to be deeper than the development default. assert config.parity_num_tests >= 20 +def test_release_workflows_share_quantized_evidence_artifact_contract() -> None: + release_workflow = Path(".github/workflows/onnx-release.yml").read_text(encoding="utf-8") + coco_workflow = Path(".github/workflows/vision-v8-coco-accuracy.yml").read_text( + encoding="utf-8" + ) + + assert "--name vision-v8-quantized-release-inputs" in release_workflow + assert "onnxruntime-gpu" in release_workflow + assert "runs-on: ${{ inputs.runner || 'self-hosted' }}" in release_workflow + assert "name: vision-v8-quantized-release-inputs" in coco_workflow + assert "fp32_provider:" in coco_workflow + assert "fp16_provider:" in coco_workflow + assert "int8_provider:" in coco_workflow + assert '--provider "${{ inputs.fp32_provider }}"' in coco_workflow + assert '--provider "${{ inputs.fp16_provider }}"' in coco_workflow + assert '--provider "${{ inputs.int8_provider }}"' in coco_workflow + assert "onnxruntime-gpu" in coco_workflow + assert "calibration.json" in coco_workflow + assert "calibration_images/**" in coco_workflow + assert "accuracy.json" in coco_workflow + assert "accuracy.md" in coco_workflow + + +def test_release_builder_runs_by_documented_file_path(tmp_path: Path) -> None: + config_path = tmp_path / "release.json" + config_path.write_text( + json.dumps( + config_mapping( + quantization={ + "enabled_precisions": ["int8"], + "calibration_manifest": str(tmp_path / "missing-calibration.json"), + "thresholds": "configs/vision_v8_quantization_thresholds.json", + "accuracy_report": str(tmp_path / "missing-accuracy.json"), + "accuracy_markdown": str(tmp_path / "missing-accuracy.md"), + } + ) + ), + encoding="utf-8", + ) + + result = subprocess.run( + [ + sys.executable, + "scripts/build_onnx_release.py", + "--config", + str(config_path), + "--output-dir", + str(tmp_path / "dist"), + "--allow-toolchain-drift", + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode != 0 + assert "No module named 'scripts." not in result.stderr + + def test_a_moving_checkpoint_ref_is_rejected() -> None: with pytest.raises(ReleaseError, match="40-character commit sha"): config_from_mapping(config_mapping(checkpoint_revision="main")) @@ -121,6 +200,21 @@ def test_an_unpinned_toolchain_is_rejected() -> None: config_from_mapping(config_mapping(toolchain={})) +def test_unsupported_quantized_precision_is_rejected() -> None: + with pytest.raises(ReleaseError, match="unsupported quantized precisions"): + config_from_mapping( + config_mapping( + quantization={ + "enabled_precisions": ["fp8"], + "calibration_manifest": "calibration.json", + "thresholds": "thresholds.json", + "accuracy_report": "accuracy.json", + "accuracy_markdown": "accuracy.md", + } + ) + ) + + def test_toolchain_drift_is_reported_per_package() -> None: pinned = {"torch": "2.13.0", "onnx": "1.21.0"} @@ -141,10 +235,12 @@ def test_output_contract_derives_the_prediction_width_from_the_sidecar() -> None assert contract["dtype"] == "float32" -def test_manifest_carries_every_field_the_release_must_document(tmp_path: Path) -> None: +def test_manifest_carries_every_field_the_release_must_document( + tmp_path: Path, +) -> None: manifest = written_release(tmp_path) - assert manifest["checkpoint_revision"] == "f3b3e659612e543ca9ff91892c0662d38dc1a1d6" + assert manifest["checkpoint_revision"] == ("f3b3e659612e543ca9ff91892c0662d38dc1a1d6") assert manifest["framework_commit"] == "0" * 40 assert manifest["opset"] == 17 assert manifest["toolchain"]["torch"] == "2.13.0" @@ -185,7 +281,9 @@ def test_verification_catches_a_truncated_or_missing_artifact(tmp_path: Path) -> assert "missing" in problems -def test_verification_binds_the_manifest_to_the_publishing_commit(tmp_path: Path) -> None: +def test_verification_binds_the_manifest_to_the_publishing_commit( + tmp_path: Path, +) -> None: manifest = written_release(tmp_path) assert verify_manifest(manifest, tmp_path, expect_commit="0" * 40) == [] diff --git a/tests/test_tr_hash_agentic_250m_lab.py b/tests/test_tr_hash_agentic_250m_lab.py index 330d8f9..d4083ee 100644 --- a/tests/test_tr_hash_agentic_250m_lab.py +++ b/tests/test_tr_hash_agentic_250m_lab.py @@ -1,11 +1,15 @@ from __future__ import annotations import json -import tomllib from pathlib import Path from complexity.training.finetuning import validate_refinement_plan +try: + import tomllib +except ModuleNotFoundError: # pragma: no cover - Python 3.10 fallback + import tomli as tomllib + PROJECT_ROOT = Path(__file__).parents[1] PLAN_DIR = PROJECT_ROOT / "configs" / "replay_plans"