From 8982b727ea7629103d069b00e439a1a26de0f8c3 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Thu, 3 Sep 2026 11:53:24 +0500 Subject: [PATCH 1/5] Add Vision v8 ONNX quantization release tooling --- complexity/deploy/onnx_detector/pipeline.py | 12 +- .../vision_v8_quantization_calibration.json | 20 + .../vision_v8_quantization_thresholds.json | 43 + docs/index.md | 1 + .../2026-08-31-vision-v8-onnx-quantization.md | 822 ++++++++++++++++++ docs/vision-v8-onnx-quantization.md | 108 +++ scripts/benchmark_onnx_artifacts.py | 173 ++++ scripts/check_onnx_quantized_artifacts.py | 239 +++++ scripts/quantize_onnx.py | 326 +++++++ tests/test_onnx_quantization_accuracy_gate.py | 72 ++ .../test_onnx_quantization_artifact_checks.py | 33 + tests/test_onnx_quantization_benchmark.py | 19 + tests/test_onnx_quantization_cli.py | 34 + tests/test_onnx_quantization_config.py | 227 +++++ tests/test_tr_hash_agentic_250m_lab.py | 6 +- 15 files changed, 2133 insertions(+), 2 deletions(-) create mode 100644 configs/vision_v8_quantization_calibration.json create mode 100644 configs/vision_v8_quantization_thresholds.json create mode 100644 docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md create mode 100644 docs/vision-v8-onnx-quantization.md create mode 100644 scripts/benchmark_onnx_artifacts.py create mode 100644 scripts/check_onnx_quantized_artifacts.py create mode 100644 scripts/quantize_onnx.py create mode 100644 tests/test_onnx_quantization_accuracy_gate.py create mode 100644 tests/test_onnx_quantization_artifact_checks.py create mode 100644 tests/test_onnx_quantization_benchmark.py create mode 100644 tests/test_onnx_quantization_cli.py create mode 100644 tests/test_onnx_quantization_config.py diff --git a/complexity/deploy/onnx_detector/pipeline.py b/complexity/deploy/onnx_detector/pipeline.py index ac84484..50779d8 100644 --- a/complexity/deploy/onnx_detector/pipeline.py +++ b/complexity/deploy/onnx_detector/pipeline.py @@ -39,9 +39,19 @@ def from_files( model_path: str | Path, metadata_path: str | Path, providers: Sequence[str] = ("CPUExecutionProvider",), + *, + warmup_runs: int = 1, + intra_op_num_threads: int | None = None, + inter_op_num_threads: int | None = None, ) -> "OnnxDetectorPipeline": metadata = load_metadata(metadata_path) - session = cls.create_session(model_path, providers).open() + session = cls.create_session( + model_path, + providers, + warmup_runs=warmup_runs, + intra_op_num_threads=intra_op_num_threads, + inter_op_num_threads=inter_op_num_threads, + ).open() validate_output_shape(metadata, session._require_session().get_outputs()[0].shape) if session.config.warmup_runs: session.warmup((1, 3, metadata.image_size, metadata.image_size)) diff --git a/configs/vision_v8_quantization_calibration.json b/configs/vision_v8_quantization_calibration.json new file mode 100644 index 0000000..15302a6 --- /dev/null +++ b/configs/vision_v8_quantization_calibration.json @@ -0,0 +1,20 @@ +{ + "schema_version": 1, + "dataset": { + "name": "coco-2017-train-calibration", + "split": "train2017-calibration", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "replace-with-annotations-sha256", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } +} diff --git a/configs/vision_v8_quantization_thresholds.json b/configs/vision_v8_quantization_thresholds.json new file mode 100644 index 0000000..659e8b3 --- /dev/null +++ b/configs/vision_v8_quantization_thresholds.json @@ -0,0 +1,43 @@ +{ + "schema_version": 1, + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "partial_release": "block", + "optional_provider_precisions": [] + }, + "precisions": { + "fp16": { + "max_raw_logit_abs_error": 0.01, + "max_decoded_box_px_error": 1.0, + "max_score_abs_error": 0.01, + "max_map50_95_drop": 0.005, + "max_map50_drop": 0.01, + "unexpected_fp32_nodes": "fail" + }, + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + "max_map50_95_drop": 0.02, + "max_map50_drop": 0.03, + "unexpected_fp32_nodes": "allow" + } + }, + "providers": { + "CPUExecutionProvider": ["fp32", "int8"], + "CUDAExecutionProvider": ["fp32", "fp16"], + "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] + }, + "benchmark": { + "warmup_iterations": 25, + "measured_iterations": 100, + "report": [ + "median_ms", + "mean_ms", + "stddev_ms", + "p95_ms", + "throughput_images_per_second", + "peak_memory_mb" + ] + } +} diff --git a/docs/index.md b/docs/index.md index 133d9db..c5ecf1c 100644 --- a/docs/index.md +++ b/docs/index.md @@ -92,6 +92,7 @@ GQA + TR-MoE architecture. - [TR-Hash object detection and serving](tr-hash-object-detection.md) - [TR-Hash Vision ONNX deployment](onnx_deploy.md) - [Vision v8 COCO accuracy gates](vision-v8-coco-accuracy-gates.md) +- [Vision v8 ONNX quantization](vision-v8-onnx-quantization.md) - [Detector specialization and ablations](TR_HASH_DETECTOR_SPECIALIZATION.md) - [TR-Hash sensor fusion](tr_hash_sensor_fusion.md) - [Vision dependency stack](vision-dependency-stack.md) diff --git a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md new file mode 100644 index 0000000..671136e --- /dev/null +++ b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md @@ -0,0 +1,822 @@ +# Vision v8 ONNX Quantization Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Build reproducible FP16 and INT8 ONNX artifacts for both Vision v8 detector branches and gate them against FP32 accuracy, decoded detections, performance, and release metadata. + +**Architecture:** Quantization is implemented as a post-export pipeline layered on top of the existing Vision v8 ONNX export, ONNX detector runtime, COCO evaluation, and release workflow. FP32 remains the reference; FP16 and INT8 artifacts are generated deterministically, validated for provider support and node dtypes, evaluated through the shared ONNX COCO evaluator, and published only after release gates pass. + +**Tech Stack:** Python, ONNX, ONNX Runtime quantization tools, ONNX Runtime providers, PyTorch, COCO evaluation helpers, GitHub Actions. + +**Spec:** User issue: “The published Vision v8 ONNX artifacts are FP32 only. We need smaller and faster deployment artifacts without trading away decoded detections or COCO accuracy silently.” + +## Global Constraints + +- Quantization commands must be deterministic from a pinned checkpoint, FP32 ONNX model, framework commit, calibration manifest, and quantization settings. +- Calibration inputs must be pinned and disjoint from the COCO evaluation inputs used by the accuracy gate. +- Unsupported provider/precision combinations must fail clearly instead of falling back silently. +- FP16 conversion must report remaining FP32 nodes and fail unless every remaining FP32 node is covered by a checked-in allowlist. +- INT8 settings must pin calibration method, per-channel/per-tensor mode, symmetric/asymmetric mode, activation type, weight type, calibration batch size, and calibration thread count. +- FP32, FP16, and INT8 reports must compare artifact size, latency, throughput, peak memory, raw-logit parity, decoded-output parity, and COCO metrics. +- Tolerances and floors must live in checked-in config, not inside scripts. +- If any required quantized artifact fails its gate, the release blocks; do not publish a partial FP32+FP16-only release unless the release config explicitly marks INT8 optional for that provider. +- ONNX binaries must remain excluded from normal source commits and be published as release assets. + +--- + +## File Structure + +- Create `configs/vision_v8_quantization_thresholds.json` for FP16/INT8 raw-logit, decoded-output, COCO, performance, dtype, and provider thresholds. +- Create `configs/vision_v8_quantization_calibration.json` for pinned calibration input metadata and quantization settings. +- Create `scripts/quantize_onnx.py` to generate FP16 and INT8 ONNX files plus JSON sidecars. +- Create `scripts/check_onnx_quantized_artifacts.py` to validate metadata, checksums, node dtypes, provider support, calibration/eval disjointness, parity reports, COCO reports, and performance reports. +- Create `scripts/benchmark_onnx_artifacts.py` to measure artifact latency, throughput, and peak memory with stable warm-up and measured iteration counts. +- Modify `scripts/check_onnx_parity.py` only if needed to expose reusable raw-logit and decoded-output comparison helpers. +- Reuse `scripts/evaluate_onnx_coco.py` from the COCO accuracy gate work for all ONNX COCO AP metrics. +- Modify `.github/workflows/detector-export.yml` or add `.github/workflows/vision-v8-onnx-quantization-release.yml` to generate, gate, and upload FP32/FP16/INT8 release assets. +- Create `docs/vision-v8-onnx-quantization.md` for reproduction commands, supported providers, precision limitations, and release policy. +- Add tests under `tests/test_onnx_quantization_*.py`. + +--- + +### Task 1: Quantization Config Contract + +**Files:** +- Create: `configs/vision_v8_quantization_thresholds.json` +- Create: `configs/vision_v8_quantization_calibration.json` +- Test: `tests/test_onnx_quantization_config.py` + +**Interfaces:** +- Produces: `load_quantization_thresholds(path: Path) -> dict[str, Any]` +- Produces: `load_calibration_manifest(path: Path) -> dict[str, Any]` + +- [ ] **Step 1: Write config validation tests** + +```python +from pathlib import Path + +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + load_calibration_manifest, + load_quantization_thresholds, +) + + +def test_threshold_config_requires_explicit_precision_policy(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text('{"schema_version": 1, "precisions": {"fp16": {}, "int8": {}}}') + + with pytest.raises(ValueError, match="release_policy"): + load_quantization_thresholds(path) + + +def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": {"name": "coco-2017-train", "image_ids_sha256": "abc"}, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } + } + """ + ) + + manifest = load_calibration_manifest(path) + + assert manifest["quantization"]["calibration_method"] == "minmax" + assert manifest["quantization"]["num_threads"] == 1 +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Expected: imports fail because `scripts/check_onnx_quantized_artifacts.py` does not exist yet. + +- [ ] **Step 3: Add config files** + +Create `configs/vision_v8_quantization_thresholds.json`: + +```json +{ + "schema_version": 1, + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "partial_release": "block", + "optional_provider_precisions": [] + }, + "precisions": { + "fp16": { + "max_raw_logit_abs_error": 0.01, + "max_decoded_box_px_error": 1.0, + "max_score_abs_error": 0.01, + "max_map50_95_drop": 0.005, + "max_map50_drop": 0.01, + "unexpected_fp32_nodes": "fail" + }, + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + "max_map50_95_drop": 0.02, + "max_map50_drop": 0.03, + "unexpected_fp32_nodes": "allow" + } + }, + "providers": { + "CPUExecutionProvider": ["fp32", "int8"], + "CUDAExecutionProvider": ["fp32", "fp16"], + "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] + }, + "benchmark": { + "warmup_iterations": 25, + "measured_iterations": 100, + "report": ["median_ms", "mean_ms", "stddev_ms", "p95_ms", "throughput_images_per_second", "peak_memory_mb"] + } +} +``` + +Create `configs/vision_v8_quantization_calibration.json`: + +```json +{ + "schema_version": 1, + "dataset": { + "name": "coco-2017-train-calibration", + "split": "train2017-calibration", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "replace-with-annotations-sha256", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } +} +``` + +- [ ] **Step 4: Add minimal config loaders** + +Create `scripts/check_onnx_quantized_artifacts.py` with: + +```python +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + + +def _load_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"JSON root must be an object: {path}") + return payload + + +def load_quantization_thresholds(path: Path) -> dict[str, Any]: + config = _load_json(path) + if "release_policy" not in config: + raise ValueError("threshold config missing release_policy") + precisions = config.get("precisions") + if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: + raise ValueError("threshold config must define fp16 and int8 precisions") + return config + + +def load_calibration_manifest(path: Path) -> dict[str, Any]: + manifest = _load_json(path) + quantization = manifest.get("quantization") + if not isinstance(quantization, dict): + raise ValueError("calibration manifest missing quantization settings") + required = { + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + "num_threads", + } + missing = sorted(required - set(quantization)) + if missing: + raise ValueError(f"calibration manifest missing settings: {missing}") + return manifest +``` + +- [ ] **Step 5: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Commit: + +```bash +git add configs/vision_v8_quantization_thresholds.json configs/vision_v8_quantization_calibration.json scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py +git commit -m "Add Vision v8 quantization gate config" +``` + +--- + +### Task 2: Calibration and Evaluation Disjointness + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Test: `tests/test_onnx_quantization_config.py` + +**Interfaces:** +- Produces: `assert_disjoint_image_ids(calibration_ids: set[int], evaluation_ids: set[int]) -> None` +- Produces: `image_id_manifest_sha256(image_ids: Sequence[int]) -> str` + +- [ ] **Step 1: Write failing overlap tests** + +```python +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, + image_id_manifest_sha256, +) + + +def test_calibration_and_eval_image_ids_must_be_disjoint() -> None: + with pytest.raises(ValueError, match="overlap"): + assert_disjoint_image_ids({1, 2, 3}, {3, 4, 5}) + + +def test_image_id_manifest_hash_is_order_stable() -> None: + assert image_id_manifest_sha256([3, 1, 2]) == image_id_manifest_sha256([1, 2, 3]) +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +- [ ] **Step 3: Implement helpers** + +Add: + +```python +import hashlib +from collections.abc import Sequence + + +def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: + digest = hashlib.sha256() + for image_id in sorted(map(int, image_ids)): + digest.update(f"{image_id}\n".encode("utf-8")) + return digest.hexdigest() + + +def assert_disjoint_image_ids(calibration_ids: set[int], evaluation_ids: set[int]) -> None: + overlap = calibration_ids & evaluation_ids + if overlap: + preview = sorted(overlap)[:10] + raise ValueError(f"calibration/evaluation image ID overlap: {preview}") +``` + +- [ ] **Step 4: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_config.py` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py +git commit -m "Reject quantization calibration eval leakage" +``` + +--- + +### Task 3: FP16 and INT8 Quantization CLI + +**Files:** +- Create: `scripts/quantize_onnx.py` +- Test: `tests/test_onnx_quantization_cli.py` + +**Interfaces:** +- Produces: `quantize_fp16(input_model: Path, output_model: Path, keep_fp32_op_types: Sequence[str]) -> None` +- Produces: `quantize_int8(input_model: Path, output_model: Path, calibration_manifest: Mapping[str, Any]) -> None` +- Produces: sidecar JSON with precision, source SHA-256, output SHA-256, toolchain versions, and quantization settings. + +- [ ] **Step 1: Write failing CLI metadata tests** + +```python +import json +from pathlib import Path + +from scripts.quantize_onnx import write_quantization_sidecar + + +def test_quantization_sidecar_binds_artifact_to_source_and_settings(tmp_path: Path) -> None: + sidecar = tmp_path / "model.fp16.json" + + write_quantization_sidecar( + sidecar, + precision="fp16", + source_model_sha256="source", + output_model_sha256="output", + framework_commit="commit", + checkpoint_revision="checkpoint", + settings={"keep_fp32_op_types": ["ReduceSum"]}, + ) + + data = json.loads(sidecar.read_text()) + assert data["precision"] == "fp16" + assert data["source_model_sha256"] == "source" + assert data["output_model_sha256"] == "output" + assert data["framework_commit"] == "commit" +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_cli.py` + +- [ ] **Step 3: Implement CLI skeleton and sidecar writer** + +Use `onnxconverter-common` or ONNX Runtime quantization APIs where available. If the dependency is unavailable, fail with: + +```text +FP16 quantization requires onnxconverter-common; install the export/quantization extra. +``` + +For INT8, fail with: + +```text +INT8 quantization requires onnxruntime.quantization and a calibration manifest. +``` + +- [ ] **Step 4: Implement FP16 conversion** + +Call the FP16 converter with explicit settings: + +```python +keep_fp32_op_types = tuple(settings.get("keep_fp32_op_types", ())) +disable_shape_infer = bool(settings.get("disable_shape_infer", False)) +``` + +Write output ONNX and sidecar. + +- [ ] **Step 5: Implement INT8 static quantization** + +Use pinned calibration settings: + +```python +calibration_method = manifest["quantization"]["calibration_method"] +per_channel = manifest["quantization"]["per_channel"] +activation_type = manifest["quantization"]["activation_type"] +weight_type = manifest["quantization"]["weight_type"] +num_threads = manifest["quantization"]["num_threads"] +``` + +Set calibration reader iteration order from the sorted calibration manifest. + +- [ ] **Step 6: Run quantization twice and assert same hash** + +Add CLI option: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model tr_hash_v8_o2m.onnx \ + --metadata tr_hash_v8_o2m.json \ + --precision fp16 \ + --output artifacts/a.onnx \ + --repeat-output artifacts/b.onnx \ + --require-identical-hash +``` + +If hashes differ, fail with: + +```text +quantization is not deterministic: first SHA-256 ... differs from repeat SHA-256 ... +``` + +- [ ] **Step 7: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_cli.py` + +Commit: + +```bash +git add scripts/quantize_onnx.py tests/test_onnx_quantization_cli.py +git commit -m "Add deterministic Vision v8 ONNX quantization CLI" +``` + +--- + +### Task 4: Node Dtype and Provider Support Validation + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Test: `tests/test_onnx_quantization_artifact_checks.py` + +**Interfaces:** +- Produces: `inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]` +- Produces: `check_provider_precision_supported(provider: str, precision: str, thresholds: Mapping[str, Any]) -> None` +- Produces: `check_unexpected_fp32_nodes(dtype_report: Mapping[str, Any], allowlist: Sequence[str]) -> list[str]` + +- [ ] **Step 1: Write provider support tests** + +```python +import pytest + +from scripts.check_onnx_quantized_artifacts import check_provider_precision_supported + + +def test_unsupported_provider_precision_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="does not support fp16"): + check_provider_precision_supported("CPUExecutionProvider", "fp16", thresholds) +``` + +- [ ] **Step 2: Write FP32 node allowlist tests** + +```python +from scripts.check_onnx_quantized_artifacts import check_unexpected_fp32_nodes + + +def test_unexpected_fp32_nodes_are_reported() -> None: + report = {"fp32_nodes": [{"name": "Conv_1", "op_type": "Conv"}, {"name": "ReduceSum_1", "op_type": "ReduceSum"}]} + + unexpected = check_unexpected_fp32_nodes(report, allowlist=["ReduceSum"]) + + assert unexpected == ["Conv_1:Conv"] +``` + +- [ ] **Step 3: Implement provider support check** + +```python +def check_provider_precision_supported(provider: str, precision: str, thresholds: Mapping[str, Any]) -> None: + supported = thresholds.get("providers", {}).get(provider) + if precision not in supported: + raise ValueError(f"{provider} does not support {precision} in quantization release config") +``` + +- [ ] **Step 4: Implement dtype allowlist check** + +```python +def check_unexpected_fp32_nodes(dtype_report: Mapping[str, Any], allowlist: Sequence[str]) -> list[str]: + allowed = set(allowlist) + unexpected = [] + for node in dtype_report.get("fp32_nodes", []): + if node["op_type"] not in allowed: + unexpected.append(f"{node['name']}:{node['op_type']}") + return unexpected +``` + +- [ ] **Step 5: Implement ONNX dtype inspection** + +Use ONNX graph initializers and value info to identify FP32 tensors that remain after FP16 conversion. Report: + +```json +{ + "fp32_nodes": [{"name": "node_name", "op_type": "Conv"}], + "fp16_nodes": 123, + "int8_nodes": 45 +} +``` + +- [ ] **Step 6: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_artifact_checks.py` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_artifact_checks.py +git commit -m "Validate quantized ONNX providers and node dtypes" +``` + +--- + +### Task 5: Quantized Parity and COCO Accuracy Gate + +**Files:** +- Modify: `scripts/check_onnx_quantized_artifacts.py` +- Modify: `scripts/check_onnx_parity.py` if helper reuse is needed +- Reuse: `scripts/evaluate_onnx_coco.py` +- Test: `tests/test_onnx_quantization_accuracy_gate.py` + +**Interfaces:** +- Produces: `check_quantized_accuracy_report(report: Mapping[str, Any], thresholds: Mapping[str, Any]) -> list[str]` +- Consumes: COCO report schema emitted by `scripts/evaluate_onnx_coco.py` + +- [ ] **Step 1: Write FP32 comparison tests** + +```python +from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report + + +def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> None: + report = { + "reference": {"precision": "fp32", "metrics": {"map50_95": 0.2, "map50": 0.32}}, + "candidate": {"precision": "int8", "metrics": {"map50_95": 0.17, "map50": 0.31}}, + } + thresholds = {"precisions": {"int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}}} + + failures = check_quantized_accuracy_report(report, thresholds) + + assert any("map50_95" in failure for failure in failures) +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_accuracy_gate.py` + +- [ ] **Step 3: Implement accuracy gate helper** + +Compare candidate metrics against FP32 reference metrics for each branch: + +```python +drop = float(reference_metric) - float(candidate_metric) +if drop > allowed_drop: + failures.append(f"{precision} {branch} {metric} dropped by {drop:.6f}") +``` + +- [ ] **Step 4: Wire to existing ONNX COCO evaluator** + +Do not create a new COCO eval path. The release workflow must call: + +```bash +python scripts/evaluate_onnx_coco.py \ + --model "$MODEL" \ + --metadata "$METADATA" \ + --annotations "$COCO_ANNOTATIONS" \ + --images "$COCO_IMAGES" \ + --output "$REPORT_DIR" \ + --provider "$PROVIDER" +``` + +- [ ] **Step 5: Add decoded-output parity checks** + +Reuse the ONNX detector pipeline to compare: + +```json +{ + "max_box_pixel_error": 0.7, + "max_score_abs_error": 0.006, + "class_id_mismatches": 0 +} +``` + +Fail if the precision-specific threshold is exceeded. + +- [ ] **Step 6: Run tests and commit** + +Run: + +```bash +python -m pytest -q tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_detector_core.py tests/test_vision_v8_coco_accuracy_gate.py +``` + +Commit: + +```bash +git add scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_accuracy_gate.py +git commit -m "Gate quantized ONNX accuracy against FP32" +``` + +--- + +### Task 6: Benchmark Methodology and Reports + +**Files:** +- Create: `scripts/benchmark_onnx_artifacts.py` +- Test: `tests/test_onnx_quantization_benchmark.py` + +**Interfaces:** +- Produces: `summarize_latency_ms(values: Sequence[float]) -> dict[str, float]` +- Produces: benchmark JSON with median, mean, stddev, p95, throughput, peak memory, provider requested, provider used, warmup count, measured count. + +- [ ] **Step 1: Write benchmark summary tests** + +```python +from scripts.benchmark_onnx_artifacts import summarize_latency_ms + + +def test_benchmark_report_uses_distribution_not_single_shot() -> None: + summary = summarize_latency_ms([10.0, 12.0, 14.0]) + + assert summary["median_ms"] == 12.0 + assert summary["mean_ms"] == 12.0 + assert summary["stddev_ms"] > 0.0 +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: `python -m pytest -q tests/test_onnx_quantization_benchmark.py` + +- [ ] **Step 3: Implement benchmark script** + +Use fixed methodology: + +```text +warmup_iterations = 25 +measured_iterations = 100 +batch_size = 1 unless explicitly configured +report median + mean + stddev + p95 +benchmark tolerance is separate from accuracy tolerance +``` + +- [ ] **Step 4: Add peak memory capture** + +For CUDA providers, use CUDA memory APIs where available. For CPU, record process RSS before/after and document it as approximate. + +- [ ] **Step 5: Run tests and commit** + +Run: `python -m pytest -q tests/test_onnx_quantization_benchmark.py` + +Commit: + +```bash +git add scripts/benchmark_onnx_artifacts.py tests/test_onnx_quantization_benchmark.py +git commit -m "Add quantized ONNX benchmark reports" +``` + +--- + +### Task 7: Release Workflow Integration + +**Files:** +- Create or modify: `.github/workflows/vision-v8-onnx-quantization-release.yml` +- Modify: `.github/workflows/detector-export.yml` only if the repo prefers one detector workflow +- Test: workflow YAML lint through `ruff` only for Python files and local shell dry-run where possible + +**Interfaces:** +- Consumes: `scripts/quantize_onnx.py` +- Consumes: `scripts/check_onnx_quantized_artifacts.py` +- Consumes: `scripts/evaluate_onnx_coco.py` +- Consumes: `scripts/benchmark_onnx_artifacts.py` + +- [ ] **Step 1: Add manual/tag-triggered workflow** + +Workflow triggers: + +```yaml +on: + workflow_dispatch: + push: + tags: + - "vision-v8-onnx-*" +``` + +- [ ] **Step 2: Export FP32 artifacts** + +Call the existing ONNX export path for: + +```text +o2m fp32 +nms-free fp32 +``` + +- [ ] **Step 3: Quantize all required artifacts** + +Generate: + +```text +o2m fp16 +o2m int8 +nms-free fp16 +nms-free int8 +``` + +Run each quantization command twice with `--require-identical-hash`. + +- [ ] **Step 4: Validate provider and node dtype policy** + +Fail if: + +```text +requested provider != actual provider +precision unsupported by provider config +FP16 output contains unexpected FP32 nodes outside allowlist +INT8 metadata does not match pinned calibration settings +``` + +- [ ] **Step 5: Run parity, COCO, and benchmark gates** + +Call: + +```bash +python scripts/evaluate_onnx_coco.py ... +python scripts/check_onnx_quantized_artifacts.py ... +python scripts/benchmark_onnx_artifacts.py ... +``` + +- [ ] **Step 6: Enforce partial-failure policy** + +If any required artifact fails, stop the workflow before release upload: + +```text +required quantized artifact failed gate; release blocked by release_policy.partial_release=block +``` + +- [ ] **Step 7: Publish release assets** + +Upload: + +```text +FP32 ONNX files +FP16 ONNX files +INT8 ONNX files +JSON sidecars +manifest JSON +accuracy reports JSON/Markdown +benchmark reports JSON/Markdown +``` + +- [ ] **Step 8: Commit** + +```bash +git add .github/workflows/vision-v8-onnx-quantization-release.yml +git commit -m "Publish gated quantized Vision v8 ONNX releases" +``` + +--- + +### Task 8: Documentation and Final Verification + +**Files:** +- Create: `docs/vision-v8-onnx-quantization.md` +- Modify: `docs/index.md` + +**Interfaces:** +- Documents exact commands for FP16, INT8, parity, COCO evaluation, benchmark, and release upload. + +- [ ] **Step 1: Document beginner-friendly concepts** + +Explain: + +```text +FP32 = bigger, safest reference +FP16 = smaller/faster float model, may leave some nodes FP32 for stability +INT8 = smallest/fastest candidate, requires calibration data +Calibration = sample inputs used to estimate activation ranges +Provider fallback = ORT did not use requested runtime +Partial FP16 fallback = some graph nodes stayed FP32 +``` + +- [ ] **Step 2: Document reproduction commands** + +Include: + +```bash +python scripts/quantize_onnx.py ... +python scripts/evaluate_onnx_coco.py ... +python scripts/check_onnx_quantized_artifacts.py ... +python scripts/benchmark_onnx_artifacts.py ... +``` + +- [ ] **Step 3: Run final local checks** + +Run: + +```bash +python -m ruff check scripts/quantize_onnx.py scripts/check_onnx_quantized_artifacts.py scripts/benchmark_onnx_artifacts.py tests/test_onnx_quantization_config.py tests/test_onnx_quantization_cli.py tests/test_onnx_quantization_artifact_checks.py tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_quantization_benchmark.py +python -m pytest -q tests/test_onnx_quantization_config.py tests/test_onnx_quantization_cli.py tests/test_onnx_quantization_artifact_checks.py tests/test_onnx_quantization_accuracy_gate.py tests/test_onnx_quantization_benchmark.py +``` + +- [ ] **Step 4: Record known untested release requirements** + +Before opening PR, state clearly whether these have run: + +```text +Full COCO FP32 vs FP16 vs INT8 +CUDA provider execution +TensorRT provider execution +GitHub Release upload +Quantize-twice identical artifact hash +``` + +- [ ] **Step 5: Commit** + +```bash +git add docs/vision-v8-onnx-quantization.md docs/index.md +git commit -m "Document Vision v8 ONNX quantization release process" +``` + +--- + +## Self-Review + +- Spec coverage: FP16 and INT8 quantization, calibration pinning, toolchain/settings metadata, size/latency/throughput/memory reports, raw-logit/decoded/COCO gates, release artifact publishing, checksums, framework commit, checkpoint revision, provider failure policy, and no source-committed ONNX binaries are all mapped to tasks. +- Reviewer gaps covered: quantize-twice hash determinism is Task 3; FP16 partial dtype fallback is Task 4; calibration/eval leakage is Task 2; shared COCO evaluator reuse is Task 5; benchmark noise methodology is Task 6; checked-in thresholds are Task 1; partial release policy is Task 7. +- Type consistency: helper names used by later tasks are introduced before they are consumed. diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md new file mode 100644 index 0000000..1d03de5 --- /dev/null +++ b/docs/vision-v8-onnx-quantization.md @@ -0,0 +1,108 @@ +# Vision v8 ONNX Quantization + +Vision v8 ONNX quantization produces smaller deployment artifacts while keeping +FP32 as the accuracy and behavior reference. Quantized artifacts are not +accepted only because they run; they must pass raw-output, decoded-detection, +COCO accuracy, provider, dtype, benchmark, checksum, and release metadata gates. + +## Precision Contract + +- `fp32`: the exported reference model and source of truth. +- `fp16`: smaller floating-point model for CUDA/TensorRT-style deployment. +- `int8`: post-training quantized model calibrated from pinned inputs. + +Each branch is quantized separately: + +- `o2m`: decode, confidence filtering, and class-aware NMS. +- `nms-free`: decode and confidence filtering only; NMS must not run. + +Unsupported provider and precision pairs fail explicitly using +`configs/vision_v8_quantization_thresholds.json`. Provider fallback is separate +from FP16 partial fallback: the release check must verify both that ONNX Runtime +used the requested provider and that unexpected graph nodes did not remain FP32. + +## Calibration Contract + +INT8 calibration is pinned by +`configs/vision_v8_quantization_calibration.json`. The manifest records the +dataset, image-ID manifest hash, annotation hash, calibration method, +per-channel/per-tensor mode, symmetric/asymmetric choices, activation type, +weight type, batch size, and calibration thread count. +Placeholder hashes are rejected by `load_calibration_manifest`; before a release +run, replace them with the approved calibration subset and real SHA-256 values. + +Calibration images must be disjoint from the COCO evaluation images used by the +accuracy gate. This prevents tuning the INT8 ranges on the same images used to +claim final AP. + +## Reproduction Commands + +Create an FP16 artifact: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model artifacts/onnx/tr_hash_v8_o2m.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m.json \ + --precision fp16 \ + --output artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --repeat-output artifacts/onnx/tr_hash_v8_o2m_fp16_repeat.onnx \ + --require-identical-hash \ + --checkpoint-revision AETHORIA-AI/TR-HASH-Vision-v8-2M-COCO-SFT@REVISION +``` + +Create an INT8 artifact: + +```bash +python scripts/quantize_onnx.py \ + --fp32-model artifacts/onnx/tr_hash_v8_o2m.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m.json \ + --precision int8 \ + --calibration-manifest configs/vision_v8_quantization_calibration.json \ + --output artifacts/onnx/tr_hash_v8_o2m_int8.onnx \ + --repeat-output artifacts/onnx/tr_hash_v8_o2m_int8_repeat.onnx \ + --require-identical-hash \ + --checkpoint-revision AETHORIA-AI/TR-HASH-Vision-v8-2M-COCO-SFT@REVISION +``` + +Evaluate a quantized ONNX artifact on COCO: + +```bash +python scripts/evaluate_onnx_coco.py \ + --model artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m_fp16.json \ + --annotations artifacts/COCO/annotations/instances_val2017.json \ + --images artifacts/COCO/images/val2017 \ + --output artifacts/vision_v8_quantized_eval/o2m_fp16 \ + --branch o2m-nms \ + --provider cuda +``` + +Benchmark an artifact: + +```bash +python scripts/benchmark_onnx_artifacts.py \ + --model artifacts/onnx/tr_hash_v8_o2m_fp16.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m_fp16.json \ + --output artifacts/vision_v8_quantized_eval/o2m_fp16/benchmark.json \ + --provider cuda \ + --warmup-iterations 25 \ + --measured-iterations 100 +``` + +## Release Policy + +The quantized release workflow blocks the release when any required artifact +fails. Do not publish a partial release, such as FP32 plus FP16 only, unless the +checked-in release policy explicitly marks the failed precision/provider as +optional. + +Release assets must include: + +- FP32, FP16, and INT8 ONNX files for both O2M and NMS-free branches. +- JSON sidecars for each artifact. +- Accuracy reports in JSON and Markdown. +- Benchmark reports in JSON and Markdown. +- A manifest binding every artifact to framework commit, checkpoint revision, + SHA-256, file size, provider, precision, and quantization settings. + +ONNX binaries remain excluded from normal source commits. diff --git a/scripts/benchmark_onnx_artifacts.py b/scripts/benchmark_onnx_artifacts.py new file mode 100644 index 0000000..38033b3 --- /dev/null +++ b/scripts/benchmark_onnx_artifacts.py @@ -0,0 +1,173 @@ +"""Benchmark Vision v8 ONNX artifacts with stable release-report methodology.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import platform +import statistics +import sys +import time +from collections.abc import Sequence +from pathlib import Path +from typing import Any + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--provider", action="append", default=None) + parser.add_argument("--batch-size", type=int, default=1) + parser.add_argument("--warmup-iterations", type=int, default=25) + parser.add_argument("--measured-iterations", type=int, default=100) + parser.add_argument("--ort-intra-op-threads", type=int, default=1) + parser.add_argument("--ort-inter-op-threads", type=int, default=1) + return parser.parse_args() + + +def _percentile(values: Sequence[float], fraction: float) -> float: + if not values: + return 0.0 + ordered = sorted(values) + index = min(round((len(ordered) - 1) * fraction), len(ordered) - 1) + return float(ordered[index]) + + +def summarize_latency_ms(values: Sequence[float]) -> dict[str, float]: + """Summarize latency as a distribution to avoid single-shot noise.""" + + if not values: + return { + "median_ms": 0.0, + "mean_ms": 0.0, + "stddev_ms": 0.0, + "p95_ms": 0.0, + "p99_ms": 0.0, + } + return { + "median_ms": float(statistics.median(values)), + "mean_ms": float(statistics.fmean(values)), + "stddev_ms": float(statistics.stdev(values)) if len(values) > 1 else 0.0, + "p95_ms": _percentile(values, 0.95), + "p99_ms": _percentile(values, 0.99), + } + + +def _peak_memory_mb() -> float | None: + try: + import psutil + except ImportError: + return None + return psutil.Process(os.getpid()).memory_info().rss / (1024 * 1024) + + +def _benchmark_session( + pipeline: Any, + *, + batch_size: int, + warmup_iterations: int, + measured_iterations: int, +) -> list[float]: + input_shape = (batch_size, 3, pipeline.metadata.image_size, pipeline.metadata.image_size) + dummy = np.zeros(input_shape, dtype=np.float32) + for _ in range(warmup_iterations): + pipeline.session.run(dummy) + + latencies: list[float] = [] + for _ in range(measured_iterations): + started = time.perf_counter() + pipeline.session.run(dummy) + latencies.append((time.perf_counter() - started) * 1000.0) + return latencies + + +def benchmark_onnx_artifact( + *, + model_path: Path, + metadata_path: Path, + providers: Sequence[str], + batch_size: int, + warmup_iterations: int, + measured_iterations: int, + ort_intra_op_threads: int, + ort_inter_op_threads: int, +) -> dict[str, Any]: + from complexity.deploy.onnx_detector import OnnxDetectorPipeline + from scripts.quantize_onnx import package_version, sha256_file + + if batch_size <= 0 or warmup_iterations < 0 or measured_iterations <= 0: + raise ValueError("batch size and measured iterations must be positive") + pipeline = OnnxDetectorPipeline.from_files( + model_path, + metadata_path, + providers=providers, + intra_op_num_threads=ort_intra_op_threads, + inter_op_num_threads=ort_inter_op_threads, + ) + latencies = _benchmark_session( + pipeline, + batch_size=batch_size, + warmup_iterations=warmup_iterations, + measured_iterations=measured_iterations, + ) + summary = summarize_latency_ms(latencies) + mean_ms = summary["mean_ms"] + throughput = (batch_size * 1000.0 / mean_ms) if mean_ms else 0.0 + return { + "schema_version": 1, + "model": str(model_path), + "metadata": str(metadata_path), + "model_sha256": sha256_file(model_path), + "model_size_bytes": model_path.stat().st_size, + "requested_provider": list(providers), + "actual_provider": pipeline.session.provider_used, + "batch_size": batch_size, + "warmup_iterations": warmup_iterations, + "measured_iterations": measured_iterations, + "latency": summary, + "throughput_images_per_second": throughput, + "peak_memory_mb": _peak_memory_mb(), + "benchmark_methodology": "fixed warmup, fixed measured iterations, latency distribution", + "environment": { + "python": sys.version.split()[0], + "os": platform.platform(), + "onnxruntime": package_version("onnxruntime"), + }, + "ort_intra_op_threads": ort_intra_op_threads, + "ort_inter_op_threads": ort_inter_op_threads, + } + + +def main() -> None: + from scripts.onnx_detect import provider_names + + args = parse_args() + if not math.isfinite(float(args.batch_size)): + raise ValueError("batch size must be finite") + report = benchmark_onnx_artifact( + model_path=args.model, + metadata_path=args.metadata, + providers=provider_names(args.provider), + batch_size=args.batch_size, + warmup_iterations=args.warmup_iterations, + measured_iterations=args.measured_iterations, + ort_intra_op_threads=args.ort_intra_op_threads, + ort_inter_op_threads=args.ort_inter_op_threads, + ) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py new file mode 100644 index 0000000..a36e626 --- /dev/null +++ b/scripts/check_onnx_quantized_artifacts.py @@ -0,0 +1,239 @@ +"""Validate Vision v8 quantized ONNX artifact metadata and reports.""" + +from __future__ import annotations + +import hashlib +import json +import string +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any + + +def _load_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"JSON root must be an object: {path}") + return payload + + +def load_quantization_thresholds(path: Path) -> dict[str, Any]: + """Load and validate the checked-in quantized artifact gate thresholds.""" + + config = _load_json(path) + if "release_policy" not in config: + raise ValueError("threshold config missing release_policy") + precisions = config.get("precisions") + if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: + raise ValueError("threshold config must define fp16 and int8 precisions") + return config + + +def load_calibration_manifest(path: Path) -> dict[str, Any]: + """Load and validate the pinned INT8 calibration manifest contract.""" + + manifest = _load_json(path) + dataset = manifest.get("dataset") + if not isinstance(dataset, dict): + raise ValueError("calibration manifest missing dataset") + for key in ("image_ids_sha256", "annotations_sha256"): + if not _is_sha256(dataset.get(key)): + raise ValueError(f"calibration manifest dataset.{key} must be a SHA-256") + if "disjoint_from" not in dataset: + raise ValueError("calibration manifest dataset.disjoint_from must be declared") + + quantization = manifest.get("quantization") + if not isinstance(quantization, dict): + raise ValueError("calibration manifest missing quantization settings") + required = { + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + "num_threads", + } + missing = sorted(required - set(quantization)) + if missing: + raise ValueError(f"calibration manifest missing settings: {missing}") + image_ids = manifest.get("image_ids") + if not isinstance(image_ids, Sequence) or isinstance(image_ids, (str, bytes)): + raise ValueError("calibration manifest image_ids must be a sequence") + if not image_ids: + raise ValueError("calibration manifest image_ids must not be empty") + actual_digest = image_id_manifest_sha256([int(image_id) for image_id in image_ids]) + expected_digest = str(dataset["image_ids_sha256"]) + if actual_digest != expected_digest: + raise ValueError( + "calibration manifest dataset.image_ids_sha256 does not match image_ids" + ) + + images = manifest.get("images") + if not isinstance(images, Sequence) or isinstance(images, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence") + if not images: + raise ValueError("calibration manifest images must not be empty") + return manifest + + +def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: + """Hash a stable sorted image-ID manifest for calibration/eval pinning.""" + + digest = hashlib.sha256() + for image_id in sorted(map(int, image_ids)): + digest.update(f"{image_id}\n".encode("utf-8")) + return digest.hexdigest() + + +def assert_disjoint_image_ids( + calibration_ids: set[int], + evaluation_ids: set[int], +) -> None: + """Fail when INT8 calibration inputs overlap accuracy-gate inputs.""" + + overlap = calibration_ids & evaluation_ids + if overlap: + preview = sorted(overlap)[:10] + raise ValueError(f"calibration/evaluation image ID overlap: {preview}") + + +def check_provider_precision_supported( + provider: str, + precision: str, + thresholds: Mapping[str, Any], +) -> None: + """Fail clearly when a provider/precision pair is not release-supported.""" + + providers = thresholds.get("providers", {}) + if not isinstance(providers, Mapping) or provider not in providers: + raise ValueError(f"{provider} is not configured in quantization release config") + supported = providers[provider] + if not isinstance(supported, Sequence) or isinstance(supported, (str, bytes)): + raise ValueError(f"{provider} precision policy must be a sequence") + if precision not in supported: + raise ValueError(f"{provider} does not support {precision} in quantization release config") + + +def check_unexpected_fp32_nodes( + dtype_report: Mapping[str, Any], + allowlist: Sequence[str], +) -> list[str]: + """Return FP32 nodes not covered by an explicit op-type allowlist.""" + + allowed = set(allowlist) + unexpected: list[str] = [] + fp32_nodes = dtype_report.get("fp32_nodes", []) + if not isinstance(fp32_nodes, Sequence) or isinstance(fp32_nodes, (str, bytes)): + raise ValueError("dtype report fp32_nodes must be a sequence") + for node in fp32_nodes: + if not isinstance(node, Mapping): + raise ValueError("dtype report fp32_nodes entries must be objects") + name = str(node.get("name", "")) + op_type = str(node.get("op_type", "")) + if op_type not in allowed: + unexpected.append(f"{name}:{op_type}") + return unexpected + + +def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: + """Inventory node-level dtype hints in an ONNX graph. + + ONNX does not assign one dtype directly to every node, so this inspector + maps typed graph values back to producer nodes and reports nodes producing + FP32, FP16, or INT8/UINT8 tensors. + """ + + try: + import onnx + from onnx import TensorProto + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError("ONNX dtype inspection requires the onnx package") from error + + model = onnx.load(str(model_path)) + value_dtypes: dict[str, int] = {} + for value_info in [ + *model.graph.input, + *model.graph.output, + *model.graph.value_info, + *model.graph.initializer, + ]: + name = getattr(value_info, "name", "") + data_type = None + if hasattr(value_info, "data_type"): + data_type = value_info.data_type + elif getattr(value_info, "type", None) is not None: + tensor_type = value_info.type.tensor_type + data_type = tensor_type.elem_type + if name and data_type: + value_dtypes[name] = int(data_type) + + fp32_nodes: list[dict[str, str]] = [] + fp16_nodes = 0 + int8_nodes = 0 + for node in model.graph.node: + output_types = {value_dtypes[name] for name in node.output if name in value_dtypes} + if TensorProto.FLOAT in output_types: + fp32_nodes.append({"name": node.name or node.output[0], "op_type": node.op_type}) + if TensorProto.FLOAT16 in output_types: + fp16_nodes += 1 + if TensorProto.INT8 in output_types or TensorProto.UINT8 in output_types: + int8_nodes += 1 + + return { + "fp32_nodes": fp32_nodes, + "fp16_nodes": fp16_nodes, + "int8_nodes": int8_nodes, + } + + +def check_quantized_accuracy_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + """Compare a quantized candidate COCO report against its FP32 reference.""" + + reference = _mapping(report.get("reference")) + candidate = _mapping(report.get("candidate")) + reference_branch = str(reference.get("branch", "")) + candidate_branch = str(candidate.get("branch", "")) + if reference_branch != candidate_branch: + return [ + "candidate branch " + f"{candidate_branch} does not match FP32 reference branch {reference_branch}" + ] + + precision = str(candidate.get("precision", "")) + precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) + if not precision_thresholds: + return [f"candidate precision {precision} has no quantization thresholds"] + + reference_metrics = _mapping(reference.get("metrics")) + candidate_metrics = _mapping(candidate.get("metrics")) + failures: list[str] = [] + for metric, threshold_name in ( + ("map50_95", "max_map50_95_drop"), + ("map50", "max_map50_drop"), + ): + if metric not in reference_metrics or metric not in candidate_metrics: + failures.append(f"missing metric {metric} in FP32 or candidate report") + continue + allowed_drop = float(precision_thresholds[threshold_name]) + drop = float(reference_metrics[metric]) - float(candidate_metrics[metric]) + if drop > allowed_drop: + failures.append( + f"{precision} {candidate_branch} {metric} dropped by {drop:.6f}; " + f"allowed drop {allowed_drop:.6f}" + ) + return failures + + +def _mapping(value: object) -> Mapping[str, Any]: + return value if isinstance(value, Mapping) else {} + + +def _is_sha256(value: object) -> bool: + if not isinstance(value, str) or len(value) != 64: + return False + return all(character in string.hexdigits for character in value) diff --git a/scripts/quantize_onnx.py b/scripts/quantize_onnx.py new file mode 100644 index 0000000..89f846a --- /dev/null +++ b/scripts/quantize_onnx.py @@ -0,0 +1,326 @@ +"""Create reproducible FP16 and INT8 Vision v8 ONNX artifacts.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import platform +import shutil +import subprocess +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from typing import Any, Literal + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +Precision = Literal["fp32", "fp16", "int8"] + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--fp32-model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--precision", choices=("fp16", "int8"), required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--sidecar", type=Path) + parser.add_argument("--calibration-manifest", type=Path) + parser.add_argument("--checkpoint-revision", default="unknown") + parser.add_argument( + "--keep-fp32-op-type", + action="append", + default=[], + help="FP16 conversion allowlist; may be repeated", + ) + parser.add_argument("--disable-shape-infer", action="store_true") + parser.add_argument("--repeat-output", type=Path) + parser.add_argument("--require-identical-hash", action="store_true") + return parser.parse_args() + + +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def framework_commit() -> str: + try: + return subprocess.check_output( + ["git", "rev-parse", "HEAD"], + stderr=subprocess.DEVNULL, + text=True, + ).strip() + except (OSError, subprocess.CalledProcessError): + return "unknown" + + +def package_version(module_name: str) -> str | None: + try: + from importlib import metadata + + return metadata.version(module_name) + except metadata.PackageNotFoundError: + return None + + +def toolchain_versions() -> dict[str, Any]: + return { + "python": sys.version.split()[0], + "os": platform.platform(), + "onnx": package_version("onnx"), + "onnxruntime": package_version("onnxruntime"), + "onnxconverter-common": package_version("onnxconverter-common"), + "numpy": np.__version__, + } + + +def assert_identical_artifact_hashes(first_sha256: str, second_sha256: str) -> None: + if first_sha256 != second_sha256: + raise ValueError( + "quantization is not deterministic: " + f"first SHA-256 {first_sha256} differs from repeat SHA-256 {second_sha256}" + ) + + +def write_quantization_sidecar( + path: Path, + *, + precision: Precision, + source_model_sha256: str, + output_model_sha256: str, + framework_commit: str, + checkpoint_revision: str, + settings: Mapping[str, Any], + toolchain: Mapping[str, Any], +) -> None: + payload = { + "schema_version": 1, + "artifact_type": "vision_v8_quantized_onnx", + "precision": precision, + "framework_commit": framework_commit, + "checkpoint_revision": checkpoint_revision, + "source_model_sha256": source_model_sha256, + "output_model_sha256": output_model_sha256, + "settings": dict(settings), + "toolchain": dict(toolchain), + } + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") + + +def quantize_fp32(input_model: Path, output_model: Path) -> None: + output_model.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(input_model, output_model) + + +def quantize_fp16( + input_model: Path, + output_model: Path, + *, + keep_fp32_op_types: Sequence[str] = (), + disable_shape_infer: bool = False, +) -> None: + try: + import onnx + from onnxruntime.transformers import float16 + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError( + "FP16 quantization requires ONNX Runtime transformer tools; " + "install the export/quantization extra." + ) from error + + output_model.parent.mkdir(parents=True, exist_ok=True) + model = onnx.load(str(input_model)) + converted = float16.convert_float_to_float16( + model, + keep_io_types=True, + disable_shape_infer=disable_shape_infer, + op_block_list=list(keep_fp32_op_types), + ) + onnx.save(converted, str(output_model)) + + +class _CalibrationReader: + def __init__( + self, + *, + image_paths: Sequence[Path], + metadata_path: Path, + input_name: str = "pixel_values", + ) -> None: + from complexity.deploy.onnx_detector.metadata import load_metadata + from complexity.deploy.onnx_detector.preprocess import preprocess_image + + metadata = load_metadata(metadata_path) + self._input_name = input_name + self._items = [ + {input_name: preprocess_image(path, metadata.image_size).pixel_values} + for path in image_paths + ] + self._index = 0 + + def get_next(self) -> dict[str, np.ndarray] | None: + if self._index >= len(self._items): + return None + item = self._items[self._index] + self._index += 1 + return item + + +def _calibration_method(name: str) -> Any: + from onnxruntime.quantization import CalibrationMethod + + normalized = name.lower() + if normalized == "minmax": + return CalibrationMethod.MinMax + if normalized in {"entropy", "kl"}: + return CalibrationMethod.Entropy + if normalized == "percentile": + return CalibrationMethod.Percentile + raise ValueError(f"unsupported INT8 calibration method: {name}") + + +def _quant_type(name: str) -> Any: + from onnxruntime.quantization import QuantType + + normalized = name.lower() + if normalized == "qint8": + return QuantType.QInt8 + if normalized == "quint8": + return QuantType.QUInt8 + raise ValueError(f"unsupported ONNX Runtime quant type: {name}") + + +def _calibration_paths(manifest: Mapping[str, Any]) -> list[Path]: + raw_paths = manifest.get("images", []) + if not isinstance(raw_paths, Sequence) or isinstance(raw_paths, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence of paths") + return [Path(str(path)) for path in sorted(raw_paths, key=str)] + + +def quantize_int8( + input_model: Path, + output_model: Path, + *, + metadata_path: Path, + calibration_manifest: Mapping[str, Any], +) -> None: + try: + from onnxruntime.quantization import QuantFormat, quantize_static + except ImportError as error: # pragma: no cover - dependency guard + raise RuntimeError( + "INT8 quantization requires onnxruntime.quantization and a calibration manifest." + ) from error + + settings = calibration_manifest["quantization"] + image_paths = _calibration_paths(calibration_manifest) + if not image_paths: + raise ValueError("INT8 quantization requires calibration manifest images") + reader = _CalibrationReader(image_paths=image_paths, metadata_path=metadata_path) + output_model.parent.mkdir(parents=True, exist_ok=True) + quantize_static( + str(input_model), + str(output_model), + reader, + quant_format=QuantFormat.QDQ, + calibrate_method=_calibration_method(str(settings["calibration_method"])), + per_channel=bool(settings["per_channel"]), + activation_type=_quant_type(str(settings["activation_type"])), + weight_type=_quant_type(str(settings["weight_type"])), + ) + + +def quantize_once( + *, + fp32_model: Path, + metadata_path: Path, + precision: Precision, + output_model: Path, + calibration_manifest: Mapping[str, Any] | None, + keep_fp32_op_types: Sequence[str], + disable_shape_infer: bool, +) -> dict[str, Any]: + if precision == "fp32": + quantize_fp32(fp32_model, output_model) + settings: dict[str, Any] = {} + elif precision == "fp16": + settings = { + "keep_fp32_op_types": list(keep_fp32_op_types), + "disable_shape_infer": disable_shape_infer, + } + quantize_fp16( + fp32_model, + output_model, + keep_fp32_op_types=keep_fp32_op_types, + disable_shape_infer=disable_shape_infer, + ) + else: + if calibration_manifest is None: + raise ValueError("INT8 quantization requires --calibration-manifest") + settings = dict(calibration_manifest["quantization"]) + quantize_int8( + fp32_model, + output_model, + metadata_path=metadata_path, + calibration_manifest=calibration_manifest, + ) + return settings + + +def main() -> None: + from scripts.check_onnx_quantized_artifacts import load_calibration_manifest + + args = parse_args() + calibration_manifest = ( + load_calibration_manifest(args.calibration_manifest) + if args.calibration_manifest is not None + else None + ) + settings = quantize_once( + fp32_model=args.fp32_model, + metadata_path=args.metadata, + precision=args.precision, + output_model=args.output, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=tuple(args.keep_fp32_op_type), + disable_shape_infer=args.disable_shape_infer, + ) + output_sha256 = sha256_file(args.output) + if args.repeat_output is not None: + quantize_once( + fp32_model=args.fp32_model, + metadata_path=args.metadata, + precision=args.precision, + output_model=args.repeat_output, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=tuple(args.keep_fp32_op_type), + disable_shape_infer=args.disable_shape_infer, + ) + repeat_sha256 = sha256_file(args.repeat_output) + if args.require_identical_hash: + assert_identical_artifact_hashes(output_sha256, repeat_sha256) + + sidecar = args.sidecar or args.output.with_suffix(".json") + write_quantization_sidecar( + sidecar, + precision=args.precision, + source_model_sha256=sha256_file(args.fp32_model), + output_model_sha256=output_sha256, + framework_commit=framework_commit(), + checkpoint_revision=args.checkpoint_revision, + settings=settings, + toolchain=toolchain_versions(), + ) + print(json.dumps({"output": str(args.output), "sha256": output_sha256}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tests/test_onnx_quantization_accuracy_gate.py b/tests/test_onnx_quantization_accuracy_gate.py new file mode 100644 index 0000000..2cd7960 --- /dev/null +++ b/tests/test_onnx_quantization_accuracy_gate.py @@ -0,0 +1,72 @@ +from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report + + +def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "int8", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.17, "map50": 0.31}, + }, + } + thresholds = { + "precisions": { + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03} + } + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert any("map50_95" in failure for failure in failures) + assert not any("map50=" in failure for failure in failures) + + +def test_quantized_accuracy_accepts_candidate_within_threshold() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "nms-free", + "metrics": {"map50_95": 0.1, "map50": 0.14}, + }, + "candidate": { + "precision": "fp16", + "branch": "nms-free", + "metrics": {"map50_95": 0.098, "map50": 0.135}, + }, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + } + } + + assert check_quantized_accuracy_report(report, thresholds) == [] + + +def test_quantized_accuracy_rejects_branch_mismatch() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "fp16", + "branch": "nms-free", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + } + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["candidate branch nms-free does not match FP32 reference branch o2m-nms"] diff --git a/tests/test_onnx_quantization_artifact_checks.py b/tests/test_onnx_quantization_artifact_checks.py new file mode 100644 index 0000000..b7e86f0 --- /dev/null +++ b/tests/test_onnx_quantization_artifact_checks.py @@ -0,0 +1,33 @@ +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + check_provider_precision_supported, + check_unexpected_fp32_nodes, +) + + +def test_unsupported_provider_precision_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="does not support fp16"): + check_provider_precision_supported("CPUExecutionProvider", "fp16", thresholds) + + +def test_unknown_provider_fails_clearly() -> None: + thresholds = {"providers": {"CPUExecutionProvider": ["fp32", "int8"]}} + + with pytest.raises(ValueError, match="not configured"): + check_provider_precision_supported("MagicExecutionProvider", "fp16", thresholds) + + +def test_unexpected_fp32_nodes_are_reported() -> None: + report = { + "fp32_nodes": [ + {"name": "Conv_1", "op_type": "Conv"}, + {"name": "ReduceSum_1", "op_type": "ReduceSum"}, + ] + } + + unexpected = check_unexpected_fp32_nodes(report, allowlist=["ReduceSum"]) + + assert unexpected == ["Conv_1:Conv"] diff --git a/tests/test_onnx_quantization_benchmark.py b/tests/test_onnx_quantization_benchmark.py new file mode 100644 index 0000000..3f1c0f2 --- /dev/null +++ b/tests/test_onnx_quantization_benchmark.py @@ -0,0 +1,19 @@ +from scripts.benchmark_onnx_artifacts import summarize_latency_ms + + +def test_benchmark_report_uses_distribution_not_single_shot() -> None: + summary = summarize_latency_ms([10.0, 12.0, 14.0]) + + assert summary["median_ms"] == 12.0 + assert summary["mean_ms"] == 12.0 + assert summary["stddev_ms"] > 0.0 + assert summary["p95_ms"] == 14.0 + + +def test_benchmark_report_handles_empty_measurements() -> None: + summary = summarize_latency_ms([]) + + assert summary["median_ms"] == 0.0 + assert summary["mean_ms"] == 0.0 + assert summary["stddev_ms"] == 0.0 + assert summary["p95_ms"] == 0.0 diff --git a/tests/test_onnx_quantization_cli.py b/tests/test_onnx_quantization_cli.py new file mode 100644 index 0000000..ae7c4fd --- /dev/null +++ b/tests/test_onnx_quantization_cli.py @@ -0,0 +1,34 @@ +import json +from pathlib import Path + +import pytest + +from scripts.quantize_onnx import assert_identical_artifact_hashes, write_quantization_sidecar + + +def test_quantization_sidecar_binds_artifact_to_source_and_settings(tmp_path: Path) -> None: + sidecar = tmp_path / "model.fp16.json" + + write_quantization_sidecar( + sidecar, + precision="fp16", + source_model_sha256="source", + output_model_sha256="output", + framework_commit="commit", + checkpoint_revision="checkpoint", + settings={"keep_fp32_op_types": ["ReduceSum"]}, + toolchain={"onnx": "1.17.0"}, + ) + + data = json.loads(sidecar.read_text(encoding="utf-8")) + assert data["precision"] == "fp16" + assert data["source_model_sha256"] == "source" + assert data["output_model_sha256"] == "output" + assert data["framework_commit"] == "commit" + assert data["checkpoint_revision"] == "checkpoint" + assert data["settings"]["keep_fp32_op_types"] == ["ReduceSum"] + + +def test_repeat_quantization_hash_mismatch_fails_loudly() -> None: + with pytest.raises(ValueError, match="quantization is not deterministic"): + assert_identical_artifact_hashes("first", "second") diff --git a/tests/test_onnx_quantization_config.py b/tests/test_onnx_quantization_config.py new file mode 100644 index 0000000..2bcbb9a --- /dev/null +++ b/tests/test_onnx_quantization_config.py @@ -0,0 +1,227 @@ +from pathlib import Path + +import pytest + +from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, + image_id_manifest_sha256, + load_calibration_manifest, + load_quantization_thresholds, +) + +HASH = "a" * 64 +IMAGE_IDS = [9, 25, 30] +IMAGE_IDS_HASH = image_id_manifest_sha256(IMAGE_IDS) + + +def test_threshold_config_requires_explicit_release_policy(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text( + '{"schema_version": 1, "precisions": {"fp16": {}, "int8": {}}}', + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="release_policy"): + load_quantization_thresholds(path) + + +def test_threshold_config_requires_fp16_and_int8_policies(tmp_path: Path) -> None: + path = tmp_path / "thresholds.json" + path.write_text( + '{"schema_version": 1, "release_policy": {}, "precisions": {"fp16": {}}}', + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="fp16 and int8"): + load_quantization_thresholds(path) + + +def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + manifest = load_calibration_manifest(path) + + assert manifest["quantization"]["calibration_method"] == "minmax" + assert manifest["quantization"]["num_threads"] == 1 + + +def test_calibration_manifest_rejects_missing_quantization_setting( + tmp_path: Path, +) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8 + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="num_threads"): + load_calibration_manifest(path) + + +def test_calibration_manifest_rejects_placeholder_dataset_hash(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "replace-with-calibration-image-id-manifest-sha256", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "images": [ + "artifacts/COCO/images/train2017/000000000009.jpg", + "artifacts/COCO/images/train2017/000000000025.jpg", + "artifacts/COCO/images/train2017/000000000030.jpg" + ], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } + } + """ + % HASH, + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="image_ids_sha256"): + load_calibration_manifest(path) + + +def test_calibration_manifest_requires_pinned_image_identity(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } + } + """ + % (HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="image_ids"): + load_calibration_manifest(path) + + +def test_calibration_manifest_requires_calibration_image_paths(tmp_path: Path) -> None: + path = tmp_path / "calibration.json" + path.write_text( + """ + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": "%s", + "annotations_sha256": "%s", + "disjoint_from": "coco-2017-val2017" + }, + "image_ids": [9, 25, 30], + "quantization": { + "calibration_method": "minmax", + "per_channel": true, + "symmetric_activations": false, + "symmetric_weights": true, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 8, + "num_threads": 1 + } + } + """ + % (IMAGE_IDS_HASH, HASH), + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="images"): + load_calibration_manifest(path) + + +def test_calibration_and_eval_image_ids_must_be_disjoint() -> None: + with pytest.raises(ValueError, match="overlap"): + assert_disjoint_image_ids({1, 2, 3}, {3, 4, 5}) + + +def test_image_id_manifest_hash_is_order_stable() -> None: + assert image_id_manifest_sha256([3, 1, 2]) == image_id_manifest_sha256([1, 2, 3]) diff --git a/tests/test_tr_hash_agentic_250m_lab.py b/tests/test_tr_hash_agentic_250m_lab.py index 330d8f9..d4083ee 100644 --- a/tests/test_tr_hash_agentic_250m_lab.py +++ b/tests/test_tr_hash_agentic_250m_lab.py @@ -1,11 +1,15 @@ from __future__ import annotations import json -import tomllib from pathlib import Path from complexity.training.finetuning import validate_refinement_plan +try: + import tomllib +except ModuleNotFoundError: # pragma: no cover - Python 3.10 fallback + import tomli as tomllib + PROJECT_ROOT = Path(__file__).parents[1] PLAN_DIR = PROJECT_ROOT / "configs" / "replay_plans" From c6093db070b6cf820aaee68c4fff29396beb6d20 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Sat, 5 Sep 2026 10:38:03 +0500 Subject: [PATCH 2/5] Wire Vision v8 quantized release gates --- .github/workflows/onnx-release.yml | 6 +- .../vision_v8_quantization_calibration.json | 3 +- docs/onnx/release.json | 12 + .../2026-08-31-vision-v8-onnx-quantization.md | 12 +- docs/vision-v8-onnx-quantization.md | 19 +- scripts/build_onnx_release.py | 299 +++++++++++++++++- scripts/check_onnx_quantized_artifacts.py | 131 +++++++- scripts/check_onnx_quantized_parity.py | 179 +++++++++++ scripts/quantize_onnx.py | 71 ++++- tests/test_onnx_quantization_accuracy_gate.py | 103 +++++- tests/test_onnx_quantization_cli.py | 146 ++++++++- tests/test_onnx_quantization_config.py | 19 +- tests/test_onnx_release.py | 45 ++- 13 files changed, 983 insertions(+), 62 deletions(-) create mode 100644 scripts/check_onnx_quantized_parity.py diff --git a/.github/workflows/onnx-release.yml b/.github/workflows/onnx-release.yml index 4ba0e70..1c92312 100644 --- a/.github/workflows/onnx-release.yml +++ b/.github/workflows/onnx-release.yml @@ -104,9 +104,5 @@ jobs: --notes-file dist/onnx/RELEASE_NOTES.md \ $DRAFT gh release upload "$TAG" \ - dist/onnx/manifest.json \ - dist/onnx/tr_hash_v8_o2m.onnx \ - dist/onnx/tr_hash_v8_o2m.json \ - dist/onnx/tr_hash_v8_nms_free.onnx \ - dist/onnx/tr_hash_v8_nms_free.json \ + dist/onnx/* \ --clobber diff --git a/configs/vision_v8_quantization_calibration.json b/configs/vision_v8_quantization_calibration.json index 15302a6..5caf831 100644 --- a/configs/vision_v8_quantization_calibration.json +++ b/configs/vision_v8_quantization_calibration.json @@ -14,7 +14,6 @@ "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } diff --git a/docs/onnx/release.json b/docs/onnx/release.json index 77f6c2d..ab080ba 100644 --- a/docs/onnx/release.json +++ b/docs/onnx/release.json @@ -3,6 +3,18 @@ "checkpoint_revision": "f3b3e659612e543ca9ff91892c0662d38dc1a1d6", "opset": 17, "parity_num_tests": 50, + "quantization": { + "enabled_precisions": ["fp16", "int8"], + "calibration_manifest": "configs/vision_v8_quantization_calibration.json", + "thresholds": "configs/vision_v8_quantization_thresholds.json", + "accuracy_report": "artifacts/vision_v8_quantized_eval/accuracy.json", + "accuracy_markdown": "artifacts/vision_v8_quantized_eval/accuracy.md", + "fp32_op_allowlist": [], + "provider_gates": [ + {"provider": "CUDAExecutionProvider", "precision": "fp16"}, + {"provider": "CPUExecutionProvider", "precision": "int8"} + ] + }, "toolchain": { "torch": "2.13.0", "onnx": "1.21.0", diff --git a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md index 671136e..e0aaae1 100644 --- a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md +++ b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md @@ -16,7 +16,7 @@ - Calibration inputs must be pinned and disjoint from the COCO evaluation inputs used by the accuracy gate. - Unsupported provider/precision combinations must fail clearly instead of falling back silently. - FP16 conversion must report remaining FP32 nodes and fail unless every remaining FP32 node is covered by a checked-in allowlist. -- INT8 settings must pin calibration method, per-channel/per-tensor mode, symmetric/asymmetric mode, activation type, weight type, calibration batch size, and calibration thread count. +- INT8 settings must pin calibration method, per-channel/per-tensor mode, symmetric/asymmetric mode, activation type, weight type, and calibration batch size. - FP32, FP16, and INT8 reports must compare artifact size, latency, throughput, peak memory, raw-logit parity, decoded-output parity, and COCO metrics. - Tolerances and floors must live in checked-in config, not inside scripts. - If any required quantized artifact fails its gate, the release blocks; do not publish a partial FP32+FP16-only release unless the release config explicitly marks INT8 optional for that provider. @@ -85,8 +85,7 @@ def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } """ @@ -95,7 +94,7 @@ def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: manifest = load_calibration_manifest(path) assert manifest["quantization"]["calibration_method"] == "minmax" - assert manifest["quantization"]["num_threads"] == 1 + assert manifest["quantization"]["batch_size"] == 8 ``` - [ ] **Step 2: Run tests and verify they fail** @@ -166,8 +165,7 @@ Create `configs/vision_v8_quantization_calibration.json`: "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } ``` @@ -214,7 +212,6 @@ def load_calibration_manifest(path: Path) -> dict[str, Any]: "activation_type", "weight_type", "batch_size", - "num_threads", } missing = sorted(required - set(quantization)) if missing: @@ -383,7 +380,6 @@ calibration_method = manifest["quantization"]["calibration_method"] per_channel = manifest["quantization"]["per_channel"] activation_type = manifest["quantization"]["activation_type"] weight_type = manifest["quantization"]["weight_type"] -num_threads = manifest["quantization"]["num_threads"] ``` Set calibration reader iteration order from the sorted calibration manifest. diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md index 1d03de5..9a4b496 100644 --- a/docs/vision-v8-onnx-quantization.md +++ b/docs/vision-v8-onnx-quantization.md @@ -27,7 +27,9 @@ INT8 calibration is pinned by `configs/vision_v8_quantization_calibration.json`. The manifest records the dataset, image-ID manifest hash, annotation hash, calibration method, per-channel/per-tensor mode, symmetric/asymmetric choices, activation type, -weight type, batch size, and calibration thread count. +weight type, and batch size. The ORT symmetry options and batch size are passed +into quantization directly; unsupported settings are not recorded in release +metadata. Placeholder hashes are rejected by `load_calibration_manifest`; before a release run, replace them with the approved calibration subset and real SHA-256 values. @@ -50,6 +52,11 @@ python scripts/quantize_onnx.py \ --checkpoint-revision AETHORIA-AI/TR-HASH-Vision-v8-2M-COCO-SFT@REVISION ``` +By default the CLI writes two JSON files for a quantized artifact: +`tr_hash_v8_o2m_fp16.json` is a detector metadata copy used by the inference +pipeline, while `tr_hash_v8_o2m_fp16.quantization.json` is the quantization +provenance sidecar used by release verification. + Create an INT8 artifact: ```bash @@ -77,6 +84,12 @@ python scripts/evaluate_onnx_coco.py \ --provider cuda ``` +Before publishing, merge the FP32 reference and quantized branch reports into +`artifacts/vision_v8_quantized_eval/accuracy.json` and +`artifacts/vision_v8_quantized_eval/accuracy.md`. The release builder validates +that JSON against `configs/vision_v8_quantization_thresholds.json`; missing or +regressed reports block publication. + Benchmark an artifact: ```bash @@ -99,7 +112,9 @@ optional. Release assets must include: - FP32, FP16, and INT8 ONNX files for both O2M and NMS-free branches. -- JSON sidecars for each artifact. +- Detector metadata JSON sidecars and quantization provenance JSON sidecars. +- FP32-vs-quantized parity reports for raw logits, decoded boxes, scores, and + class/count stability. - Accuracy reports in JSON and Markdown. - Benchmark reports in JSON and Markdown. - A manifest binding every artifact to framework commit, checkpoint revision, diff --git a/scripts/build_onnx_release.py b/scripts/build_onnx_release.py index 7c31101..a3bace1 100644 --- a/scripts/build_onnx_release.py +++ b/scripts/build_onnx_release.py @@ -61,6 +61,20 @@ class ReleaseConfig: parity_num_tests: int toolchain: Mapping[str, str] branches: tuple[BranchSpec, ...] + quantization: QuantizationConfig | None = None + + +@dataclass(frozen=True) +class QuantizationConfig: + """Pinned inputs for optional quantized release artifacts.""" + + enabled_precisions: tuple[str, ...] + calibration_manifest: Path + thresholds: Path + accuracy_report: Path + accuracy_markdown: Path + fp32_op_allowlist: tuple[str, ...] + provider_gates: tuple[tuple[str, str], ...] class ReleaseError(RuntimeError): @@ -102,6 +116,33 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: if not isinstance(toolchain, Mapping) or not toolchain: raise ReleaseError("release config must pin a toolchain") + quantization = None + if isinstance(values.get("quantization"), Mapping): + raw_quantization = values["quantization"] + enabled_precisions = tuple( + str(precision) + for precision in raw_quantization.get("enabled_precisions", ()) + ) + unsupported = sorted(set(enabled_precisions) - {"fp16", "int8"}) + if unsupported: + raise ReleaseError(f"unsupported quantized precisions: {unsupported}") + provider_gates = tuple( + (str(gate["provider"]), str(gate["precision"])) + for gate in raw_quantization.get("provider_gates", ()) + ) + quantization = QuantizationConfig( + enabled_precisions=enabled_precisions, + calibration_manifest=Path(str(raw_quantization["calibration_manifest"])), + thresholds=Path(str(raw_quantization["thresholds"])), + accuracy_report=Path(str(raw_quantization["accuracy_report"])), + accuracy_markdown=Path(str(raw_quantization["accuracy_markdown"])), + fp32_op_allowlist=tuple( + str(op_type) + for op_type in raw_quantization.get("fp32_op_allowlist", ()) + ), + provider_gates=provider_gates, + ) + return ReleaseConfig( checkpoint_repo=str(values["checkpoint_repo"]), checkpoint_revision=revision, @@ -109,6 +150,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: parity_num_tests=int(values.get("parity_num_tests", 5)), toolchain={str(k): str(v) for k, v in toolchain.items()}, branches=branches, + quantization=quantization, ) @@ -175,6 +217,18 @@ def artifact_entry(path: Path, **fields: Any) -> dict[str, Any]: } +def provider_chain(provider: str) -> tuple[str, ...]: + if provider == "TensorrtExecutionProvider": + return ( + "TensorrtExecutionProvider", + "CUDAExecutionProvider", + "CPUExecutionProvider", + ) + if provider == "CUDAExecutionProvider": + return ("CUDAExecutionProvider", "CPUExecutionProvider") + return (provider,) + + def output_contract(sidecar: Mapping[str, Any]) -> dict[str, Any]: """Describe the ONNX input/output contract from an export sidecar.""" @@ -267,14 +321,15 @@ def render_release_notes(manifest: Mapping[str, Any]) -> str: "Both models expose raw detector logits only; decode and post-processing " "run outside the graph.", "", - "| Branch | Model | Post-processing | Size | SHA-256 |", - "|---|---|---|---:|---|", + "| Branch | Precision | Model | Post-processing | Size | SHA-256 |", + "|---|---|---|---|---:|---|", ] for artifact in manifest["artifacts"]: if artifact.get("kind") != "model": continue lines.append( - f"| {artifact['branch']} | `{artifact['name']}` | " + f"| {artifact['branch']} | {artifact.get('precision', 'unknown')} | " + f"`{artifact['name']}` | " f"{artifact['post_processing']} | {artifact['size_bytes']:,} bytes | " f"`{artifact['sha256']}` |" ) @@ -318,13 +373,34 @@ def build_release( from huggingface_hub import snapshot_download from scripts.check_onnx_parity import check_parity + from scripts.check_onnx_quantized_artifacts import ( + check_provider_precision_supported, + check_quantized_accuracy_report, + check_quantized_parity_report, + check_unexpected_fp32_nodes, + inspect_onnx_node_dtypes, + load_calibration_manifest, + load_quantization_thresholds, + ) + from scripts.check_onnx_quantized_parity import build_parity_report from scripts.export_onnx import export_onnx + from scripts.quantize_onnx import ( + assert_identical_artifact_hashes, + copy_detector_metadata, + default_quantization_sidecar, + quantize_once, + write_quantization_sidecar, + ) + from scripts.quantize_onnx import ( + toolchain_versions as quantization_toolchain_versions, + ) installed = installed_toolchain() mismatches = toolchain_mismatches(config.toolchain, installed) if mismatches: - message = "toolchain does not match the pinned release toolchain:\n " + "\n ".join( - mismatches + message = ( + "toolchain does not match the pinned release toolchain:\n " + + "\n ".join(mismatches) ) if not allow_toolchain_drift: raise ReleaseError( @@ -335,6 +411,76 @@ def build_release( destination = Path(output_dir) destination.mkdir(parents=True, exist_ok=True) + commit = framework_commit() + artifacts: list[dict[str, Any]] = [] + + quantization_thresholds: Mapping[str, Any] | None = None + calibration_manifest: Mapping[str, Any] | None = None + provider_by_precision: dict[str, str] = {} + if config.quantization is not None: + quantization_thresholds = load_quantization_thresholds( + config.quantization.thresholds + ) + for provider, precision in config.quantization.provider_gates: + check_provider_precision_supported( + provider, + precision, + quantization_thresholds, + ) + provider_by_precision[precision] = provider + if config.quantization.enabled_precisions: + calibration_manifest = load_calibration_manifest( + config.quantization.calibration_manifest + ) + if config.quantization.enabled_precisions and calibration_manifest is None: + raise ReleaseError("quantized release parity requires calibration images") + parity_image = Path(str(calibration_manifest["images"][0])) + if not config.quantization.accuracy_report.is_file(): + raise ReleaseError( + "quantized release requires an accuracy comparison report: " + f"{config.quantization.accuracy_report}" + ) + if not config.quantization.accuracy_markdown.is_file(): + raise ReleaseError( + "quantized release requires a Markdown accuracy report: " + f"{config.quantization.accuracy_markdown}" + ) + accuracy_report = json.loads( + config.quantization.accuracy_report.read_text(encoding="utf-8") + ) + accuracy_failures = check_quantized_accuracy_report( + accuracy_report, + quantization_thresholds, + ) + if accuracy_failures: + raise ReleaseError( + "quantized COCO accuracy gate failed:\n " + + "\n ".join(accuracy_failures) + ) + accuracy_report_path = destination / "quantized_accuracy.json" + accuracy_markdown_path = destination / "quantized_accuracy.md" + accuracy_report_path.write_text( + json.dumps(accuracy_report, indent=2) + "\n", + encoding="utf-8", + ) + accuracy_markdown_path.write_text( + config.quantization.accuracy_markdown.read_text(encoding="utf-8"), + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + accuracy_report_path, + kind="accuracy_report", + precision="mixed", + ) + ) + artifacts.append( + artifact_entry( + accuracy_markdown_path, + kind="accuracy_report_markdown", + precision="mixed", + ) + ) print(f"Downloading {config.checkpoint_repo}@{config.checkpoint_revision}") checkpoint_root = Path( @@ -344,7 +490,6 @@ def build_release( ) ) - artifacts: list[dict[str, Any]] = [] for spec in config.branches: checkpoint = ( checkpoint_root @@ -377,19 +522,157 @@ def build_release( model_path, kind="model", branch=spec.branch, + precision="fp32", requires_nms=bool(sidecar["requires_nms"]), post_processing=spec.post_processing, contract=output_contract(sidecar), ) ) artifacts.append( - artifact_entry(sidecar_path, kind="metadata", branch=spec.branch) + artifact_entry( + sidecar_path, + kind="metadata", + branch=spec.branch, + precision="fp32", + ) ) + if config.quantization is not None and quantization_thresholds is not None: + for precision in config.quantization.enabled_precisions: + quantized_model_path = destination / f"{spec.stem}_{precision}.onnx" + repeat_model_path = destination / f"{spec.stem}_{precision}_repeat.onnx" + print(f"Quantizing {spec.branch} to {precision}") + settings = quantize_once( + fp32_model=model_path, + metadata_path=sidecar_path, + precision=precision, + output_model=quantized_model_path, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=config.quantization.fp32_op_allowlist, + disable_shape_infer=False, + ) + quantize_once( + fp32_model=model_path, + metadata_path=sidecar_path, + precision=precision, + output_model=repeat_model_path, + calibration_manifest=calibration_manifest, + keep_fp32_op_types=config.quantization.fp32_op_allowlist, + disable_shape_infer=False, + ) + assert_identical_artifact_hashes( + sha256_file(quantized_model_path), + sha256_file(repeat_model_path), + ) + repeat_model_path.unlink() + + precision_thresholds = quantization_thresholds["precisions"][precision] + if precision_thresholds.get("unexpected_fp32_nodes") == "fail": + dtype_report = inspect_onnx_node_dtypes(quantized_model_path) + unexpected = check_unexpected_fp32_nodes( + dtype_report, + allowlist=config.quantization.fp32_op_allowlist, + ) + if unexpected: + raise ReleaseError( + f"{quantized_model_path.name} retained unexpected " + "FP32 nodes: " + + ", ".join(unexpected) + ) + + quantized_metadata_path = quantized_model_path.with_suffix(".json") + copy_detector_metadata(sidecar_path, quantized_metadata_path) + quantized_sidecar_path = default_quantization_sidecar( + quantized_model_path + ) + write_quantization_sidecar( + quantized_sidecar_path, + precision=precision, + source_model_sha256=sha256_file(model_path), + output_model_sha256=sha256_file(quantized_model_path), + framework_commit=commit, + checkpoint_revision=config.checkpoint_revision, + settings=settings, + toolchain=quantization_toolchain_versions(), + ) + parity_report = build_parity_report( + reference_model=model_path, + candidate_model=quantized_model_path, + metadata_path=sidecar_path, + image_path=parity_image, + precision=precision, + providers=provider_chain( + provider_by_precision.get(precision, "CPUExecutionProvider") + ), + ) + provider_used = str(parity_report["provider_used"]) + check_provider_precision_supported( + provider_used, + precision, + quantization_thresholds, + ) + parity_failures = check_quantized_parity_report( + parity_report, + quantization_thresholds, + ) + if int(parity_report["class_mismatch_count"]) > 0: + parity_failures.append( + f"{spec.branch} {precision} decoded class/count mismatch: " + f"{parity_report['class_mismatch_count']}" + ) + if parity_failures: + raise ReleaseError( + f"{quantized_model_path.name} failed quantized parity:\n " + + "\n ".join(parity_failures) + ) + parity_report_path = ( + destination / f"{spec.stem}_{precision}_parity.json" + ) + parity_report_path.write_text( + json.dumps(parity_report, indent=2) + "\n", + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + quantized_model_path, + kind="model", + branch=spec.branch, + precision=precision, + source_model=Path(model_path).name, + requires_nms=bool(sidecar["requires_nms"]), + post_processing=spec.post_processing, + contract=output_contract(sidecar), + ) + ) + artifacts.append( + artifact_entry( + quantized_metadata_path, + kind="metadata", + branch=spec.branch, + precision=precision, + ) + ) + artifacts.append( + artifact_entry( + quantized_sidecar_path, + kind="quantization_metadata", + branch=spec.branch, + precision=precision, + source_model=Path(model_path).name, + ) + ) + artifacts.append( + artifact_entry( + parity_report_path, + kind="parity_report", + branch=spec.branch, + precision=precision, + ) + ) manifest = build_manifest( config, artifacts, - commit=framework_commit(), + commit=commit, toolchain=installed, ) manifest_path = destination / MANIFEST_NAME diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py index a36e626..dce400f 100644 --- a/scripts/check_onnx_quantized_artifacts.py +++ b/scripts/check_onnx_quantized_artifacts.py @@ -4,6 +4,7 @@ import hashlib import json +import math import string from collections.abc import Mapping, Sequence from pathlib import Path @@ -24,7 +25,11 @@ def load_quantization_thresholds(path: Path) -> dict[str, Any]: if "release_policy" not in config: raise ValueError("threshold config missing release_policy") precisions = config.get("precisions") - if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: + if ( + not isinstance(precisions, dict) + or "fp16" not in precisions + or "int8" not in precisions + ): raise ValueError("threshold config must define fp16 and int8 precisions") return config @@ -53,7 +58,6 @@ def load_calibration_manifest(path: Path) -> dict[str, Any]: "activation_type", "weight_type", "batch_size", - "num_threads", } missing = sorted(required - set(quantization)) if missing: @@ -108,12 +112,16 @@ def check_provider_precision_supported( providers = thresholds.get("providers", {}) if not isinstance(providers, Mapping) or provider not in providers: - raise ValueError(f"{provider} is not configured in quantization release config") + raise ValueError( + f"{provider} is not configured in quantization release config" + ) supported = providers[provider] if not isinstance(supported, Sequence) or isinstance(supported, (str, bytes)): raise ValueError(f"{provider} precision policy must be a sequence") if precision not in supported: - raise ValueError(f"{provider} does not support {precision} in quantization release config") + raise ValueError( + f"{provider} does not support {precision} in quantization release config" + ) def check_unexpected_fp32_nodes( @@ -149,7 +157,9 @@ def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: import onnx from onnx import TensorProto except ImportError as error: # pragma: no cover - dependency guard - raise RuntimeError("ONNX dtype inspection requires the onnx package") from error + raise RuntimeError( + "ONNX dtype inspection requires the onnx package" + ) from error model = onnx.load(str(model_path)) value_dtypes: dict[str, int] = {} @@ -173,9 +183,13 @@ def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: fp16_nodes = 0 int8_nodes = 0 for node in model.graph.node: - output_types = {value_dtypes[name] for name in node.output if name in value_dtypes} + output_types = { + value_dtypes[name] for name in node.output if name in value_dtypes + } if TensorProto.FLOAT in output_types: - fp32_nodes.append({"name": node.name or node.output[0], "op_type": node.op_type}) + fp32_nodes.append( + {"name": node.name or node.output[0], "op_type": node.op_type} + ) if TensorProto.FLOAT16 in output_types: fp16_nodes += 1 if TensorProto.INT8 in output_types or TensorProto.UINT8 in output_types: @@ -194,18 +208,100 @@ def check_quantized_accuracy_report( ) -> list[str]: """Compare a quantized candidate COCO report against its FP32 reference.""" + if "branches" in report: + return _check_branch_accuracy_report(report, thresholds) + reference = _mapping(report.get("reference")) candidate = _mapping(report.get("candidate")) + return _check_accuracy_pair(reference, candidate, thresholds) + + +def check_quantized_parity_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + """Gate raw-logit and decoded-output drift against checked-in thresholds.""" + + precision = str(report.get("precision", "")) + branch = str(report.get("branch", "")) + precision_thresholds = _mapping( + _mapping(thresholds.get("precisions")).get(precision) + ) + if not precision_thresholds: + return [f"candidate precision {precision} has no quantization thresholds"] + + failures: list[str] = [] + for metric, threshold_name in ( + ("max_raw_logit_abs_error", "max_raw_logit_abs_error"), + ("max_decoded_box_px_error", "max_decoded_box_px_error"), + ("max_score_abs_error", "max_score_abs_error"), + ): + if metric not in report: + failures.append(f"missing parity metric {metric}") + continue + value = _finite_float(report[metric]) + if value is None: + failures.append(f"non-finite parity metric {metric}") + continue + allowed = float(precision_thresholds[threshold_name]) + if value > allowed: + failures.append( + f"{precision} {branch} {metric} {value:.6f} exceeds {allowed:.6f}" + ) + return failures + + +def _check_branch_accuracy_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + branches = _mapping(report.get("branches")) + if not branches: + return ["quantized COCO report must contain branch comparisons"] + reference_precision = str(report.get("reference_precision", "fp32")) + candidate_precision = str( + report.get("candidate_precision", report.get("precision", "")) + ) + if not candidate_precision: + return ["candidate precision missing from quantized COCO report"] + + failures: list[str] = [] + for branch, branch_report in branches.items(): + branch_data = _mapping(branch_report) + reference = _mapping( + branch_data.get(reference_precision, branch_data.get("reference")) + ) + candidate = _mapping( + branch_data.get(candidate_precision, branch_data.get("candidate")) + ) + if not reference or not candidate: + failures.append( + f"{branch} missing {reference_precision} " + f"or {candidate_precision} metrics" + ) + continue + failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) + return failures + + +def _check_accuracy_pair( + reference: Mapping[str, Any], + candidate: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: reference_branch = str(reference.get("branch", "")) candidate_branch = str(candidate.get("branch", "")) if reference_branch != candidate_branch: return [ "candidate branch " - f"{candidate_branch} does not match FP32 reference branch {reference_branch}" + f"{candidate_branch} does not match FP32 reference branch " + f"{reference_branch}" ] precision = str(candidate.get("precision", "")) - precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) + precision_thresholds = _mapping( + _mapping(thresholds.get("precisions")).get(precision) + ) if not precision_thresholds: return [f"candidate precision {precision} has no quantization thresholds"] @@ -219,8 +315,13 @@ def check_quantized_accuracy_report( if metric not in reference_metrics or metric not in candidate_metrics: failures.append(f"missing metric {metric} in FP32 or candidate report") continue + reference_value = _finite_float(reference_metrics[metric]) + candidate_value = _finite_float(candidate_metrics[metric]) + if reference_value is None or candidate_value is None: + failures.append(f"non-finite metric {metric} in FP32 or candidate report") + continue allowed_drop = float(precision_thresholds[threshold_name]) - drop = float(reference_metrics[metric]) - float(candidate_metrics[metric]) + drop = reference_value - candidate_value if drop > allowed_drop: failures.append( f"{precision} {candidate_branch} {metric} dropped by {drop:.6f}; " @@ -233,6 +334,16 @@ def _mapping(value: object) -> Mapping[str, Any]: return value if isinstance(value, Mapping) else {} +def _finite_float(value: object) -> float | None: + try: + converted = float(value) + except (TypeError, ValueError): + return None + if not math.isfinite(converted): + return None + return converted + + def _is_sha256(value: object) -> bool: if not isinstance(value, str) or len(value) != 64: return False diff --git a/scripts/check_onnx_quantized_parity.py b/scripts/check_onnx_quantized_parity.py new file mode 100644 index 0000000..ba7a9ca --- /dev/null +++ b/scripts/check_onnx_quantized_parity.py @@ -0,0 +1,179 @@ +"""Gate FP32-vs-quantized Vision v8 ONNX raw and decoded parity.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from typing import Sequence + +import numpy as np + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--reference-model", type=Path, required=True) + parser.add_argument("--candidate-model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--image", type=Path, required=True) + parser.add_argument("--precision", choices=("fp16", "int8"), required=True) + parser.add_argument("--thresholds", type=Path, required=True) + parser.add_argument( + "--provider", + action="append", + default=None, + help="Provider alias or ORT provider name. May be repeated or comma-separated.", + ) + parser.add_argument("--output", type=Path) + return parser.parse_args() + + +def decoded_parity_metrics( + reference_detections: Sequence[object], + candidate_detections: Sequence[object], +) -> dict[str, float | int]: + """Compare decoded detections in stable score order.""" + + count = min(len(reference_detections), len(candidate_detections)) + if count == 0: + return { + "reference_detection_count": len(reference_detections), + "candidate_detection_count": len(candidate_detections), + "class_mismatch_count": int( + len(reference_detections) != len(candidate_detections) + ), + "max_decoded_box_px_error": 0.0, + "max_score_abs_error": 0.0, + } + + box_errors = [] + score_errors = [] + class_mismatches = abs(len(reference_detections) - len(candidate_detections)) + for reference, candidate in zip( + reference_detections[:count], + candidate_detections[:count], + ): + box_errors.append( + float( + np.max( + np.abs( + np.asarray(reference.box_pixel, dtype=np.float32) + - np.asarray(candidate.box_pixel, dtype=np.float32) + ) + ) + ) + ) + score_errors.append(abs(float(reference.score) - float(candidate.score))) + class_mismatches += int(reference.class_id != candidate.class_id) + + return { + "reference_detection_count": len(reference_detections), + "candidate_detection_count": len(candidate_detections), + "class_mismatch_count": class_mismatches, + "max_decoded_box_px_error": max(box_errors), + "max_score_abs_error": max(score_errors), + } + + +def build_parity_report( + *, + reference_model: Path, + candidate_model: Path, + metadata_path: Path, + image_path: Path, + precision: str, + providers: Sequence[str], +) -> dict[str, object]: + from complexity.deploy.onnx_detector import OnnxDetectorPipeline + from complexity.deploy.onnx_detector.metadata import load_metadata + from complexity.deploy.onnx_detector.preprocess import preprocess_image + + metadata = load_metadata(metadata_path) + preprocessed = preprocess_image(image_path, metadata.image_size) + + reference_session = OnnxDetectorPipeline.create_session( + reference_model, + providers=providers, + warmup_runs=0, + ).open() + candidate_session = OnnxDetectorPipeline.create_session( + candidate_model, + providers=providers, + warmup_runs=0, + ).open() + + reference_raw = reference_session.run(preprocessed.pixel_values) + candidate_raw = candidate_session.run(preprocessed.pixel_values) + reference_pipeline = OnnxDetectorPipeline( + metadata=metadata, + session=reference_session, + ) + candidate_pipeline = OnnxDetectorPipeline( + metadata=metadata, + session=candidate_session, + ) + reference_result = reference_pipeline.predict(image_path) + candidate_result = candidate_pipeline.predict(image_path) + + return { + "schema_version": 1, + "branch": metadata.branch, + "precision": precision, + "provider_used": candidate_session.provider_used, + "max_raw_logit_abs_error": float(np.max(np.abs(reference_raw - candidate_raw))), + **decoded_parity_metrics( + reference_result.detections, + candidate_result.detections, + ), + } + + +def main() -> None: + from scripts.check_onnx_quantized_artifacts import ( + check_provider_precision_supported, + check_quantized_parity_report, + load_quantization_thresholds, + ) + from scripts.onnx_detect import provider_names + + args = parse_args() + thresholds = load_quantization_thresholds(args.thresholds) + providers = provider_names(args.provider) + report = build_parity_report( + reference_model=args.reference_model, + candidate_model=args.candidate_model, + metadata_path=args.metadata, + image_path=args.image, + precision=args.precision, + providers=providers, + ) + check_provider_precision_supported( + str(report["provider_used"]), + args.precision, + thresholds, + ) + failures = check_quantized_parity_report(report, thresholds) + if int(report["class_mismatch_count"]) > 0: + failures.append( + f"{report['branch']} {args.precision} decoded class/count mismatch: " + f"{report['class_mismatch_count']}" + ) + + if args.output is not None: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + if failures: + print("Quantized parity FAILED:", file=sys.stderr) + for failure in failures: + print(f" {failure}", file=sys.stderr) + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/scripts/quantize_onnx.py b/scripts/quantize_onnx.py index 89f846a..b45d2e3 100644 --- a/scripts/quantize_onnx.py +++ b/scripts/quantize_onnx.py @@ -28,7 +28,19 @@ def parse_args() -> argparse.Namespace: parser.add_argument("--metadata", type=Path, required=True) parser.add_argument("--precision", choices=("fp16", "int8"), required=True) parser.add_argument("--output", type=Path, required=True) - parser.add_argument("--sidecar", type=Path) + parser.add_argument( + "--sidecar", + type=Path, + help="Quantization provenance sidecar. Defaults to .quantization.json.", + ) + parser.add_argument( + "--detector-metadata-output", + type=Path, + help=( + "Detector metadata sidecar copied from --metadata. " + "Defaults to .json." + ), + ) parser.add_argument("--calibration-manifest", type=Path) parser.add_argument("--checkpoint-revision", default="unknown") parser.add_argument( @@ -116,6 +128,19 @@ def write_quantization_sidecar( path.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") +def default_quantization_sidecar(output_model: Path) -> Path: + return output_model.with_name(f"{output_model.stem}.quantization.json") + + +def default_detector_metadata_output(output_model: Path) -> Path: + return output_model.with_suffix(".json") + + +def copy_detector_metadata(source: Path, destination: Path) -> None: + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, destination) + + def quantize_fp32(input_model: Path, output_model: Path) -> None: output_model.parent.mkdir(parents=True, exist_ok=True) shutil.copyfile(input_model, output_model) @@ -154,6 +179,7 @@ def __init__( *, image_paths: Sequence[Path], metadata_path: Path, + batch_size: int = 1, input_name: str = "pixel_values", ) -> None: from complexity.deploy.onnx_detector.metadata import load_metadata @@ -161,8 +187,11 @@ def __init__( metadata = load_metadata(metadata_path) self._input_name = input_name + if batch_size <= 0: + raise ValueError("calibration batch_size must be positive") + self._batch_size = batch_size self._items = [ - {input_name: preprocess_image(path, metadata.image_size).pixel_values} + preprocess_image(path, metadata.image_size).pixel_values for path in image_paths ] self._index = 0 @@ -170,9 +199,9 @@ def __init__( def get_next(self) -> dict[str, np.ndarray] | None: if self._index >= len(self._items): return None - item = self._items[self._index] - self._index += 1 - return item + batch = self._items[self._index : self._index + self._batch_size] + self._index += self._batch_size + return {self._input_name: np.concatenate(batch, axis=0)} def _calibration_method(name: str) -> Any: @@ -217,14 +246,19 @@ def quantize_int8( from onnxruntime.quantization import QuantFormat, quantize_static except ImportError as error: # pragma: no cover - dependency guard raise RuntimeError( - "INT8 quantization requires onnxruntime.quantization and a calibration manifest." + "INT8 quantization requires onnxruntime.quantization " + "and a calibration manifest." ) from error settings = calibration_manifest["quantization"] image_paths = _calibration_paths(calibration_manifest) if not image_paths: raise ValueError("INT8 quantization requires calibration manifest images") - reader = _CalibrationReader(image_paths=image_paths, metadata_path=metadata_path) + reader = _CalibrationReader( + image_paths=image_paths, + metadata_path=metadata_path, + batch_size=int(settings["batch_size"]), + ) output_model.parent.mkdir(parents=True, exist_ok=True) quantize_static( str(input_model), @@ -235,6 +269,10 @@ def quantize_int8( per_channel=bool(settings["per_channel"]), activation_type=_quant_type(str(settings["activation_type"])), weight_type=_quant_type(str(settings["weight_type"])), + extra_options={ + "ActivationSymmetric": bool(settings["symmetric_activations"]), + "WeightSymmetric": bool(settings["symmetric_weights"]), + }, ) @@ -266,6 +304,18 @@ def quantize_once( if calibration_manifest is None: raise ValueError("INT8 quantization requires --calibration-manifest") settings = dict(calibration_manifest["quantization"]) + settings = { + key: settings[key] + for key in ( + "calibration_method", + "per_channel", + "symmetric_activations", + "symmetric_weights", + "activation_type", + "weight_type", + "batch_size", + ) + } quantize_int8( fp32_model, output_model, @@ -308,7 +358,12 @@ def main() -> None: if args.require_identical_hash: assert_identical_artifact_hashes(output_sha256, repeat_sha256) - sidecar = args.sidecar or args.output.with_suffix(".json") + detector_metadata_output = ( + args.detector_metadata_output or default_detector_metadata_output(args.output) + ) + copy_detector_metadata(args.metadata, detector_metadata_output) + + sidecar = args.sidecar or default_quantization_sidecar(args.output) write_quantization_sidecar( sidecar, precision=args.precision, diff --git a/tests/test_onnx_quantization_accuracy_gate.py b/tests/test_onnx_quantization_accuracy_gate.py index 2cd7960..f8b16d0 100644 --- a/tests/test_onnx_quantization_accuracy_gate.py +++ b/tests/test_onnx_quantization_accuracy_gate.py @@ -1,4 +1,7 @@ -from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report +from scripts.check_onnx_quantized_artifacts import ( + check_quantized_accuracy_report, + check_quantized_parity_report, +) def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> None: @@ -69,4 +72,100 @@ def test_quantized_accuracy_rejects_branch_mismatch() -> None: failures = check_quantized_accuracy_report(report, thresholds) - assert failures == ["candidate branch nms-free does not match FP32 reference branch o2m-nms"] + assert failures == [ + "candidate branch nms-free does not match FP32 reference branch o2m-nms" + ] + + +def test_quantized_accuracy_accepts_evaluator_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.315}, + }, + } + }, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + } + } + + assert check_quantized_accuracy_report(report, thresholds) == [] + + +def test_quantized_accuracy_rejects_empty_evaluator_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": {}, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + } + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["quantized COCO report must contain branch comparisons"] + + +def test_quantized_accuracy_rejects_non_finite_metrics() -> None: + report = { + "reference": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "candidate": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": float("nan"), "map50": 0.32}, + }, + } + thresholds = { + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + } + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["non-finite metric map50_95 in FP32 or candidate report"] + + +def test_quantized_parity_consumes_raw_and_decoded_thresholds() -> None: + report = { + "precision": "int8", + "branch": "o2m-nms", + "max_raw_logit_abs_error": 0.13, + "max_decoded_box_px_error": 3.0, + "max_score_abs_error": 0.04, + } + thresholds = { + "precisions": { + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + } + } + } + + failures = check_quantized_parity_report(report, thresholds) + + assert failures == [ + "int8 o2m-nms max_raw_logit_abs_error 0.130000 exceeds 0.120000" + ] diff --git a/tests/test_onnx_quantization_cli.py b/tests/test_onnx_quantization_cli.py index ae7c4fd..64831bd 100644 --- a/tests/test_onnx_quantization_cli.py +++ b/tests/test_onnx_quantization_cli.py @@ -1,12 +1,24 @@ import json +import sys +import types from pathlib import Path import pytest -from scripts.quantize_onnx import assert_identical_artifact_hashes, write_quantization_sidecar +from scripts import quantize_onnx +from scripts.quantize_onnx import ( + assert_identical_artifact_hashes, + copy_detector_metadata, + default_detector_metadata_output, + default_quantization_sidecar, + quantize_int8, + write_quantization_sidecar, +) -def test_quantization_sidecar_binds_artifact_to_source_and_settings(tmp_path: Path) -> None: +def test_quantization_sidecar_binds_artifact_to_source_and_settings( + tmp_path: Path, +) -> None: sidecar = tmp_path / "model.fp16.json" write_quantization_sidecar( @@ -32,3 +44,133 @@ def test_quantization_sidecar_binds_artifact_to_source_and_settings(tmp_path: Pa def test_repeat_quantization_hash_mismatch_fails_loudly() -> None: with pytest.raises(ValueError, match="quantization is not deterministic"): assert_identical_artifact_hashes("first", "second") + + +def test_default_sidecars_keep_detector_metadata_separate(tmp_path: Path) -> None: + output = tmp_path / "tr_hash_v8_o2m_fp16.onnx" + source_metadata = tmp_path / "tr_hash_v8_o2m.json" + source_metadata.write_text('{"branch": "o2m", "image_size": 640}', encoding="utf-8") + + detector_sidecar = default_detector_metadata_output(output) + copy_detector_metadata(source_metadata, detector_sidecar) + + assert detector_sidecar == tmp_path / "tr_hash_v8_o2m_fp16.json" + assert default_quantization_sidecar(output) == ( + tmp_path / "tr_hash_v8_o2m_fp16.quantization.json" + ) + assert json.loads(detector_sidecar.read_text(encoding="utf-8"))["branch"] == "o2m" + + +def test_int8_quantization_passes_claimed_settings_to_ort( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: dict[str, object] = {} + + class QuantFormat: + QDQ = "QDQ" + + class CalibrationMethod: + MinMax = "MinMax" + Entropy = "Entropy" + Percentile = "Percentile" + + class QuantType: + QInt8 = "QInt8" + QUInt8 = "QUInt8" + + def quantize_static(*args: object, **kwargs: object) -> None: + calls["args"] = args + calls["kwargs"] = kwargs + Path(args[1]).write_bytes(b"int8") + + fake_quantization = types.SimpleNamespace( + CalibrationMethod=CalibrationMethod, + QuantFormat=QuantFormat, + QuantType=QuantType, + quantize_static=quantize_static, + ) + monkeypatch.setitem(sys.modules, "onnxruntime.quantization", fake_quantization) + monkeypatch.setattr( + quantize_onnx, + "_calibration_paths", + lambda _manifest: [tmp_path / "a.jpg", tmp_path / "b.jpg"], + ) + + class Reader: + def __init__(self, **kwargs: object) -> None: + calls["reader_kwargs"] = kwargs + + monkeypatch.setattr(quantize_onnx, "_CalibrationReader", Reader) + + manifest = { + "images": ["a.jpg", "b.jpg"], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + }, + } + + quantize_int8( + tmp_path / "fp32.onnx", + tmp_path / "int8.onnx", + metadata_path=tmp_path / "metadata.json", + calibration_manifest=manifest, + ) + + assert calls["reader_kwargs"] == { + "image_paths": [tmp_path / "a.jpg", tmp_path / "b.jpg"], + "metadata_path": tmp_path / "metadata.json", + "batch_size": 2, + } + assert calls["kwargs"]["per_channel"] is True + assert calls["kwargs"]["activation_type"] == "QUInt8" + assert calls["kwargs"]["weight_type"] == "QInt8" + assert calls["kwargs"]["extra_options"] == { + "ActivationSymmetric": False, + "WeightSymmetric": True, + } + + +def test_int8_sidecar_settings_only_include_applied_contract( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(quantize_onnx, "quantize_int8", lambda *_, **__: None) + manifest = { + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + "num_threads": 1, + } + } + + settings = quantize_onnx.quantize_once( + fp32_model=tmp_path / "fp32.onnx", + metadata_path=tmp_path / "metadata.json", + precision="int8", + output_model=tmp_path / "int8.onnx", + calibration_manifest=manifest, + keep_fp32_op_types=(), + disable_shape_infer=False, + ) + + assert settings == { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 2, + } diff --git a/tests/test_onnx_quantization_config.py b/tests/test_onnx_quantization_config.py index 2bcbb9a..419df81 100644 --- a/tests/test_onnx_quantization_config.py +++ b/tests/test_onnx_quantization_config.py @@ -61,8 +61,7 @@ def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } """ @@ -73,7 +72,7 @@ def test_calibration_manifest_pins_int8_settings(tmp_path: Path) -> None: manifest = load_calibration_manifest(path) assert manifest["quantization"]["calibration_method"] == "minmax" - assert manifest["quantization"]["num_threads"] == 1 + assert manifest["quantization"]["batch_size"] == 8 def test_calibration_manifest_rejects_missing_quantization_setting( @@ -102,8 +101,7 @@ def test_calibration_manifest_rejects_missing_quantization_setting( "symmetric_activations": false, "symmetric_weights": true, "activation_type": "quint8", - "weight_type": "qint8", - "batch_size": 8 + "weight_type": "qint8" } } """ @@ -111,7 +109,7 @@ def test_calibration_manifest_rejects_missing_quantization_setting( encoding="utf-8", ) - with pytest.raises(ValueError, match="num_threads"): + with pytest.raises(ValueError, match="batch_size"): load_calibration_manifest(path) @@ -140,8 +138,7 @@ def test_calibration_manifest_rejects_placeholder_dataset_hash(tmp_path: Path) - "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } """ @@ -172,8 +169,7 @@ def test_calibration_manifest_requires_pinned_image_identity(tmp_path: Path) -> "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } """ @@ -205,8 +201,7 @@ def test_calibration_manifest_requires_calibration_image_paths(tmp_path: Path) - "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8, - "num_threads": 1 + "batch_size": 8 } } """ diff --git a/tests/test_onnx_release.py b/tests/test_onnx_release.py index ca60084..b02518b 100644 --- a/tests/test_onnx_release.py +++ b/tests/test_onnx_release.py @@ -100,6 +100,24 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: assert {spec.branch for spec in config.branches} == {"o2m", "nms-free"} assert config.opset == 17 assert set(config.toolchain) == {"torch", "onnx", "onnxruntime"} + assert config.quantization is not None + assert config.quantization.enabled_precisions == ("fp16", "int8") + assert config.quantization.calibration_manifest == Path( + "configs/vision_v8_quantization_calibration.json" + ) + assert config.quantization.thresholds == Path( + "configs/vision_v8_quantization_thresholds.json" + ) + assert config.quantization.accuracy_report == Path( + "artifacts/vision_v8_quantized_eval/accuracy.json" + ) + assert config.quantization.accuracy_markdown == Path( + "artifacts/vision_v8_quantized_eval/accuracy.md" + ) + assert config.quantization.provider_gates == ( + ("CUDAExecutionProvider", "fp16"), + ("CPUExecutionProvider", "int8"), + ) # Five seeds underestimate the observed maxima (see the validation report), # so a release gate has to be deeper than the development default. assert config.parity_num_tests >= 20 @@ -121,6 +139,21 @@ def test_an_unpinned_toolchain_is_rejected() -> None: config_from_mapping(config_mapping(toolchain={})) +def test_unsupported_quantized_precision_is_rejected() -> None: + with pytest.raises(ReleaseError, match="unsupported quantized precisions"): + config_from_mapping( + config_mapping( + quantization={ + "enabled_precisions": ["fp8"], + "calibration_manifest": "calibration.json", + "thresholds": "thresholds.json", + "accuracy_report": "accuracy.json", + "accuracy_markdown": "accuracy.md", + } + ) + ) + + def test_toolchain_drift_is_reported_per_package() -> None: pinned = {"torch": "2.13.0", "onnx": "1.21.0"} @@ -141,10 +174,14 @@ def test_output_contract_derives_the_prediction_width_from_the_sidecar() -> None assert contract["dtype"] == "float32" -def test_manifest_carries_every_field_the_release_must_document(tmp_path: Path) -> None: +def test_manifest_carries_every_field_the_release_must_document( + tmp_path: Path, +) -> None: manifest = written_release(tmp_path) - assert manifest["checkpoint_revision"] == "f3b3e659612e543ca9ff91892c0662d38dc1a1d6" + assert manifest["checkpoint_revision"] == ( + "f3b3e659612e543ca9ff91892c0662d38dc1a1d6" + ) assert manifest["framework_commit"] == "0" * 40 assert manifest["opset"] == 17 assert manifest["toolchain"]["torch"] == "2.13.0" @@ -185,7 +222,9 @@ def test_verification_catches_a_truncated_or_missing_artifact(tmp_path: Path) -> assert "missing" in problems -def test_verification_binds_the_manifest_to_the_publishing_commit(tmp_path: Path) -> None: +def test_verification_binds_the_manifest_to_the_publishing_commit( + tmp_path: Path, +) -> None: manifest = written_release(tmp_path) assert verify_manifest(manifest, tmp_path, expect_commit="0" * 40) == [] From 0c63ed4c7cd0741a05f1c3ff530875635a264dcd Mon Sep 17 00:00:00 2001 From: Illiyin Date: Sat, 5 Sep 2026 17:25:29 +0500 Subject: [PATCH 3/5] Fix Vision v8 quantized release gating --- .github/workflows/onnx-release.yml | 24 +- .github/workflows/vision-v8-coco-accuracy.yml | 70 ++++- complexity/deploy/onnx_detector/pipeline.py | 12 +- ..._v8_quantization_calibration.example.json} | 2 +- .../vision_v8_quantization_thresholds.json | 79 ++--- docs/onnx/release.json | 7 +- .../2026-08-31-vision-v8-onnx-quantization.md | 8 +- docs/vision-v8-onnx-quantization.md | 45 ++- scripts/benchmark_onnx_artifacts.py | 30 +- scripts/build_onnx_release.py | 249 ++++++++++++---- scripts/check_onnx_quantized_artifacts.py | 211 ++++++++++---- scripts/evaluate_onnx_coco.py | 9 + scripts/evaluate_tr_hash_coco.py | 50 ++-- scripts/merge_vision_v8_coco_reports.py | 38 ++- tests/test_coco_release_evaluation.py | 57 +++- tests/test_onnx_quantization_accuracy_gate.py | 269 ++++++++++++++++-- tests/test_onnx_quantization_benchmark.py | 71 ++++- tests/test_onnx_release.py | 67 ++++- 18 files changed, 1056 insertions(+), 242 deletions(-) rename configs/{vision_v8_quantization_calibration.json => vision_v8_quantization_calibration.example.json} (96%) diff --git a/.github/workflows/onnx-release.yml b/.github/workflows/onnx-release.yml index 1c92312..3e52613 100644 --- a/.github/workflows/onnx-release.yml +++ b/.github/workflows/onnx-release.yml @@ -12,11 +12,16 @@ on: required: false default: true type: boolean + evidence_run_id: + description: "Workflow run ID containing vision-v8-quantized-release-inputs" + required: false + type: string push: tags: - "onnx-v8-*" permissions: + actions: read contents: write concurrency: @@ -55,10 +60,25 @@ jobs: --index-url https://download.pytorch.org/whl/cpu python -m pip install \ "onnx==${{ steps.pins.outputs.onnx }}" \ - "onnxruntime==${{ steps.pins.outputs.onnxruntime }}" + "onnxruntime==${{ steps.pins.outputs.onnxruntime }}" \ + psutil + + - name: Download quantized release evidence + env: + GH_TOKEN: ${{ github.token }} + EVIDENCE_RUN_ID: ${{ github.event.inputs.evidence_run_id || vars.ONNX_RELEASE_EVIDENCE_RUN_ID }} + run: | + test -n "$EVIDENCE_RUN_ID" + mkdir -p artifacts/vision_v8_quantized_eval + gh run download "$EVIDENCE_RUN_ID" \ + --name vision-v8-quantized-release-inputs \ + --dir artifacts/vision_v8_quantized_eval + test -f artifacts/vision_v8_quantized_eval/calibration.json + test -f artifacts/vision_v8_quantized_eval/accuracy.json + test -f artifacts/vision_v8_quantized_eval/accuracy.md - name: Test the release tooling - run: pytest -q tests/test_onnx_release.py + run: pytest -q tests/test_onnx_release.py tests/test_onnx_quantization_benchmark.py - name: Build, gate and verify the release env: diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 6adbb62..e2ca47c 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -47,6 +47,33 @@ on: onnx_nms_free_metadata: description: "Local path to the NMS-free Vision v8 ONNX metadata sidecar on the runner" required: false + onnx_o2m_fp16_model: + description: "Local path to the FP16 O2M Vision v8 ONNX model on the runner" + required: false + onnx_o2m_fp16_metadata: + description: "Local path to the FP16 O2M Vision v8 ONNX metadata sidecar" + required: false + onnx_nms_free_fp16_model: + description: "Local path to the FP16 NMS-free Vision v8 ONNX model" + required: false + onnx_nms_free_fp16_metadata: + description: "Local path to the FP16 NMS-free Vision v8 ONNX metadata sidecar" + required: false + onnx_o2m_int8_model: + description: "Local path to the INT8 O2M Vision v8 ONNX model on the runner" + required: false + onnx_o2m_int8_metadata: + description: "Local path to the INT8 O2M Vision v8 ONNX metadata sidecar" + required: false + onnx_nms_free_int8_model: + description: "Local path to the INT8 NMS-free Vision v8 ONNX model" + required: false + onnx_nms_free_int8_metadata: + description: "Local path to the INT8 NMS-free Vision v8 ONNX metadata sidecar" + required: false + calibration_manifest: + description: "Local path to the quantization calibration manifest" + required: false annotations: description: "Local path to instances_val2017.json on the runner" required: true @@ -147,11 +174,11 @@ jobs: test -n "${{ inputs.onnx_o2m_metadata }}" test -n "${{ inputs.onnx_nms_free_model }}" test -n "${{ inputs.onnx_nms_free_metadata }}" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b fi - name: Gate full COCO report @@ -160,6 +187,33 @@ jobs: artifacts/vision_v8_coco_eval/run_a/evaluation.json --repeat-report artifacts/vision_v8_coco_eval/run_b/evaluation.json --config configs/vision_v8_coco_accuracy_gate.json + - name: Build quantized release evidence artifact + if: inputs.backend == 'onnx' && inputs.calibration_manifest != '' + run: | + test -n "${{ inputs.onnx_o2m_fp16_model }}" + test -n "${{ inputs.onnx_o2m_fp16_metadata }}" + test -n "${{ inputs.onnx_nms_free_fp16_model }}" + test -n "${{ inputs.onnx_nms_free_fp16_metadata }}" + test -n "${{ inputs.onnx_o2m_int8_model }}" + test -n "${{ inputs.onnx_o2m_int8_metadata }}" + test -n "${{ inputs.onnx_nms_free_int8_model }}" + test -n "${{ inputs.onnx_nms_free_int8_metadata }}" + mkdir -p artifacts/vision_v8_quantized_eval + cp "${{ inputs.calibration_manifest }}" artifacts/vision_v8_quantized_eval/calibration.json + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_fp16_model }}" --metadata "${{ inputs.onnx_o2m_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_fp16 --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_fp16_model }}" --metadata "${{ inputs.onnx_nms_free_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_fp16 --branch nms-free --provider "${{ inputs.provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_int8_model }}" --metadata "${{ inputs.onnx_o2m_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_int8 --branch o2m-nms --provider "${{ inputs.provider }}" --precision int8 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_int8_model }}" --metadata "${{ inputs.onnx_nms_free_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_int8 --branch nms-free --provider "${{ inputs.provider }}" --precision int8 --seed 0 + python scripts/merge_vision_v8_coco_reports.py \ + artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json \ + artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_int8/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_int8/evaluation.json \ + --output artifacts/vision_v8_quantized_eval + cp artifacts/vision_v8_quantized_eval/evaluation.json artifacts/vision_v8_quantized_eval/accuracy.json + cp artifacts/vision_v8_quantized_eval/evaluation.md artifacts/vision_v8_quantized_eval/accuracy.md - uses: actions/upload-artifact@v4 with: name: vision-v8-coco-accuracy-report @@ -168,6 +222,14 @@ jobs: artifacts/vision_v8_coco_eval/run_a/evaluation.md artifacts/vision_v8_coco_eval/run_b/evaluation.json artifacts/vision_v8_coco_eval/run_b/evaluation.md + - uses: actions/upload-artifact@v4 + if: inputs.backend == 'onnx' && inputs.calibration_manifest != '' + with: + name: vision-v8-quantized-release-inputs + path: | + artifacts/vision_v8_quantized_eval/calibration.json + artifacts/vision_v8_quantized_eval/accuracy.json + artifacts/vision_v8_quantized_eval/accuracy.md - name: Upload gated reports to GitHub Release if: inputs.release_tag != '' env: diff --git a/complexity/deploy/onnx_detector/pipeline.py b/complexity/deploy/onnx_detector/pipeline.py index 50779d8..037e531 100644 --- a/complexity/deploy/onnx_detector/pipeline.py +++ b/complexity/deploy/onnx_detector/pipeline.py @@ -84,13 +84,11 @@ def postprocess_single_image( scores: np.ndarray, classes: np.ndarray, ) -> tuple[np.ndarray, np.ndarray, np.ndarray]: - filtered_boxes, filtered_scores, filtered_classes, _ = ( - postprocess.filter_by_confidence( - boxes, - scores, - classes, - self.metadata.confidence_threshold, - ) + filtered_boxes, filtered_scores, filtered_classes, _ = postprocess.filter_by_confidence( + boxes, + scores, + classes, + self.metadata.confidence_threshold, ) if self.metadata.branch == "o2m": keep = postprocess.class_aware_nms( diff --git a/configs/vision_v8_quantization_calibration.json b/configs/vision_v8_quantization_calibration.example.json similarity index 96% rename from configs/vision_v8_quantization_calibration.json rename to configs/vision_v8_quantization_calibration.example.json index 5caf831..e6fb9c0 100644 --- a/configs/vision_v8_quantization_calibration.json +++ b/configs/vision_v8_quantization_calibration.example.json @@ -14,6 +14,6 @@ "symmetric_weights": true, "activation_type": "quint8", "weight_type": "qint8", - "batch_size": 8 + "batch_size": 1 } } diff --git a/configs/vision_v8_quantization_thresholds.json b/configs/vision_v8_quantization_thresholds.json index 659e8b3..992f0fa 100644 --- a/configs/vision_v8_quantization_thresholds.json +++ b/configs/vision_v8_quantization_thresholds.json @@ -1,43 +1,44 @@ { - "schema_version": 1, - "release_policy": { - "required_precisions": ["fp32", "fp16", "int8"], - "partial_release": "block", - "optional_provider_precisions": [] - }, - "precisions": { - "fp16": { - "max_raw_logit_abs_error": 0.01, - "max_decoded_box_px_error": 1.0, - "max_score_abs_error": 0.01, - "max_map50_95_drop": 0.005, - "max_map50_drop": 0.01, - "unexpected_fp32_nodes": "fail" + "schema_version": 1, + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "require_artifact_bindings": true, + "partial_release": "block", + "optional_provider_precisions": [] }, - "int8": { - "max_raw_logit_abs_error": 0.12, - "max_decoded_box_px_error": 4.0, - "max_score_abs_error": 0.05, - "max_map50_95_drop": 0.02, - "max_map50_drop": 0.03, - "unexpected_fp32_nodes": "allow" + "precisions": { + "fp16": { + "max_raw_logit_abs_error": 0.01, + "max_decoded_box_px_error": 1.0, + "max_score_abs_error": 0.01, + "max_map50_95_drop": 0.005, + "max_map50_drop": 0.01, + "unexpected_fp32_nodes": "fail" + }, + "int8": { + "max_raw_logit_abs_error": 0.12, + "max_decoded_box_px_error": 4.0, + "max_score_abs_error": 0.05, + "max_map50_95_drop": 0.02, + "max_map50_drop": 0.03, + "unexpected_fp32_nodes": "allow" + } + }, + "providers": { + "CPUExecutionProvider": ["fp32", "fp16", "int8"], + "CUDAExecutionProvider": ["fp32", "fp16"], + "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] + }, + "benchmark": { + "warmup_iterations": 25, + "measured_iterations": 100, + "report": [ + "median_ms", + "mean_ms", + "stddev_ms", + "p95_ms", + "throughput_images_per_second", + "peak_memory_mb" + ] } - }, - "providers": { - "CPUExecutionProvider": ["fp32", "int8"], - "CUDAExecutionProvider": ["fp32", "fp16"], - "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] - }, - "benchmark": { - "warmup_iterations": 25, - "measured_iterations": 100, - "report": [ - "median_ms", - "mean_ms", - "stddev_ms", - "p95_ms", - "throughput_images_per_second", - "peak_memory_mb" - ] - } } diff --git a/docs/onnx/release.json b/docs/onnx/release.json index ab080ba..f3ca266 100644 --- a/docs/onnx/release.json +++ b/docs/onnx/release.json @@ -5,20 +5,21 @@ "parity_num_tests": 50, "quantization": { "enabled_precisions": ["fp16", "int8"], - "calibration_manifest": "configs/vision_v8_quantization_calibration.json", + "calibration_manifest": "artifacts/vision_v8_quantized_eval/calibration.json", "thresholds": "configs/vision_v8_quantization_thresholds.json", "accuracy_report": "artifacts/vision_v8_quantized_eval/accuracy.json", "accuracy_markdown": "artifacts/vision_v8_quantized_eval/accuracy.md", "fp32_op_allowlist": [], "provider_gates": [ - {"provider": "CUDAExecutionProvider", "precision": "fp16"}, + {"provider": "CPUExecutionProvider", "precision": "fp32"}, + {"provider": "CPUExecutionProvider", "precision": "fp16"}, {"provider": "CPUExecutionProvider", "precision": "int8"} ] }, "toolchain": { "torch": "2.13.0", "onnx": "1.21.0", - "onnxruntime": "1.24.4" + "onnxruntime": "1.23.2" }, "branches": [ { diff --git a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md index e0aaae1..f8fe0f3 100644 --- a/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md +++ b/docs/superpowers/plans/2026-08-31-vision-v8-onnx-quantization.md @@ -27,7 +27,7 @@ ## File Structure - Create `configs/vision_v8_quantization_thresholds.json` for FP16/INT8 raw-logit, decoded-output, COCO, performance, dtype, and provider thresholds. -- Create `configs/vision_v8_quantization_calibration.json` for pinned calibration input metadata and quantization settings. +- Create `configs/vision_v8_quantization_calibration.example.json` for pinned calibration input metadata and quantization settings. - Create `scripts/quantize_onnx.py` to generate FP16 and INT8 ONNX files plus JSON sidecars. - Create `scripts/check_onnx_quantized_artifacts.py` to validate metadata, checksums, node dtypes, provider support, calibration/eval disjointness, parity reports, COCO reports, and performance reports. - Create `scripts/benchmark_onnx_artifacts.py` to measure artifact latency, throughput, and peak memory with stable warm-up and measured iteration counts. @@ -43,7 +43,7 @@ **Files:** - Create: `configs/vision_v8_quantization_thresholds.json` -- Create: `configs/vision_v8_quantization_calibration.json` +- Create: `configs/vision_v8_quantization_calibration.example.json` - Test: `tests/test_onnx_quantization_config.py` **Interfaces:** @@ -146,7 +146,7 @@ Create `configs/vision_v8_quantization_thresholds.json`: } ``` -Create `configs/vision_v8_quantization_calibration.json`: +Create `configs/vision_v8_quantization_calibration.example.json`: ```json { @@ -226,7 +226,7 @@ Run: `python -m pytest -q tests/test_onnx_quantization_config.py` Commit: ```bash -git add configs/vision_v8_quantization_thresholds.json configs/vision_v8_quantization_calibration.json scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py +git add configs/vision_v8_quantization_thresholds.json configs/vision_v8_quantization_calibration.example.json scripts/check_onnx_quantized_artifacts.py tests/test_onnx_quantization_config.py git commit -m "Add Vision v8 quantization gate config" ``` diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md index 9a4b496..9530ecb 100644 --- a/docs/vision-v8-onnx-quantization.md +++ b/docs/vision-v8-onnx-quantization.md @@ -24,12 +24,16 @@ used the requested provider and that unexpected graph nodes did not remain FP32. ## Calibration Contract INT8 calibration is pinned by -`configs/vision_v8_quantization_calibration.json`. The manifest records the +the release evidence artifact. `configs/vision_v8_quantization_calibration.example.json` +documents the expected shape. The manifest records the dataset, image-ID manifest hash, annotation hash, calibration method, per-channel/per-tensor mode, symmetric/asymmetric choices, activation type, weight type, and batch size. The ORT symmetry options and batch size are passed into quantization directly; unsupported settings are not recorded in release metadata. +The default release export uses a fixed batch of 1, so the default calibration +manifest must also use `batch_size: 1`. If the export is changed to dynamic +batching, the calibration batch size can be raised in the same release contract. Placeholder hashes are rejected by `load_calibration_manifest`; before a release run, replace them with the approved calibration subset and real SHA-256 values. @@ -64,7 +68,7 @@ python scripts/quantize_onnx.py \ --fp32-model artifacts/onnx/tr_hash_v8_o2m.onnx \ --metadata artifacts/onnx/tr_hash_v8_o2m.json \ --precision int8 \ - --calibration-manifest configs/vision_v8_quantization_calibration.json \ + --calibration-manifest artifacts/vision_v8_quantized_eval/calibration.json \ --output artifacts/onnx/tr_hash_v8_o2m_int8.onnx \ --repeat-output artifacts/onnx/tr_hash_v8_o2m_int8_repeat.onnx \ --require-identical-hash \ @@ -88,7 +92,28 @@ Before publishing, merge the FP32 reference and quantized branch reports into `artifacts/vision_v8_quantized_eval/accuracy.json` and `artifacts/vision_v8_quantized_eval/accuracy.md`. The release builder validates that JSON against `configs/vision_v8_quantization_thresholds.json`; missing or -regressed reports block publication. +regressed reports block publication. The report must include the evaluated image +IDs so the release gate can prove the calibration set and evaluation set are +actually disjoint. + +The merge report is precision-nested by branch. For example, each branch must +contain `fp32`, `fp16`, and `int8` entries so the gate can compare every +quantized candidate against the FP32 reference: + +```bash +python scripts/merge_vision_v8_coco_reports.py \ + artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json \ + artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_fp16/evaluation.json \ + artifacts/vision_v8_quantized_eval/o2m_int8/evaluation.json \ + artifacts/vision_v8_quantized_eval/nms_free_int8/evaluation.json \ + --output artifacts/vision_v8_quantized_eval +cp artifacts/vision_v8_quantized_eval/evaluation.json \ + artifacts/vision_v8_quantized_eval/accuracy.json +cp artifacts/vision_v8_quantized_eval/evaluation.md \ + artifacts/vision_v8_quantized_eval/accuracy.md +``` Benchmark an artifact: @@ -102,6 +127,20 @@ python scripts/benchmark_onnx_artifacts.py \ --measured-iterations 100 ``` +The release build also generates `quantized_benchmarks.json` and +`quantized_benchmarks.md` for every FP32, FP16, and INT8 branch artifact using +the benchmark methodology from the checked-in threshold config. + +The GitHub release workflow downloads a prior Actions artifact named +`vision-v8-quantized-release-inputs`. The manual Vision v8 COCO accuracy +workflow produces that artifact when `backend=onnx`, `calibration_manifest` is +set, and FP32/FP16/INT8 model plus metadata paths are provided for both +branches. The artifact must provide `calibration.json`, `accuracy.json`, and +`accuracy.md` under `artifacts/vision_v8_quantized_eval/` before +`build_onnx_release.py` runs. Pass the source run as workflow input +`evidence_run_id`, or set repository variable `ONNX_RELEASE_EVIDENCE_RUN_ID` +for tag-triggered releases. + ## Release Policy The quantized release workflow blocks the release when any required artifact diff --git a/scripts/benchmark_onnx_artifacts.py b/scripts/benchmark_onnx_artifacts.py index 38033b3..445ef23 100644 --- a/scripts/benchmark_onnx_artifacts.py +++ b/scripts/benchmark_onnx_artifacts.py @@ -63,7 +63,7 @@ def summarize_latency_ms(values: Sequence[float]) -> dict[str, float]: } -def _peak_memory_mb() -> float | None: +def _current_memory_mb() -> float | None: try: import psutil except ImportError: @@ -77,18 +77,34 @@ def _benchmark_session( batch_size: int, warmup_iterations: int, measured_iterations: int, -) -> list[float]: - input_shape = (batch_size, 3, pipeline.metadata.image_size, pipeline.metadata.image_size) +) -> tuple[list[float], float | None]: + input_shape = ( + batch_size, + 3, + pipeline.metadata.image_size, + pipeline.metadata.image_size, + ) dummy = np.zeros(input_shape, dtype=np.float32) + peak_memory_mb = _current_memory_mb() for _ in range(warmup_iterations): pipeline.session.run(dummy) + peak_memory_mb = _max_optional(peak_memory_mb, _current_memory_mb()) latencies: list[float] = [] for _ in range(measured_iterations): started = time.perf_counter() pipeline.session.run(dummy) latencies.append((time.perf_counter() - started) * 1000.0) - return latencies + peak_memory_mb = _max_optional(peak_memory_mb, _current_memory_mb()) + return latencies, peak_memory_mb + + +def _max_optional(first: float | None, second: float | None) -> float | None: + if first is None: + return second + if second is None: + return first + return max(first, second) def benchmark_onnx_artifact( @@ -114,7 +130,7 @@ def benchmark_onnx_artifact( intra_op_num_threads=ort_intra_op_threads, inter_op_num_threads=ort_inter_op_threads, ) - latencies = _benchmark_session( + latencies, peak_memory_mb = _benchmark_session( pipeline, batch_size=batch_size, warmup_iterations=warmup_iterations, @@ -136,8 +152,8 @@ def benchmark_onnx_artifact( "measured_iterations": measured_iterations, "latency": summary, "throughput_images_per_second": throughput, - "peak_memory_mb": _peak_memory_mb(), - "benchmark_methodology": "fixed warmup, fixed measured iterations, latency distribution", + "peak_memory_mb": peak_memory_mb, + "benchmark_methodology": ("fixed warmup, fixed measured iterations, latency distribution"), "environment": { "python": sys.version.split()[0], "os": platform.platform(), diff --git a/scripts/build_onnx_release.py b/scripts/build_onnx_release.py index a3bace1..39e5acf 100644 --- a/scripts/build_onnx_release.py +++ b/scripts/build_onnx_release.py @@ -21,10 +21,15 @@ import json import os import subprocess +import sys from dataclasses import dataclass from pathlib import Path from typing import Any, Mapping, Sequence +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + os.environ.setdefault("COMPLEXITY_DISABLE_KERNELS", "1") MANIFEST_NAME = "manifest.json" @@ -92,8 +97,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: BranchSpec( branch=str(entry["branch"]), checkpoint_subdir=( - None if entry.get("checkpoint_subdir") is None - else str(entry["checkpoint_subdir"]) + None if entry.get("checkpoint_subdir") is None else str(entry["checkpoint_subdir"]) ), stem=str(entry["stem"]), post_processing=str(entry["post_processing"]), @@ -120,8 +124,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: if isinstance(values.get("quantization"), Mapping): raw_quantization = values["quantization"] enabled_precisions = tuple( - str(precision) - for precision in raw_quantization.get("enabled_precisions", ()) + str(precision) for precision in raw_quantization.get("enabled_precisions", ()) ) unsupported = sorted(set(enabled_precisions) - {"fp16", "int8"}) if unsupported: @@ -137,8 +140,7 @@ def config_from_mapping(values: Mapping[str, Any]) -> ReleaseConfig: accuracy_report=Path(str(raw_quantization["accuracy_report"])), accuracy_markdown=Path(str(raw_quantization["accuracy_markdown"])), fp32_op_allowlist=tuple( - str(op_type) - for op_type in raw_quantization.get("fp32_op_allowlist", ()) + str(op_type) for op_type in raw_quantization.get("fp32_op_allowlist", ()) ), provider_gates=provider_gates, ) @@ -229,6 +231,10 @@ def provider_chain(provider: str) -> tuple[str, ...]: return (provider,) +def coco_report_branch(branch: str) -> str: + return "o2m-nms" if branch == "o2m" else branch + + def output_contract(sidecar: Mapping[str, Any]) -> dict[str, Any]: """Describe the ONNX input/output contract from an export sidecar.""" @@ -285,9 +291,7 @@ def verify_manifest( if expect_commit is not None: recorded = str(manifest.get("framework_commit", "")) if recorded != expect_commit: - problems.append( - f"framework_commit {recorded or 'missing'}, expected {expect_commit}" - ) + problems.append(f"framework_commit {recorded or 'missing'}, expected {expect_commit}") for artifact in manifest["artifacts"]: path = Path(directory) / str(artifact["name"]) if not path.is_file(): @@ -296,14 +300,12 @@ def verify_manifest( actual_size = path.stat().st_size if actual_size != int(artifact["size_bytes"]): problems.append( - f"{artifact['name']}: size {actual_size}, " - f"manifest {artifact['size_bytes']}" + f"{artifact['name']}: size {actual_size}, manifest {artifact['size_bytes']}" ) actual_digest = sha256_file(path) if actual_digest != str(artifact["sha256"]): problems.append( - f"{artifact['name']}: sha256 {actual_digest}, " - f"manifest {artifact['sha256']}" + f"{artifact['name']}: sha256 {actual_digest}, manifest {artifact['sha256']}" ) return problems @@ -362,6 +364,34 @@ def render_release_notes(manifest: Mapping[str, Any]) -> str: return "\n".join(lines) +def render_benchmark_report(report: Mapping[str, Any]) -> str: + """Render benchmark evidence into a release-friendly Markdown table.""" + + lines = [ + "# Vision v8 ONNX Quantization Benchmarks", + "", + "| Branch | Precision | Provider | Median ms | P95 ms | Throughput | Peak MB |", + "|---|---|---|---:|---:|---:|---:|", + ] + for branch, branch_report in _mapping(report.get("branches")).items(): + for precision, precision_report in _mapping(branch_report).items(): + data = _mapping(precision_report) + latency = _mapping(data.get("latency")) + lines.append( + f"| {branch} | {precision} | {data.get('actual_provider', '')} | " + f"{float(latency.get('median_ms', 0.0)):.3f} | " + f"{float(latency.get('p95_ms', 0.0)):.3f} | " + f"{float(data.get('throughput_images_per_second', 0.0)):.3f} | " + f"{float(data.get('peak_memory_mb', 0.0)):.3f} |" + ) + lines.append("") + return "\n".join(lines) + + +def _mapping(value: object) -> Mapping[str, Any]: + return value if isinstance(value, Mapping) else {} + + def build_release( config: ReleaseConfig, output_dir: Path, @@ -372,12 +402,16 @@ def build_release( from huggingface_hub import snapshot_download + from scripts.benchmark_onnx_artifacts import benchmark_onnx_artifact from scripts.check_onnx_parity import check_parity from scripts.check_onnx_quantized_artifacts import ( + assert_disjoint_image_ids, check_provider_precision_supported, check_quantized_accuracy_report, + check_quantized_benchmark_report, check_quantized_parity_report, check_unexpected_fp32_nodes, + evaluation_image_ids_from_report, inspect_onnx_node_dtypes, load_calibration_manifest, load_quantization_thresholds, @@ -398,9 +432,8 @@ def build_release( installed = installed_toolchain() mismatches = toolchain_mismatches(config.toolchain, installed) if mismatches: - message = ( - "toolchain does not match the pinned release toolchain:\n " - + "\n ".join(mismatches) + message = "toolchain does not match the pinned release toolchain:\n " + "\n ".join( + mismatches ) if not allow_toolchain_drift: raise ReleaseError( @@ -416,11 +449,13 @@ def build_release( quantization_thresholds: Mapping[str, Any] | None = None calibration_manifest: Mapping[str, Any] | None = None + accuracy_report: Mapping[str, Any] | None = None provider_by_precision: dict[str, str] = {} + benchmark_report: dict[str, Any] | None = None + benchmark_settings: Mapping[str, Any] = {} + expected_accuracy_artifacts: dict[str, dict[str, dict[str, str]]] = {} if config.quantization is not None: - quantization_thresholds = load_quantization_thresholds( - config.quantization.thresholds - ) + quantization_thresholds = load_quantization_thresholds(config.quantization.thresholds) for provider, precision in config.quantization.provider_gates: check_provider_precision_supported( provider, @@ -432,6 +467,12 @@ def build_release( calibration_manifest = load_calibration_manifest( config.quantization.calibration_manifest ) + batch_size = int(calibration_manifest["quantization"]["batch_size"]) + if batch_size != 1: + raise ReleaseError( + "default ONNX release exports fixed batch size 1, " + f"but calibration batch_size is {batch_size}" + ) if config.quantization.enabled_precisions and calibration_manifest is None: raise ReleaseError("quantized release parity requires calibration images") parity_image = Path(str(calibration_manifest["images"][0])) @@ -448,39 +489,12 @@ def build_release( accuracy_report = json.loads( config.quantization.accuracy_report.read_text(encoding="utf-8") ) - accuracy_failures = check_quantized_accuracy_report( - accuracy_report, - quantization_thresholds, - ) - if accuracy_failures: - raise ReleaseError( - "quantized COCO accuracy gate failed:\n " - + "\n ".join(accuracy_failures) - ) - accuracy_report_path = destination / "quantized_accuracy.json" - accuracy_markdown_path = destination / "quantized_accuracy.md" - accuracy_report_path.write_text( - json.dumps(accuracy_report, indent=2) + "\n", - encoding="utf-8", - ) - accuracy_markdown_path.write_text( - config.quantization.accuracy_markdown.read_text(encoding="utf-8"), - encoding="utf-8", - ) - artifacts.append( - artifact_entry( - accuracy_report_path, - kind="accuracy_report", - precision="mixed", - ) - ) - artifacts.append( - artifact_entry( - accuracy_markdown_path, - kind="accuracy_report_markdown", - precision="mixed", - ) + assert_disjoint_image_ids( + {int(image_id) for image_id in calibration_manifest["image_ids"]}, + evaluation_image_ids_from_report(accuracy_report), ) + benchmark_report = {"schema_version": 1, "branches": {}} + benchmark_settings = _mapping(quantization_thresholds.get("benchmark")) print(f"Downloading {config.checkpoint_repo}@{config.checkpoint_revision}") checkpoint_root = Path( @@ -536,6 +550,27 @@ def build_release( precision="fp32", ) ) + if config.quantization is not None and quantization_thresholds is not None: + expected_accuracy_artifacts.setdefault(coco_report_branch(spec.branch), {})["fp32"] = { + "checkpoint_sha256": sha256_file(model_path), + "metadata_sha256": sha256_file(sidecar_path), + } + fp32_benchmark = benchmark_onnx_artifact( + model_path=model_path, + metadata_path=sidecar_path, + providers=provider_chain(provider_by_precision.get("fp32", "CPUExecutionProvider")), + batch_size=1, + warmup_iterations=int(benchmark_settings["warmup_iterations"]), + measured_iterations=int(benchmark_settings["measured_iterations"]), + ort_intra_op_threads=1, + ort_inter_op_threads=1, + ) + check_provider_precision_supported( + str(fp32_benchmark["actual_provider"]), + "fp32", + quantization_thresholds, + ) + benchmark_report["branches"].setdefault(spec.branch, {})["fp32"] = fp32_benchmark if config.quantization is not None and quantization_thresholds is not None: for precision in config.quantization.enabled_precisions: quantized_model_path = destination / f"{spec.stem}_{precision}.onnx" @@ -575,15 +610,12 @@ def build_release( if unexpected: raise ReleaseError( f"{quantized_model_path.name} retained unexpected " - "FP32 nodes: " - + ", ".join(unexpected) + "FP32 nodes: " + ", ".join(unexpected) ) quantized_metadata_path = quantized_model_path.with_suffix(".json") copy_detector_metadata(sidecar_path, quantized_metadata_path) - quantized_sidecar_path = default_quantization_sidecar( - quantized_model_path - ) + quantized_sidecar_path = default_quantization_sidecar(quantized_model_path) write_quantization_sidecar( quantized_sidecar_path, precision=precision, @@ -624,9 +656,7 @@ def build_release( f"{quantized_model_path.name} failed quantized parity:\n " + "\n ".join(parity_failures) ) - parity_report_path = ( - destination / f"{spec.stem}_{precision}_parity.json" - ) + parity_report_path = destination / f"{spec.stem}_{precision}_parity.json" parity_report_path.write_text( json.dumps(parity_report, indent=2) + "\n", encoding="utf-8", @@ -668,6 +698,107 @@ def build_release( precision=precision, ) ) + expected_accuracy_artifacts.setdefault( + coco_report_branch(spec.branch), + {}, + )[precision] = { + "checkpoint_sha256": sha256_file(quantized_model_path), + "metadata_sha256": sha256_file(quantized_metadata_path), + } + quantized_benchmark = benchmark_onnx_artifact( + model_path=quantized_model_path, + metadata_path=quantized_metadata_path, + providers=provider_chain( + provider_by_precision.get(precision, "CPUExecutionProvider") + ), + batch_size=1, + warmup_iterations=int(benchmark_settings["warmup_iterations"]), + measured_iterations=int(benchmark_settings["measured_iterations"]), + ort_intra_op_threads=1, + ort_inter_op_threads=1, + ) + check_provider_precision_supported( + str(quantized_benchmark["actual_provider"]), + precision, + quantization_thresholds, + ) + benchmark_report["branches"].setdefault(spec.branch, {})[precision] = ( + quantized_benchmark + ) + + if ( + config.quantization is not None + and quantization_thresholds is not None + and benchmark_report is not None + ): + benchmark_failures = check_quantized_benchmark_report( + benchmark_report, + quantization_thresholds, + required_branches=[spec.branch for spec in config.branches], + ) + if benchmark_failures: + raise ReleaseError( + "quantized benchmark gate failed:\n " + "\n ".join(benchmark_failures) + ) + assert accuracy_report is not None + accuracy_failures = check_quantized_accuracy_report( + accuracy_report, + quantization_thresholds, + required_branches=[coco_report_branch(spec.branch) for spec in config.branches], + expected_artifacts=expected_accuracy_artifacts, + ) + if accuracy_failures: + raise ReleaseError( + "quantized COCO accuracy gate failed:\n " + "\n ".join(accuracy_failures) + ) + accuracy_report_path = destination / "quantized_accuracy.json" + accuracy_markdown_path = destination / "quantized_accuracy.md" + accuracy_report_path.write_text( + json.dumps(accuracy_report, indent=2) + "\n", + encoding="utf-8", + ) + accuracy_markdown_path.write_text( + config.quantization.accuracy_markdown.read_text(encoding="utf-8"), + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + accuracy_report_path, + kind="accuracy_report", + precision="mixed", + ) + ) + artifacts.append( + artifact_entry( + accuracy_markdown_path, + kind="accuracy_report_markdown", + precision="mixed", + ) + ) + benchmark_json_path = destination / "quantized_benchmarks.json" + benchmark_markdown_path = destination / "quantized_benchmarks.md" + benchmark_json_path.write_text( + json.dumps(benchmark_report, indent=2) + "\n", + encoding="utf-8", + ) + benchmark_markdown_path.write_text( + render_benchmark_report(benchmark_report), + encoding="utf-8", + ) + artifacts.append( + artifact_entry( + benchmark_json_path, + kind="benchmark_report", + precision="mixed", + ) + ) + artifacts.append( + artifact_entry( + benchmark_markdown_path, + kind="benchmark_report_markdown", + precision="mixed", + ) + ) manifest = build_manifest( config, diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py index dce400f..c7563c2 100644 --- a/scripts/check_onnx_quantized_artifacts.py +++ b/scripts/check_onnx_quantized_artifacts.py @@ -25,11 +25,7 @@ def load_quantization_thresholds(path: Path) -> dict[str, Any]: if "release_policy" not in config: raise ValueError("threshold config missing release_policy") precisions = config.get("precisions") - if ( - not isinstance(precisions, dict) - or "fp16" not in precisions - or "int8" not in precisions - ): + if not isinstance(precisions, dict) or "fp16" not in precisions or "int8" not in precisions: raise ValueError("threshold config must define fp16 and int8 precisions") return config @@ -70,9 +66,7 @@ def load_calibration_manifest(path: Path) -> dict[str, Any]: actual_digest = image_id_manifest_sha256([int(image_id) for image_id in image_ids]) expected_digest = str(dataset["image_ids_sha256"]) if actual_digest != expected_digest: - raise ValueError( - "calibration manifest dataset.image_ids_sha256 does not match image_ids" - ) + raise ValueError("calibration manifest dataset.image_ids_sha256 does not match image_ids") images = manifest.get("images") if not isinstance(images, Sequence) or isinstance(images, (str, bytes)): @@ -103,6 +97,47 @@ def assert_disjoint_image_ids( raise ValueError(f"calibration/evaluation image ID overlap: {preview}") +def evaluation_image_ids_from_report(report: Mapping[str, Any]) -> set[int]: + """Extract the actual evaluated image IDs from an accuracy report.""" + + raw_ids = report.get("evaluation_image_ids") + if raw_ids is None: + raw_ids = _mapping(report.get("dataset")).get("image_ids") + if not isinstance(raw_ids, Sequence) or isinstance(raw_ids, (str, bytes)): + raise ValueError("accuracy report must include evaluation image IDs") + if not raw_ids: + raise ValueError("accuracy report evaluation image IDs must not be empty") + return {int(image_id) for image_id in raw_ids} + + +def check_accuracy_artifact_bindings( + report: Mapping[str, Any], + generated_artifacts: Mapping[str, Mapping[str, Mapping[str, str]]], +) -> list[str]: + """Verify accuracy evidence hashes match the artifacts being published.""" + + branches = _mapping(report.get("branches")) + failures: list[str] = [] + for branch, precision_hashes in generated_artifacts.items(): + branch_report = _mapping(branches.get(branch)) + if not branch_report: + failures.append(f"accuracy report missing branch {branch}") + continue + for precision, expected_hashes in precision_hashes.items(): + precision_report = _mapping(branch_report.get(precision)) + if not precision_report: + failures.append(f"accuracy report missing {branch} {precision}") + continue + for field, expected in expected_hashes.items(): + actual = precision_report.get(field) + if actual != expected: + failures.append( + f"{branch} {precision} {field} {actual or 'missing'} " + f"does not match generated {expected}" + ) + return failures + + def check_provider_precision_supported( provider: str, precision: str, @@ -112,16 +147,12 @@ def check_provider_precision_supported( providers = thresholds.get("providers", {}) if not isinstance(providers, Mapping) or provider not in providers: - raise ValueError( - f"{provider} is not configured in quantization release config" - ) + raise ValueError(f"{provider} is not configured in quantization release config") supported = providers[provider] if not isinstance(supported, Sequence) or isinstance(supported, (str, bytes)): raise ValueError(f"{provider} precision policy must be a sequence") if precision not in supported: - raise ValueError( - f"{provider} does not support {precision} in quantization release config" - ) + raise ValueError(f"{provider} does not support {precision} in quantization release config") def check_unexpected_fp32_nodes( @@ -157,9 +188,7 @@ def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: import onnx from onnx import TensorProto except ImportError as error: # pragma: no cover - dependency guard - raise RuntimeError( - "ONNX dtype inspection requires the onnx package" - ) from error + raise RuntimeError("ONNX dtype inspection requires the onnx package") from error model = onnx.load(str(model_path)) value_dtypes: dict[str, int] = {} @@ -183,13 +212,9 @@ def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: fp16_nodes = 0 int8_nodes = 0 for node in model.graph.node: - output_types = { - value_dtypes[name] for name in node.output if name in value_dtypes - } + output_types = {value_dtypes[name] for name in node.output if name in value_dtypes} if TensorProto.FLOAT in output_types: - fp32_nodes.append( - {"name": node.name or node.output[0], "op_type": node.op_type} - ) + fp32_nodes.append({"name": node.name or node.output[0], "op_type": node.op_type}) if TensorProto.FLOAT16 in output_types: fp16_nodes += 1 if TensorProto.INT8 in output_types or TensorProto.UINT8 in output_types: @@ -205,11 +230,25 @@ def inspect_onnx_node_dtypes(model_path: Path) -> dict[str, Any]: def check_quantized_accuracy_report( report: Mapping[str, Any], thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str] = (), + expected_artifacts: Mapping[str, Mapping[str, Mapping[str, str]]] | None = None, ) -> list[str]: """Compare a quantized candidate COCO report against its FP32 reference.""" if "branches" in report: - return _check_branch_accuracy_report(report, thresholds) + failures = _check_branch_accuracy_report( + report, + thresholds, + required_branches=required_branches, + ) + release_policy = _mapping(thresholds.get("release_policy")) + if release_policy.get("require_artifact_bindings") is True: + if expected_artifacts is None: + failures.append("generated artifact bindings are required by release policy") + else: + failures.extend(check_accuracy_artifact_bindings(report, expected_artifacts)) + return failures reference = _mapping(report.get("reference")) candidate = _mapping(report.get("candidate")) @@ -224,9 +263,7 @@ def check_quantized_parity_report( precision = str(report.get("precision", "")) branch = str(report.get("branch", "")) - precision_thresholds = _mapping( - _mapping(thresholds.get("precisions")).get(precision) - ) + precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) if not precision_thresholds: return [f"candidate precision {precision} has no quantization thresholds"] @@ -245,45 +282,121 @@ def check_quantized_parity_report( continue allowed = float(precision_thresholds[threshold_name]) if value > allowed: - failures.append( - f"{precision} {branch} {metric} {value:.6f} exceeds {allowed:.6f}" - ) + failures.append(f"{precision} {branch} {metric} {value:.6f} exceeds {allowed:.6f}") + return failures + + +def check_quantized_benchmark_report( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str], +) -> list[str]: + """Validate benchmark evidence covers every release branch and precision.""" + + branches = _mapping(report.get("branches")) + if not branches: + return ["benchmark report must contain branches"] + required_precisions = _required_precisions(thresholds) + required_fields = tuple(_mapping(thresholds.get("benchmark")).get("report", ())) + failures: list[str] = [] + for branch in required_branches: + branch_report = _mapping(branches.get(branch)) + if not branch_report: + failures.append(f"benchmark report missing branch {branch}") + continue + for precision in required_precisions: + precision_report = _mapping(branch_report.get(precision)) + if not precision_report: + failures.append(f"benchmark report missing {branch} {precision}") + continue + for field in required_fields: + value = _benchmark_field(precision_report, str(field)) + if _finite_float(value) is None: + failures.append(f"benchmark report has invalid {branch} {precision} {field}") return failures def _check_branch_accuracy_report( report: Mapping[str, Any], thresholds: Mapping[str, Any], + *, + required_branches: Sequence[str], ) -> list[str]: branches = _mapping(report.get("branches")) if not branches: return ["quantized COCO report must contain branch comparisons"] reference_precision = str(report.get("reference_precision", "fp32")) - candidate_precision = str( - report.get("candidate_precision", report.get("precision", "")) + threshold_precisions = set(_mapping(thresholds.get("precisions"))) + candidate_precisions = _candidate_precisions(report, branches, threshold_precisions) + required_candidates = tuple( + precision + for precision in _required_precisions(thresholds) + if precision != reference_precision and precision in threshold_precisions ) - if not candidate_precision: - return ["candidate precision missing from quantized COCO report"] + if required_candidates: + candidate_precisions = required_candidates failures: list[str] = [] + if not candidate_precisions: + return ["candidate precision missing from quantized COCO report"] + for branch in required_branches: + if branch not in branches: + failures.append(f"quantized COCO report missing branch {branch}") for branch, branch_report in branches.items(): branch_data = _mapping(branch_report) - reference = _mapping( - branch_data.get(reference_precision, branch_data.get("reference")) - ) - candidate = _mapping( - branch_data.get(candidate_precision, branch_data.get("candidate")) - ) - if not reference or not candidate: - failures.append( - f"{branch} missing {reference_precision} " - f"or {candidate_precision} metrics" - ) - continue - failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) + for candidate_precision in candidate_precisions: + reference = _mapping(branch_data.get(reference_precision, branch_data.get("reference"))) + candidate = _mapping(branch_data.get(candidate_precision, branch_data.get("candidate"))) + if not reference or not candidate: + failures.append( + f"{branch} missing {reference_precision} or {candidate_precision} metrics" + ) + continue + failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) return failures +def _required_precisions(thresholds: Mapping[str, Any]) -> tuple[str, ...]: + release_policy = _mapping(thresholds.get("release_policy")) + raw_precisions = release_policy.get("required_precisions", ()) + if not isinstance(raw_precisions, Sequence) or isinstance( + raw_precisions, + (str, bytes), + ): + return () + return tuple(str(precision) for precision in raw_precisions) + + +def _benchmark_field(report: Mapping[str, Any], field: str) -> object: + if field in report: + return report[field] + latency = _mapping(report.get("latency")) + return latency.get(field) + + +def _candidate_precisions( + report: Mapping[str, Any], + branches: Mapping[str, Any], + threshold_precisions: set[str], +) -> tuple[str, ...]: + raw_candidate_precisions = report.get("candidate_precisions") + if isinstance(raw_candidate_precisions, Sequence) and not isinstance( + raw_candidate_precisions, + (str, bytes), + ): + candidates = {str(precision) for precision in raw_candidate_precisions} + else: + candidates = set() + raw_candidate_precision = report.get("candidate_precision", report.get("precision")) + if raw_candidate_precision is not None: + candidates.add(str(raw_candidate_precision)) + for branch_report in branches.values(): + branch_data = _mapping(branch_report) + candidates.update(str(key) for key in branch_data if str(key) in threshold_precisions) + return tuple(sorted(candidates)) + + def _check_accuracy_pair( reference: Mapping[str, Any], candidate: Mapping[str, Any], @@ -299,9 +412,7 @@ def _check_accuracy_pair( ] precision = str(candidate.get("precision", "")) - precision_thresholds = _mapping( - _mapping(thresholds.get("precisions")).get(precision) - ) + precision_thresholds = _mapping(_mapping(thresholds.get("precisions")).get(precision)) if not precision_thresholds: return [f"candidate precision {precision} has no quantization thresholds"] diff --git a/scripts/evaluate_onnx_coco.py b/scripts/evaluate_onnx_coco.py index 7e709ee..9d6c059 100644 --- a/scripts/evaluate_onnx_coco.py +++ b/scripts/evaluate_onnx_coco.py @@ -56,6 +56,12 @@ def parse_args() -> argparse.Namespace: help="Provider alias or ORT provider name. May be repeated or comma-separated.", ) parser.add_argument("--seed", type=int, default=0) + parser.add_argument( + "--precision", + choices=("fp32", "fp16", "int8"), + default="fp32", + help="artifact precision label recorded for quantized release gates", + ) parser.add_argument("--confidence", type=float, default=None) parser.add_argument("--nms-iou", type=float, default=None) parser.add_argument("--max-detections", type=int, default=None) @@ -198,6 +204,7 @@ def main() -> None: "schema_version": 1, "format_version": 1, "backend": "onnx", + "precision": args.precision, "framework_commit": _framework_commit(), "checkpoint": str(args.model), "checkpoint_sha256": _checkpoint_sha256(args.model), @@ -209,6 +216,7 @@ def main() -> None: "split": "val2017", "images": len(image_ids), "evaluated_images": len(image_ids), + "image_ids": image_ids, "annotations": str(args.annotations), "annotations_sha256": _sha256_file(args.annotations), "image_list_sha256": _image_list_sha256(coco, image_ids), @@ -225,6 +233,7 @@ def main() -> None: "branches": { branch: { "branch": branch, + "precision": args.precision, "contract": _branch_contract(branch), "predictions": str(prediction_path), "detections": len(predictions), diff --git a/scripts/evaluate_tr_hash_coco.py b/scripts/evaluate_tr_hash_coco.py index a490441..2255628 100644 --- a/scripts/evaluate_tr_hash_coco.py +++ b/scripts/evaluate_tr_hash_coco.py @@ -18,7 +18,7 @@ import time from contextlib import nullcontext from pathlib import Path -from typing import Any, Iterable +from typing import Any, Iterable, Mapping import numpy as np import torch @@ -338,23 +338,23 @@ def _write_markdown_report(report: dict[str, Any], path: Path) -> None: "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |", ] for branch, branch_report in report["branches"].items(): - metrics = branch_report["metrics"] - lines.append( - "| " - + " | ".join( - [ - branch, - f"{float(metrics['map50_95']):.6f}", - f"{float(metrics['map50']):.6f}", - f"{float(metrics['map75']):.6f}", - f"{float(metrics['ap_small']):.6f}", - f"{float(metrics['ap_medium']):.6f}", - f"{float(metrics['ap_large']):.6f}", - f"{float(metrics['ar_100']):.6f}", - ] + for label, metrics in _branch_metric_rows(branch, branch_report): + lines.append( + "| " + + " | ".join( + [ + label, + f"{float(metrics['map50_95']):.6f}", + f"{float(metrics['map50']):.6f}", + f"{float(metrics['map75']):.6f}", + f"{float(metrics['ap_small']):.6f}", + f"{float(metrics['ap_medium']):.6f}", + f"{float(metrics['ap_large']):.6f}", + f"{float(metrics['ar_100']):.6f}", + ] + ) + + " |" ) - + " |" - ) lines.extend( [ "", @@ -373,6 +373,21 @@ def _write_markdown_report(report: dict[str, Any], path: Path) -> None: path.write_text("\n".join(lines), encoding="utf-8") +def _branch_metric_rows( + branch: str, + branch_report: Mapping[str, Any], +) -> list[tuple[str, Mapping[str, Any]]]: + if "metrics" in branch_report: + return [(branch, branch_report["metrics"])] + + rows: list[tuple[str, Mapping[str, Any]]] = [] + for precision, precision_report in branch_report.items(): + if not isinstance(precision_report, Mapping) or "metrics" not in precision_report: + continue + rows.append((f"{branch} {precision}", precision_report["metrics"])) + return rows + + def main() -> None: args = parse_args() if args.batch_size <= 0 or args.warmup < 0 or args.max_detections <= 0: @@ -415,6 +430,7 @@ def main() -> None: "split": "val2017", "images": len(image_ids), "evaluated_images": len(image_ids), + "image_ids": image_ids, "annotations": str(args.annotations), "annotations_sha256": _sha256_file(args.annotations), "image_list_sha256": _image_list_sha256(coco, image_ids), diff --git a/scripts/merge_vision_v8_coco_reports.py b/scripts/merge_vision_v8_coco_reports.py index 90a6a47..c616dfa 100644 --- a/scripts/merge_vision_v8_coco_reports.py +++ b/scripts/merge_vision_v8_coco_reports.py @@ -35,6 +35,10 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: merged["metadata"] = "multiple ONNX branch sidecars" merged["branches"] = {} shared_keys = ("backend", "framework_commit", "dataset", "protocol") + precision_reports = [_report_precision(report) for report in reports] + nest_by_precision = any(precision is not None for precision in precision_reports) + if nest_by_precision and not all(precision_reports): + raise ValueError("cannot merge mixed precision and non-precision reports") for report in reports: for key in shared_keys: if report.get(key) != reports[0].get(key): @@ -44,8 +48,6 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: if not isinstance(branches, Mapping): raise ValueError("each report must contain a branches object") for branch, branch_report in branches.items(): - if branch in merged["branches"]: - raise ValueError(f"duplicate branch in merged reports: {branch}") if not isinstance(branch_report, Mapping): raise ValueError(f"branch report must be an object: {branch}") merged_branch = deepcopy(dict(branch_report)) @@ -57,11 +59,41 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: "metadata_sha256", ): merged_branch.setdefault(key, report.get(key)) - merged["branches"][branch] = merged_branch + precision = _report_precision(report) + if nest_by_precision: + assert precision is not None + merged_branch.setdefault("precision", precision) + branch_precisions = merged["branches"].setdefault(branch, {}) + if precision in branch_precisions: + raise ValueError(f"duplicate precision in merged reports: {branch} {precision}") + branch_precisions[precision] = merged_branch + else: + if branch in merged["branches"]: + raise ValueError(f"duplicate branch in merged reports: {branch}") + merged["branches"][branch] = merged_branch + + if nest_by_precision: + precisions = tuple(dict.fromkeys(str(precision) for precision in precision_reports)) + if "fp32" in precisions: + merged["reference_precision"] = "fp32" + merged["candidate_precisions"] = [ + precision for precision in precisions if precision != "fp32" + ] + else: + merged["candidate_precisions"] = list(precisions) return merged +def _report_precision(report: Mapping[str, Any]) -> str | None: + precision = report.get("precision") + if precision is None: + protocol = report.get("protocol") + if isinstance(protocol, Mapping): + precision = protocol.get("precision") + return str(precision) if precision is not None else None + + def main() -> None: args = parse_args() report = merge_reports([load_json(path) for path in args.reports]) diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index 64647fb..997abcc 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -2,6 +2,7 @@ import pytest +from scripts.check_onnx_quantized_artifacts import check_quantized_accuracy_report from scripts.evaluate_tr_hash_coco import ( BRANCHES, _branch_contract, @@ -69,9 +70,7 @@ def test_checkpoint_sha256_hashes_loaded_directory_weights(tmp_path): ema_weights.unlink() - assert _checkpoint_sha256(checkpoint) == hashlib.sha256( - b"model weights" - ).hexdigest() + assert _checkpoint_sha256(checkpoint) == hashlib.sha256(b"model weights").hexdigest() def test_branch_contract_distinguishes_nms_requirements(): @@ -156,6 +155,58 @@ def test_merge_onnx_reports_keeps_both_branch_artifact_hashes(): assert merged["branches"]["nms-free"]["metadata_sha256"] == "d" * 64 +def test_merge_onnx_reports_can_build_precision_nested_quantized_report(): + fp32 = { + "backend": "onnx", + "precision": "fp32", + "framework_commit": "abc123", + "checkpoint": "o2m_fp32.onnx", + "checkpoint_sha256": "a" * 64, + "model": "o2m_fp32.onnx", + "metadata": "o2m.json", + "metadata_sha256": "b" * 64, + "dataset": {"name": "coco-2017", "image_ids": [1, 2]}, + "environment": {}, + "protocol": {"seed": 0}, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.3}, + } + }, + } + fp16 = { + **fp32, + "precision": "fp16", + "checkpoint": "o2m_fp16.onnx", + "checkpoint_sha256": "c" * 64, + "model": "o2m_fp16.onnx", + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.295}, + } + }, + } + + merged = merge_reports([fp32, fp16]) + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16"]}, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + assert set(merged["branches"]["o2m-nms"]) == {"fp32", "fp16"} + assert merged["branches"]["o2m-nms"]["fp16"]["checkpoint_sha256"] == "c" * 64 + assert ( + check_quantized_accuracy_report( + merged, + thresholds, + required_branches=["o2m-nms"], + ) + == [] + ) + + def test_merge_onnx_reports_rejects_mismatched_protocol(): first = { "backend": "onnx", diff --git a/tests/test_onnx_quantization_accuracy_gate.py b/tests/test_onnx_quantization_accuracy_gate.py index f8b16d0..798a279 100644 --- a/tests/test_onnx_quantization_accuracy_gate.py +++ b/tests/test_onnx_quantization_accuracy_gate.py @@ -1,6 +1,10 @@ +import pytest + from scripts.check_onnx_quantized_artifacts import ( + check_accuracy_artifact_bindings, check_quantized_accuracy_report, check_quantized_parity_report, + evaluation_image_ids_from_report, ) @@ -17,11 +21,7 @@ def test_quantized_accuracy_fails_when_map_drop_exceeds_precision_threshold() -> "metrics": {"map50_95": 0.17, "map50": 0.31}, }, } - thresholds = { - "precisions": { - "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03} - } - } + thresholds = {"precisions": {"int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}}} failures = check_quantized_accuracy_report(report, thresholds) @@ -42,11 +42,7 @@ def test_quantized_accuracy_accepts_candidate_within_threshold() -> None: "metrics": {"map50_95": 0.098, "map50": 0.135}, }, } - thresholds = { - "precisions": { - "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} - } - } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} assert check_quantized_accuracy_report(report, thresholds) == [] @@ -64,17 +60,11 @@ def test_quantized_accuracy_rejects_branch_mismatch() -> None: "metrics": {"map50_95": 0.2, "map50": 0.32}, }, } - thresholds = { - "precisions": { - "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} - } - } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} failures = check_quantized_accuracy_report(report, thresholds) - assert failures == [ - "candidate branch nms-free does not match FP32 reference branch o2m-nms" - ] + assert failures == ["candidate branch nms-free does not match FP32 reference branch o2m-nms"] def test_quantized_accuracy_accepts_evaluator_branch_report() -> None: @@ -96,32 +86,122 @@ def test_quantized_accuracy_accepts_evaluator_branch_report() -> None: } }, } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + assert check_quantized_accuracy_report(report, thresholds) == [] + + +def test_quantized_accuracy_checks_every_precision_in_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + "int8": { + "precision": "int8", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.01, "map50": 0.02}, + }, + } + }, + } thresholds = { "precisions": { - "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, } } - assert check_quantized_accuracy_report(report, thresholds) == [] + failures = check_quantized_accuracy_report(report, thresholds) + assert any("int8 o2m-nms map50_95" in failure for failure in failures) -def test_quantized_accuracy_rejects_empty_evaluator_branch_report() -> None: + +def test_quantized_accuracy_requires_release_policy_precisions() -> None: report = { "reference_precision": "fp32", "candidate_precision": "fp16", - "branches": {}, + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, } thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16", "int8"]}, "precisions": { - "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} - } + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, + }, } failures = check_quantized_accuracy_report(report, thresholds) + assert failures == ["o2m-nms missing fp32 or int8 metrics"] + + +def test_quantized_accuracy_rejects_empty_evaluator_branch_report() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": {}, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report(report, thresholds) + assert failures == ["quantized COCO report must contain branch comparisons"] +def test_quantized_accuracy_requires_configured_branches() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.315}, + }, + } + }, + } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} + + failures = check_quantized_accuracy_report( + report, + thresholds, + required_branches=["o2m-nms", "nms-free"], + ) + + assert "quantized COCO report missing branch nms-free" in failures + + def test_quantized_accuracy_rejects_non_finite_metrics() -> None: report = { "reference": { @@ -135,11 +215,7 @@ def test_quantized_accuracy_rejects_non_finite_metrics() -> None: "metrics": {"map50_95": float("nan"), "map50": 0.32}, }, } - thresholds = { - "precisions": { - "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01} - } - } + thresholds = {"precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}} failures = check_quantized_accuracy_report(report, thresholds) @@ -166,6 +242,139 @@ def test_quantized_parity_consumes_raw_and_decoded_thresholds() -> None: failures = check_quantized_parity_report(report, thresholds) + assert failures == ["int8 o2m-nms max_raw_logit_abs_error 0.130000 exceeds 0.120000"] + + +def test_accuracy_report_must_include_actual_evaluation_ids() -> None: + assert evaluation_image_ids_from_report({"dataset": {"image_ids": [3, 2, 2]}}) == {2, 3} + + +def test_accuracy_report_rejects_missing_evaluation_ids() -> None: + with pytest.raises(ValueError, match="evaluation image IDs"): + evaluation_image_ids_from_report({"dataset": {"disjoint_from": "train2017"}}) + + +def test_accuracy_report_artifact_hashes_must_match_generated_release_artifacts() -> None: + report = { + "branches": { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + } + generated = { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "e" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + + failures = check_accuracy_artifact_bindings(report, generated) + assert failures == [ - "int8 o2m-nms max_raw_logit_abs_error 0.130000 exceeds 0.120000" + "o2m-nms fp16 checkpoint_sha256 " + "cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc " + "does not match generated " + "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee" ] + + +def test_release_accuracy_gate_requires_generated_artifact_bindings() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16"], + "require_artifact_bindings": True, + }, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == ["generated artifact bindings are required by release policy"] + + +def test_release_accuracy_gate_rejects_bogus_hashes_against_generated_artifacts() -> None: + report = { + "reference_precision": "fp32", + "candidate_precision": "fp16", + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "checkpoint_sha256": "c" * 64, + "metadata_sha256": "d" * 64, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16"], + "require_artifact_bindings": True, + }, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + generated = { + "o2m-nms": { + "fp32": { + "checkpoint_sha256": "a" * 64, + "metadata_sha256": "b" * 64, + }, + "fp16": { + "checkpoint_sha256": "e" * 64, + "metadata_sha256": "d" * 64, + }, + } + } + + failures = check_quantized_accuracy_report( + report, + thresholds, + expected_artifacts=generated, + ) + + assert any("o2m-nms fp16 checkpoint_sha256" in failure for failure in failures) diff --git a/tests/test_onnx_quantization_benchmark.py b/tests/test_onnx_quantization_benchmark.py index 3f1c0f2..ff66134 100644 --- a/tests/test_onnx_quantization_benchmark.py +++ b/tests/test_onnx_quantization_benchmark.py @@ -1,4 +1,7 @@ -from scripts.benchmark_onnx_artifacts import summarize_latency_ms +import numpy as np + +from scripts.benchmark_onnx_artifacts import _benchmark_session, summarize_latency_ms +from scripts.check_onnx_quantized_artifacts import check_quantized_benchmark_report def test_benchmark_report_uses_distribution_not_single_shot() -> None: @@ -17,3 +20,69 @@ def test_benchmark_report_handles_empty_measurements() -> None: assert summary["mean_ms"] == 0.0 assert summary["stddev_ms"] == 0.0 assert summary["p95_ms"] == 0.0 + + +def test_benchmark_report_requires_every_branch_and_precision() -> None: + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16", "int8"]}, + "benchmark": { + "report": [ + "median_ms", + "throughput_images_per_second", + "peak_memory_mb", + ] + }, + } + report = { + "branches": { + "o2m": { + "fp32": { + "latency": {"median_ms": 1.0}, + "throughput_images_per_second": 10.0, + "peak_memory_mb": 100.0, + } + } + } + } + + failures = check_quantized_benchmark_report( + report, + thresholds, + required_branches=["o2m", "nms-free"], + ) + + assert "benchmark report missing o2m fp16" in failures + assert "benchmark report missing branch nms-free" in failures + + +def test_benchmark_session_reports_observed_peak_memory( + monkeypatch, +) -> None: + memory_samples = iter([100.0, 110.0, 105.0, 125.0]) + + class Session: + def run(self, values): + assert values.shape == (1, 3, 4, 4) + return np.zeros((1, 1), dtype=np.float32) + + class Pipeline: + class Metadata: + image_size = 4 + + metadata = Metadata() + session = Session() + + monkeypatch.setattr( + "scripts.benchmark_onnx_artifacts._current_memory_mb", + lambda: next(memory_samples), + ) + + latencies, peak_memory_mb = _benchmark_session( + Pipeline(), + batch_size=1, + warmup_iterations=1, + measured_iterations=2, + ) + + assert len(latencies) == 2 + assert peak_memory_mb == 125.0 diff --git a/tests/test_onnx_release.py b/tests/test_onnx_release.py index b02518b..993e2dc 100644 --- a/tests/test_onnx_release.py +++ b/tests/test_onnx_release.py @@ -1,6 +1,8 @@ from __future__ import annotations import json +import subprocess +import sys from pathlib import Path from typing import Any @@ -43,7 +45,7 @@ def config_mapping(**overrides: Any) -> dict[str, Any]: "checkpoint_revision": "f3b3e659612e543ca9ff91892c0662d38dc1a1d6", "opset": 17, "parity_num_tests": 5, - "toolchain": {"torch": "2.13.0", "onnx": "1.21.0", "onnxruntime": "1.24.4"}, + "toolchain": {"torch": "2.13.0", "onnx": "1.21.0", "onnxruntime": "1.23.2"}, "branches": [ { "branch": "o2m", @@ -100,14 +102,13 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: assert {spec.branch for spec in config.branches} == {"o2m", "nms-free"} assert config.opset == 17 assert set(config.toolchain) == {"torch", "onnx", "onnxruntime"} + assert config.toolchain["onnxruntime"] == "1.23.2" assert config.quantization is not None assert config.quantization.enabled_precisions == ("fp16", "int8") assert config.quantization.calibration_manifest == Path( - "configs/vision_v8_quantization_calibration.json" - ) - assert config.quantization.thresholds == Path( - "configs/vision_v8_quantization_thresholds.json" + "artifacts/vision_v8_quantized_eval/calibration.json" ) + assert config.quantization.thresholds == Path("configs/vision_v8_quantization_thresholds.json") assert config.quantization.accuracy_report == Path( "artifacts/vision_v8_quantized_eval/accuracy.json" ) @@ -115,7 +116,8 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: "artifacts/vision_v8_quantized_eval/accuracy.md" ) assert config.quantization.provider_gates == ( - ("CUDAExecutionProvider", "fp16"), + ("CPUExecutionProvider", "fp32"), + ("CPUExecutionProvider", "fp16"), ("CPUExecutionProvider", "int8"), ) # Five seeds underestimate the observed maxima (see the validation report), @@ -123,6 +125,55 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: assert config.parity_num_tests >= 20 +def test_release_workflows_share_quantized_evidence_artifact_contract() -> None: + release_workflow = Path(".github/workflows/onnx-release.yml").read_text(encoding="utf-8") + coco_workflow = Path(".github/workflows/vision-v8-coco-accuracy.yml").read_text( + encoding="utf-8" + ) + + assert "--name vision-v8-quantized-release-inputs" in release_workflow + assert "name: vision-v8-quantized-release-inputs" in coco_workflow + assert "calibration.json" in coco_workflow + assert "accuracy.json" in coco_workflow + assert "accuracy.md" in coco_workflow + + +def test_release_builder_runs_by_documented_file_path(tmp_path: Path) -> None: + config_path = tmp_path / "release.json" + config_path.write_text( + json.dumps( + config_mapping( + quantization={ + "enabled_precisions": ["int8"], + "calibration_manifest": str(tmp_path / "missing-calibration.json"), + "thresholds": "configs/vision_v8_quantization_thresholds.json", + "accuracy_report": str(tmp_path / "missing-accuracy.json"), + "accuracy_markdown": str(tmp_path / "missing-accuracy.md"), + } + ) + ), + encoding="utf-8", + ) + + result = subprocess.run( + [ + sys.executable, + "scripts/build_onnx_release.py", + "--config", + str(config_path), + "--output-dir", + str(tmp_path / "dist"), + "--allow-toolchain-drift", + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode != 0 + assert "No module named 'scripts." not in result.stderr + + def test_a_moving_checkpoint_ref_is_rejected() -> None: with pytest.raises(ReleaseError, match="40-character commit sha"): config_from_mapping(config_mapping(checkpoint_revision="main")) @@ -179,9 +230,7 @@ def test_manifest_carries_every_field_the_release_must_document( ) -> None: manifest = written_release(tmp_path) - assert manifest["checkpoint_revision"] == ( - "f3b3e659612e543ca9ff91892c0662d38dc1a1d6" - ) + assert manifest["checkpoint_revision"] == ("f3b3e659612e543ca9ff91892c0662d38dc1a1d6") assert manifest["framework_commit"] == "0" * 40 assert manifest["opset"] == 17 assert manifest["toolchain"]["torch"] == "2.13.0" From e57ab2784a0a848432c27a28e2db0f631fbc2b64 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Sat, 5 Sep 2026 20:50:03 +0500 Subject: [PATCH 4/5] Fix quantized release runner evidence --- .github/workflows/onnx-release.yml | 10 +- .github/workflows/vision-v8-coco-accuracy.yml | 5 +- .../vision_v8_quantization_thresholds.json | 2 +- docs/onnx/release.json | 2 +- docs/vision-v8-onnx-quantization.md | 16 ++- scripts/check_onnx_quantized_artifacts.py | 39 +++++++ scripts/check_onnx_quantized_parity.py | 4 +- scripts/quantize_onnx.py | 15 +-- tests/test_onnx_quantization_config.py | 102 ++++++++++++++++++ tests/test_onnx_release.py | 5 +- 10 files changed, 176 insertions(+), 24 deletions(-) diff --git a/.github/workflows/onnx-release.yml b/.github/workflows/onnx-release.yml index 3e52613..fd5252b 100644 --- a/.github/workflows/onnx-release.yml +++ b/.github/workflows/onnx-release.yml @@ -16,6 +16,11 @@ on: description: "Workflow run ID containing vision-v8-quantized-release-inputs" required: false type: string + runner: + description: "Runner label with CUDA support for FP16 ONNX gates" + required: false + default: "self-hosted" + type: string push: tags: - "onnx-v8-*" @@ -30,7 +35,7 @@ concurrency: jobs: publish: - runs-on: ubuntu-latest + runs-on: ${{ inputs.runner || 'self-hosted' }} # Two 640px exports plus 50-seed parity gates on both branches. timeout-minutes: 60 steps: @@ -60,7 +65,7 @@ jobs: --index-url https://download.pytorch.org/whl/cpu python -m pip install \ "onnx==${{ steps.pins.outputs.onnx }}" \ - "onnxruntime==${{ steps.pins.outputs.onnxruntime }}" \ + "onnxruntime-gpu==${{ steps.pins.outputs.onnxruntime }}" \ psutil - name: Download quantized release evidence @@ -74,6 +79,7 @@ jobs: --name vision-v8-quantized-release-inputs \ --dir artifacts/vision_v8_quantized_eval test -f artifacts/vision_v8_quantized_eval/calibration.json + test -d artifacts/vision_v8_quantized_eval/calibration_images test -f artifacts/vision_v8_quantized_eval/accuracy.json test -f artifacts/vision_v8_quantized_eval/accuracy.md diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index e2ca47c..40600c3 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -189,6 +189,8 @@ jobs: --config configs/vision_v8_coco_accuracy_gate.json - name: Build quantized release evidence artifact if: inputs.backend == 'onnx' && inputs.calibration_manifest != '' + env: + CALIBRATION_MANIFEST: ${{ inputs.calibration_manifest }} run: | test -n "${{ inputs.onnx_o2m_fp16_model }}" test -n "${{ inputs.onnx_o2m_fp16_metadata }}" @@ -199,7 +201,7 @@ jobs: test -n "${{ inputs.onnx_nms_free_int8_model }}" test -n "${{ inputs.onnx_nms_free_int8_metadata }}" mkdir -p artifacts/vision_v8_quantized_eval - cp "${{ inputs.calibration_manifest }}" artifacts/vision_v8_quantized_eval/calibration.json + python -c "import os; from pathlib import Path; from scripts.check_onnx_quantized_artifacts import materialize_calibration_manifest_images; materialize_calibration_manifest_images(Path(os.environ['CALIBRATION_MANIFEST']), artifact_root=Path('artifacts/vision_v8_quantized_eval'), output_manifest_path=Path('artifacts/vision_v8_quantized_eval/calibration.json'))" python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_fp16_model }}" --metadata "${{ inputs.onnx_o2m_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_fp16 --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp16 --seed 0 python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_fp16_model }}" --metadata "${{ inputs.onnx_nms_free_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_fp16 --branch nms-free --provider "${{ inputs.provider }}" --precision fp16 --seed 0 python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_int8_model }}" --metadata "${{ inputs.onnx_o2m_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_int8 --branch o2m-nms --provider "${{ inputs.provider }}" --precision int8 --seed 0 @@ -228,6 +230,7 @@ jobs: name: vision-v8-quantized-release-inputs path: | artifacts/vision_v8_quantized_eval/calibration.json + artifacts/vision_v8_quantized_eval/calibration_images/** artifacts/vision_v8_quantized_eval/accuracy.json artifacts/vision_v8_quantized_eval/accuracy.md - name: Upload gated reports to GitHub Release diff --git a/configs/vision_v8_quantization_thresholds.json b/configs/vision_v8_quantization_thresholds.json index 992f0fa..f0f64e3 100644 --- a/configs/vision_v8_quantization_thresholds.json +++ b/configs/vision_v8_quantization_thresholds.json @@ -25,7 +25,7 @@ } }, "providers": { - "CPUExecutionProvider": ["fp32", "fp16", "int8"], + "CPUExecutionProvider": ["fp32", "int8"], "CUDAExecutionProvider": ["fp32", "fp16"], "TensorrtExecutionProvider": ["fp32", "fp16", "int8"] }, diff --git a/docs/onnx/release.json b/docs/onnx/release.json index f3ca266..e64308f 100644 --- a/docs/onnx/release.json +++ b/docs/onnx/release.json @@ -12,7 +12,7 @@ "fp32_op_allowlist": [], "provider_gates": [ {"provider": "CPUExecutionProvider", "precision": "fp32"}, - {"provider": "CPUExecutionProvider", "precision": "fp16"}, + {"provider": "CUDAExecutionProvider", "precision": "fp16"}, {"provider": "CPUExecutionProvider", "precision": "int8"} ] }, diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md index 9530ecb..517d36e 100644 --- a/docs/vision-v8-onnx-quantization.md +++ b/docs/vision-v8-onnx-quantization.md @@ -20,6 +20,8 @@ Unsupported provider and precision pairs fail explicitly using `configs/vision_v8_quantization_thresholds.json`. Provider fallback is separate from FP16 partial fallback: the release check must verify both that ONNX Runtime used the requested provider and that unexpected graph nodes did not remain FP32. +The default release contract runs FP32 and INT8 on CPU, but FP16 on CUDA because +the CPU ONNX Runtime build does not support executing FP16 detector graphs. ## Calibration Contract @@ -135,11 +137,15 @@ The GitHub release workflow downloads a prior Actions artifact named `vision-v8-quantized-release-inputs`. The manual Vision v8 COCO accuracy workflow produces that artifact when `backend=onnx`, `calibration_manifest` is set, and FP32/FP16/INT8 model plus metadata paths are provided for both -branches. The artifact must provide `calibration.json`, `accuracy.json`, and -`accuracy.md` under `artifacts/vision_v8_quantized_eval/` before -`build_onnx_release.py` runs. Pass the source run as workflow input -`evidence_run_id`, or set repository variable `ONNX_RELEASE_EVIDENCE_RUN_ID` -for tag-triggered releases. +branches. The artifact must provide `calibration.json`, `calibration_images/`, +`accuracy.json`, and `accuracy.md` under +`artifacts/vision_v8_quantized_eval/` before `build_onnx_release.py` runs. The +COCO workflow copies the pinned calibration images into `calibration_images/` +and rewrites the manifest paths so the fresh release runner can open them. +Pass the source run as workflow input `evidence_run_id`, or set repository +variable `ONNX_RELEASE_EVIDENCE_RUN_ID` for tag-triggered releases. +Run the ONNX release workflow on a CUDA-capable self-hosted runner so the FP16 +parity and benchmark gates use `onnxruntime-gpu` instead of CPU fallback. ## Release Policy diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py index c7563c2..7e40563 100644 --- a/scripts/check_onnx_quantized_artifacts.py +++ b/scripts/check_onnx_quantized_artifacts.py @@ -5,6 +5,7 @@ import hashlib import json import math +import shutil import string from collections.abc import Mapping, Sequence from pathlib import Path @@ -73,9 +74,43 @@ def load_calibration_manifest(path: Path) -> dict[str, Any]: raise ValueError("calibration manifest images must be a sequence") if not images: raise ValueError("calibration manifest images must not be empty") + manifest["images"] = [ + str(_resolve_manifest_path(path.parent, Path(str(image)))) for image in images + ] return manifest +def materialize_calibration_manifest_images( + manifest_path: Path, + *, + artifact_root: Path, + output_manifest_path: Path | None = None, + image_directory_name: str = "calibration_images", +) -> None: + """Copy calibration images beside an evidence manifest and rewrite paths.""" + + manifest = _load_json(manifest_path) + images = manifest.get("images") + if not isinstance(images, Sequence) or isinstance(images, (str, bytes)): + raise ValueError("calibration manifest images must be a sequence") + + destination_dir = artifact_root / image_directory_name + destination_dir.mkdir(parents=True, exist_ok=True) + rewritten: list[str] = [] + for index, image in enumerate(images): + source = _resolve_manifest_path(manifest_path.parent, Path(str(image))) + if not source.is_file(): + raise ValueError(f"calibration image does not exist: {source}") + destination = destination_dir / f"{index:06d}_{source.name}" + shutil.copyfile(source, destination) + rewritten.append(destination.relative_to(artifact_root).as_posix()) + + manifest["images"] = rewritten + output_path = output_manifest_path or manifest_path + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") + + def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: """Hash a stable sorted image-ID manifest for calibration/eval pinning.""" @@ -85,6 +120,10 @@ def image_id_manifest_sha256(image_ids: Sequence[int]) -> str: return digest.hexdigest() +def _resolve_manifest_path(base: Path, path: Path) -> Path: + return path if path.is_absolute() else base / path + + def assert_disjoint_image_ids( calibration_ids: set[int], evaluation_ids: set[int], diff --git a/scripts/check_onnx_quantized_parity.py b/scripts/check_onnx_quantized_parity.py index ba7a9ca..bf988c7 100644 --- a/scripts/check_onnx_quantized_parity.py +++ b/scripts/check_onnx_quantized_parity.py @@ -44,9 +44,7 @@ def decoded_parity_metrics( return { "reference_detection_count": len(reference_detections), "candidate_detection_count": len(candidate_detections), - "class_mismatch_count": int( - len(reference_detections) != len(candidate_detections) - ), + "class_mismatch_count": int(len(reference_detections) != len(candidate_detections)), "max_decoded_box_px_error": 0.0, "max_score_abs_error": 0.0, } diff --git a/scripts/quantize_onnx.py b/scripts/quantize_onnx.py index b45d2e3..3fcb1a5 100644 --- a/scripts/quantize_onnx.py +++ b/scripts/quantize_onnx.py @@ -36,10 +36,7 @@ def parse_args() -> argparse.Namespace: parser.add_argument( "--detector-metadata-output", type=Path, - help=( - "Detector metadata sidecar copied from --metadata. " - "Defaults to .json." - ), + help=("Detector metadata sidecar copied from --metadata. Defaults to .json."), ) parser.add_argument("--calibration-manifest", type=Path) parser.add_argument("--checkpoint-revision", default="unknown") @@ -191,8 +188,7 @@ def __init__( raise ValueError("calibration batch_size must be positive") self._batch_size = batch_size self._items = [ - preprocess_image(path, metadata.image_size).pixel_values - for path in image_paths + preprocess_image(path, metadata.image_size).pixel_values for path in image_paths ] self._index = 0 @@ -246,8 +242,7 @@ def quantize_int8( from onnxruntime.quantization import QuantFormat, quantize_static except ImportError as error: # pragma: no cover - dependency guard raise RuntimeError( - "INT8 quantization requires onnxruntime.quantization " - "and a calibration manifest." + "INT8 quantization requires onnxruntime.quantization and a calibration manifest." ) from error settings = calibration_manifest["quantization"] @@ -358,8 +353,8 @@ def main() -> None: if args.require_identical_hash: assert_identical_artifact_hashes(output_sha256, repeat_sha256) - detector_metadata_output = ( - args.detector_metadata_output or default_detector_metadata_output(args.output) + detector_metadata_output = args.detector_metadata_output or default_detector_metadata_output( + args.output ) copy_detector_metadata(args.metadata, detector_metadata_output) diff --git a/tests/test_onnx_quantization_config.py b/tests/test_onnx_quantization_config.py index 419df81..6073a7f 100644 --- a/tests/test_onnx_quantization_config.py +++ b/tests/test_onnx_quantization_config.py @@ -1,3 +1,4 @@ +import json from pathlib import Path import pytest @@ -7,6 +8,7 @@ image_id_manifest_sha256, load_calibration_manifest, load_quantization_thresholds, + materialize_calibration_manifest_images, ) HASH = "a" * 64 @@ -220,3 +222,103 @@ def test_calibration_and_eval_image_ids_must_be_disjoint() -> None: def test_image_id_manifest_hash_is_order_stable() -> None: assert image_id_manifest_sha256([3, 1, 2]) == image_id_manifest_sha256([1, 2, 3]) + + +def test_materializes_calibration_images_and_rewrites_manifest_paths(tmp_path: Path) -> None: + source_dir = tmp_path / "source" + source_dir.mkdir() + first = source_dir / "000000000009.jpg" + second = source_dir / "000000000025.jpg" + first.write_bytes(b"first") + second.write_bytes(b"second") + manifest_path = tmp_path / "calibration.json" + manifest_path.write_text( + json.dumps( + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": image_id_manifest_sha256([9, 25]), + "annotations_sha256": HASH, + "disjoint_from": "coco-2017-val2017", + }, + "image_ids": [9, 25], + "images": [str(first), str(second)], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 1, + }, + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + materialize_calibration_manifest_images( + manifest_path, + artifact_root=tmp_path, + ) + manifest = load_calibration_manifest(manifest_path) + expected_first = tmp_path / "calibration_images" / "000000_000000000009.jpg" + expected_second = tmp_path / "calibration_images" / "000001_000000000025.jpg" + + assert manifest["images"] == [str(expected_first), str(expected_second)] + assert expected_first.read_bytes() == b"first" + assert expected_second.read_bytes() == b"second" + + +def test_materializes_calibration_manifest_to_separate_artifact_path( + tmp_path: Path, +) -> None: + source_dir = tmp_path / "source" + source_dir.mkdir() + (source_dir / "000000000009.jpg").write_bytes(b"first") + source_manifest = source_dir / "calibration.json" + artifact_root = tmp_path / "artifact" + output_manifest = artifact_root / "calibration.json" + source_manifest.write_text( + json.dumps( + { + "schema_version": 1, + "dataset": { + "name": "coco-2017-train", + "image_ids_sha256": image_id_manifest_sha256([9]), + "annotations_sha256": HASH, + "disjoint_from": "coco-2017-val2017", + }, + "image_ids": [9], + "images": ["000000000009.jpg"], + "quantization": { + "calibration_method": "minmax", + "per_channel": True, + "symmetric_activations": False, + "symmetric_weights": True, + "activation_type": "quint8", + "weight_type": "qint8", + "batch_size": 1, + }, + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + materialize_calibration_manifest_images( + source_manifest, + artifact_root=artifact_root, + output_manifest_path=output_manifest, + ) + + assert source_manifest.read_text(encoding="utf-8") != output_manifest.read_text( + encoding="utf-8" + ) + assert load_calibration_manifest(output_manifest)["images"] == [ + str(artifact_root / "calibration_images" / "000000_000000000009.jpg") + ] diff --git a/tests/test_onnx_release.py b/tests/test_onnx_release.py index 993e2dc..322c52a 100644 --- a/tests/test_onnx_release.py +++ b/tests/test_onnx_release.py @@ -117,7 +117,7 @@ def test_committed_release_config_is_valid_and_pins_both_branches() -> None: ) assert config.quantization.provider_gates == ( ("CPUExecutionProvider", "fp32"), - ("CPUExecutionProvider", "fp16"), + ("CUDAExecutionProvider", "fp16"), ("CPUExecutionProvider", "int8"), ) # Five seeds underestimate the observed maxima (see the validation report), @@ -132,8 +132,11 @@ def test_release_workflows_share_quantized_evidence_artifact_contract() -> None: ) assert "--name vision-v8-quantized-release-inputs" in release_workflow + assert "onnxruntime-gpu" in release_workflow + assert "runs-on: ${{ inputs.runner || 'self-hosted' }}" in release_workflow assert "name: vision-v8-quantized-release-inputs" in coco_workflow assert "calibration.json" in coco_workflow + assert "calibration_images/**" in coco_workflow assert "accuracy.json" in coco_workflow assert "accuracy.md" in coco_workflow From 18d8874b1b3046fe5b38325a64df832d940e8044 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Sun, 6 Sep 2026 23:09:38 +0500 Subject: [PATCH 5/5] Enforce quantized release evidence providers --- .github/workflows/vision-v8-coco-accuracy.yml | 32 +++++-- .../vision_v8_quantization_thresholds.json | 5 + docs/vision-v8-onnx-quantization.md | 7 +- scripts/build_onnx_release.py | 1 + scripts/check_onnx_quantized_artifacts.py | 80 +++++++++++++++- scripts/merge_vision_v8_coco_reports.py | 1 + tests/test_coco_release_evaluation.py | 46 +++++++++ tests/test_onnx_quantization_accuracy_gate.py | 93 +++++++++++++++++++ tests/test_onnx_release.py | 7 ++ 9 files changed, 257 insertions(+), 15 deletions(-) diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 40600c3..d44324d 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -93,7 +93,19 @@ on: - pytorch - onnx provider: - description: "ONNX Runtime provider alias for backend=onnx" + description: "Legacy ONNX provider alias; precision-specific provider inputs are used for release evidence" + required: false + default: "cpu" + fp32_provider: + description: "ONNX Runtime provider alias for FP32 evaluation" + required: true + default: "cpu" + fp16_provider: + description: "ONNX Runtime provider alias for FP16 evaluation" + required: true + default: "cuda" + int8_provider: + description: "ONNX Runtime provider alias for INT8 evaluation" required: true default: "cpu" release_tag: @@ -163,6 +175,8 @@ jobs: run: | python -m pip install --upgrade pip python -m pip install -e ".[detection,export]" + ONNXRUNTIME_VERSION=$(python -c "import json; print(json.load(open('docs/onnx/release.json'))['toolchain']['onnxruntime'])") + python -m pip install "onnxruntime-gpu==$ONNXRUNTIME_VERSION" - name: Run deterministic COCO evaluation twice run: | if [ "${{ inputs.backend }}" = "pytorch" ]; then @@ -174,11 +188,11 @@ jobs: test -n "${{ inputs.onnx_o2m_metadata }}" test -n "${{ inputs.onnx_nms_free_model }}" test -n "${{ inputs.onnx_nms_free_metadata }}" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp32 --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp32 --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.fp32_provider }}" --precision fp32 --seed 0 python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b fi - name: Gate full COCO report @@ -202,10 +216,10 @@ jobs: test -n "${{ inputs.onnx_nms_free_int8_metadata }}" mkdir -p artifacts/vision_v8_quantized_eval python -c "import os; from pathlib import Path; from scripts.check_onnx_quantized_artifacts import materialize_calibration_manifest_images; materialize_calibration_manifest_images(Path(os.environ['CALIBRATION_MANIFEST']), artifact_root=Path('artifacts/vision_v8_quantized_eval'), output_manifest_path=Path('artifacts/vision_v8_quantized_eval/calibration.json'))" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_fp16_model }}" --metadata "${{ inputs.onnx_o2m_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_fp16 --branch o2m-nms --provider "${{ inputs.provider }}" --precision fp16 --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_fp16_model }}" --metadata "${{ inputs.onnx_nms_free_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_fp16 --branch nms-free --provider "${{ inputs.provider }}" --precision fp16 --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_int8_model }}" --metadata "${{ inputs.onnx_o2m_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_int8 --branch o2m-nms --provider "${{ inputs.provider }}" --precision int8 --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_int8_model }}" --metadata "${{ inputs.onnx_nms_free_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_int8 --branch nms-free --provider "${{ inputs.provider }}" --precision int8 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_fp16_model }}" --metadata "${{ inputs.onnx_o2m_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_fp16 --branch o2m-nms --provider "${{ inputs.fp16_provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_fp16_model }}" --metadata "${{ inputs.onnx_nms_free_fp16_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_fp16 --branch nms-free --provider "${{ inputs.fp16_provider }}" --precision fp16 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_int8_model }}" --metadata "${{ inputs.onnx_o2m_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/o2m_int8 --branch o2m-nms --provider "${{ inputs.int8_provider }}" --precision int8 --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_int8_model }}" --metadata "${{ inputs.onnx_nms_free_int8_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_quantized_eval/nms_free_int8 --branch nms-free --provider "${{ inputs.int8_provider }}" --precision int8 --seed 0 python scripts/merge_vision_v8_coco_reports.py \ artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json \ artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json \ diff --git a/configs/vision_v8_quantization_thresholds.json b/configs/vision_v8_quantization_thresholds.json index f0f64e3..5944cf6 100644 --- a/configs/vision_v8_quantization_thresholds.json +++ b/configs/vision_v8_quantization_thresholds.json @@ -3,6 +3,11 @@ "release_policy": { "required_precisions": ["fp32", "fp16", "int8"], "require_artifact_bindings": true, + "provider_by_precision": { + "fp32": "CPUExecutionProvider", + "fp16": "CUDAExecutionProvider", + "int8": "CPUExecutionProvider" + }, "partial_release": "block", "optional_provider_precisions": [] }, diff --git a/docs/vision-v8-onnx-quantization.md b/docs/vision-v8-onnx-quantization.md index 517d36e..34076f6 100644 --- a/docs/vision-v8-onnx-quantization.md +++ b/docs/vision-v8-onnx-quantization.md @@ -137,7 +137,12 @@ The GitHub release workflow downloads a prior Actions artifact named `vision-v8-quantized-release-inputs`. The manual Vision v8 COCO accuracy workflow produces that artifact when `backend=onnx`, `calibration_manifest` is set, and FP32/FP16/INT8 model plus metadata paths are provided for both -branches. The artifact must provide `calibration.json`, `calibration_images/`, +branches. Dispatch it with `fp32_provider=cpu`, `fp16_provider=cuda`, and +`int8_provider=cpu` so the evidence matches the checked-in release provider +policy. The merged accuracy report preserves `requested_provider` and +`actual_provider` inside each precision entry, and the release gate rejects +provider fallback or stale evidence generated from a different framework +commit. The artifact must provide `calibration.json`, `calibration_images/`, `accuracy.json`, and `accuracy.md` under `artifacts/vision_v8_quantized_eval/` before `build_onnx_release.py` runs. The COCO workflow copies the pinned calibration images into `calibration_images/` diff --git a/scripts/build_onnx_release.py b/scripts/build_onnx_release.py index 39e5acf..c839912 100644 --- a/scripts/build_onnx_release.py +++ b/scripts/build_onnx_release.py @@ -746,6 +746,7 @@ def build_release( quantization_thresholds, required_branches=[coco_report_branch(spec.branch) for spec in config.branches], expected_artifacts=expected_accuracy_artifacts, + expected_framework_commit=commit, ) if accuracy_failures: raise ReleaseError( diff --git a/scripts/check_onnx_quantized_artifacts.py b/scripts/check_onnx_quantized_artifacts.py index 7e40563..f76e747 100644 --- a/scripts/check_onnx_quantized_artifacts.py +++ b/scripts/check_onnx_quantized_artifacts.py @@ -272,15 +272,20 @@ def check_quantized_accuracy_report( *, required_branches: Sequence[str] = (), expected_artifacts: Mapping[str, Mapping[str, Mapping[str, str]]] | None = None, + expected_framework_commit: str | None = None, ) -> list[str]: """Compare a quantized candidate COCO report against its FP32 reference.""" if "branches" in report: - failures = _check_branch_accuracy_report( - report, - thresholds, - required_branches=required_branches, + failures = _check_framework_commit(report, expected_framework_commit) + failures.extend( + _check_branch_accuracy_report( + report, + thresholds, + required_branches=required_branches, + ) ) + failures.extend(_check_provider_contract(report, thresholds)) release_policy = _mapping(thresholds.get("release_policy")) if release_policy.get("require_artifact_bindings") is True: if expected_artifacts is None: @@ -291,7 +296,9 @@ def check_quantized_accuracy_report( reference = _mapping(report.get("reference")) candidate = _mapping(report.get("candidate")) - return _check_accuracy_pair(reference, candidate, thresholds) + failures = _check_framework_commit(report, expected_framework_commit) + failures.extend(_check_accuracy_pair(reference, candidate, thresholds)) + return failures def check_quantized_parity_report( @@ -396,6 +403,61 @@ def _check_branch_accuracy_report( return failures +def _check_framework_commit( + report: Mapping[str, Any], + expected_framework_commit: str | None, +) -> list[str]: + if expected_framework_commit is None: + return [] + actual = str(report.get("framework_commit", "")) + if actual != expected_framework_commit: + return [ + f"accuracy report framework_commit {actual or 'missing'}, " + f"expected {expected_framework_commit}" + ] + return [] + + +def _check_provider_contract( + report: Mapping[str, Any], + thresholds: Mapping[str, Any], +) -> list[str]: + release_policy = _mapping(thresholds.get("release_policy")) + provider_by_precision = _mapping(release_policy.get("provider_by_precision")) + if not provider_by_precision: + return [] + + branches = _mapping(report.get("branches")) + failures: list[str] = [] + for branch, branch_report in branches.items(): + branch_data = _mapping(branch_report) + for precision, expected_provider_value in provider_by_precision.items(): + expected_provider = str(expected_provider_value) + precision_report = _mapping(branch_data.get(str(precision))) + if not precision_report: + continue + environment = _mapping(precision_report.get("environment")) + if not environment: + failures.append(f"{branch} {precision} missing environment provider metadata") + continue + + requested = _provider_values(environment.get("requested_provider")) + actual = str(environment.get("actual_provider", "")) + if not requested: + failures.append(f"{branch} {precision} missing requested_provider") + elif requested[0] != expected_provider: + failures.append( + f"{branch} {precision} requested_provider {requested[0]}, " + f"expected {expected_provider}" + ) + if actual != expected_provider: + failures.append( + f"{branch} {precision} actual_provider {actual or 'missing'}, " + f"expected {expected_provider}" + ) + return failures + + def _required_precisions(thresholds: Mapping[str, Any]) -> tuple[str, ...]: release_policy = _mapping(thresholds.get("release_policy")) raw_precisions = release_policy.get("required_precisions", ()) @@ -436,6 +498,14 @@ def _candidate_precisions( return tuple(sorted(candidates)) +def _provider_values(value: object) -> tuple[str, ...]: + if isinstance(value, Sequence) and not isinstance(value, (str, bytes)): + return tuple(str(provider) for provider in value) + if value is None: + return () + return (str(value),) + + def _check_accuracy_pair( reference: Mapping[str, Any], candidate: Mapping[str, Any], diff --git a/scripts/merge_vision_v8_coco_reports.py b/scripts/merge_vision_v8_coco_reports.py index c616dfa..d0715d2 100644 --- a/scripts/merge_vision_v8_coco_reports.py +++ b/scripts/merge_vision_v8_coco_reports.py @@ -54,6 +54,7 @@ def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: for key in ( "checkpoint", "checkpoint_sha256", + "environment", "model", "metadata", "metadata_sha256", diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index 997abcc..c773734 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -207,6 +207,52 @@ def test_merge_onnx_reports_can_build_precision_nested_quantized_report(): ) +def test_merge_onnx_reports_preserves_environment_per_precision(): + fp32 = { + "backend": "onnx", + "precision": "fp32", + "framework_commit": "abc123", + "checkpoint": "o2m_fp32.onnx", + "checkpoint_sha256": "a" * 64, + "model": "o2m_fp32.onnx", + "metadata": "o2m.json", + "metadata_sha256": "b" * 64, + "dataset": {"name": "coco-2017", "image_ids": [1, 2]}, + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "protocol": {"seed": 0}, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.3}, + } + }, + } + fp16 = { + **fp32, + "precision": "fp16", + "checkpoint": "o2m_fp16.onnx", + "checkpoint_sha256": "c" * 64, + "environment": { + "requested_provider": ["CUDAExecutionProvider"], + "actual_provider": "CUDAExecutionProvider", + }, + "branches": { + "o2m-nms": { + "branch": "o2m-nms", + "metrics": {"map50_95": 0.198, "map50": 0.295}, + } + }, + } + + merged = merge_reports([fp32, fp16]) + + assert merged["branches"]["o2m-nms"]["fp32"]["environment"] == fp32["environment"] + assert merged["branches"]["o2m-nms"]["fp16"]["environment"] == fp16["environment"] + + def test_merge_onnx_reports_rejects_mismatched_protocol(): first = { "backend": "onnx", diff --git a/tests/test_onnx_quantization_accuracy_gate.py b/tests/test_onnx_quantization_accuracy_gate.py index 798a279..3aefeb5 100644 --- a/tests/test_onnx_quantization_accuracy_gate.py +++ b/tests/test_onnx_quantization_accuracy_gate.py @@ -378,3 +378,96 @@ def test_release_accuracy_gate_rejects_bogus_hashes_against_generated_artifacts( ) assert any("o2m-nms fp16 checkpoint_sha256" in failure for failure in failures) + + +def test_release_accuracy_gate_validates_provider_per_precision() -> None: + report = { + "reference_precision": "fp32", + "candidate_precisions": ["fp16", "int8"], + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["WrongExecutionProvider"], + "actual_provider": "WrongExecutionProvider", + }, + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + "int8": { + "precision": "int8", + "branch": "o2m-nms", + "environment": { + "requested_provider": ["CPUExecutionProvider"], + "actual_provider": "CPUExecutionProvider", + }, + "metrics": {"map50_95": 0.19, "map50": 0.31}, + }, + } + }, + } + thresholds = { + "release_policy": { + "required_precisions": ["fp32", "fp16", "int8"], + "provider_by_precision": { + "fp32": "CPUExecutionProvider", + "fp16": "CUDAExecutionProvider", + "int8": "CPUExecutionProvider", + }, + }, + "precisions": { + "fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}, + "int8": {"max_map50_95_drop": 0.02, "max_map50_drop": 0.03}, + }, + } + + failures = check_quantized_accuracy_report(report, thresholds) + + assert failures == [ + "o2m-nms fp16 requested_provider WrongExecutionProvider, expected CUDAExecutionProvider", + "o2m-nms fp16 actual_provider WrongExecutionProvider, expected CUDAExecutionProvider", + ] + + +def test_release_accuracy_gate_binds_evidence_to_framework_commit() -> None: + report = { + "framework_commit": "old", + "reference_precision": "fp32", + "candidate_precisions": ["fp16"], + "branches": { + "o2m-nms": { + "fp32": { + "precision": "fp32", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.2, "map50": 0.32}, + }, + "fp16": { + "precision": "fp16", + "branch": "o2m-nms", + "metrics": {"map50_95": 0.199, "map50": 0.319}, + }, + } + }, + } + thresholds = { + "release_policy": {"required_precisions": ["fp32", "fp16"]}, + "precisions": {"fp16": {"max_map50_95_drop": 0.005, "max_map50_drop": 0.01}}, + } + + failures = check_quantized_accuracy_report( + report, + thresholds, + expected_framework_commit="new", + ) + + assert failures == ["accuracy report framework_commit old, expected new"] diff --git a/tests/test_onnx_release.py b/tests/test_onnx_release.py index 322c52a..a31a9d7 100644 --- a/tests/test_onnx_release.py +++ b/tests/test_onnx_release.py @@ -135,6 +135,13 @@ def test_release_workflows_share_quantized_evidence_artifact_contract() -> None: assert "onnxruntime-gpu" in release_workflow assert "runs-on: ${{ inputs.runner || 'self-hosted' }}" in release_workflow assert "name: vision-v8-quantized-release-inputs" in coco_workflow + assert "fp32_provider:" in coco_workflow + assert "fp16_provider:" in coco_workflow + assert "int8_provider:" in coco_workflow + assert '--provider "${{ inputs.fp32_provider }}"' in coco_workflow + assert '--provider "${{ inputs.fp16_provider }}"' in coco_workflow + assert '--provider "${{ inputs.int8_provider }}"' in coco_workflow + assert "onnxruntime-gpu" in coco_workflow assert "calibration.json" in coco_workflow assert "calibration_images/**" in coco_workflow assert "accuracy.json" in coco_workflow