From 8bd3ce8b30ca6b076390059aa1aac5ae27acbae7 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Mon, 31 Aug 2026 15:37:20 +0500 Subject: [PATCH 1/4] Add Vision v8 COCO accuracy gates --- .github/workflows/vision-v8-coco-accuracy.yml | 179 +++++++++++++ complexity/deploy/onnx_detector/pipeline.py | 12 +- complexity/deploy/onnx_detector/session.py | 8 + configs/vision_v8_coco_accuracy_gate.json | 53 ++++ docs/index.md | 1 + docs/vision-v8-coco-accuracy-gates.md | 114 ++++++++ scripts/check_vision_v8_coco_report.py | 246 +++++++++++++++++ scripts/evaluate_onnx_coco.py | 252 ++++++++++++++++++ scripts/evaluate_tr_hash_coco.py | 167 ++++++++++++ tests/test_coco_release_evaluation.py | 68 +++++ tests/test_onnx_detector_core.py | 14 + tests/test_vision_v8_coco_accuracy_gate.py | 109 ++++++++ 12 files changed, 1222 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/vision-v8-coco-accuracy.yml create mode 100644 configs/vision_v8_coco_accuracy_gate.json create mode 100644 docs/vision-v8-coco-accuracy-gates.md create mode 100644 scripts/check_vision_v8_coco_report.py create mode 100644 scripts/evaluate_onnx_coco.py create mode 100644 tests/test_vision_v8_coco_accuracy_gate.py diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml new file mode 100644 index 00000000..37d1f264 --- /dev/null +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -0,0 +1,179 @@ +name: Vision v8 COCO accuracy + +on: + pull_request: + paths: + - ".github/workflows/vision-v8-coco-accuracy.yml" + - "configs/vision_v8_coco_accuracy_gate.json" + - "docs/vision-v8-coco-accuracy-gates.md" + - "docs/index.md" + - "scripts/check_vision_v8_coco_report.py" + - "scripts/evaluate_tr_hash_coco.py" + - "scripts/evaluate_onnx_coco.py" + - "complexity/deploy/onnx_detector/**" + - "tests/test_coco_release_evaluation.py" + - "tests/test_onnx_detector_core.py" + - "tests/test_vision_v8_coco_accuracy_gate.py" + push: + branches: [main] + paths: + - ".github/workflows/vision-v8-coco-accuracy.yml" + - "configs/vision_v8_coco_accuracy_gate.json" + - "docs/vision-v8-coco-accuracy-gates.md" + - "docs/index.md" + - "scripts/check_vision_v8_coco_report.py" + - "scripts/evaluate_tr_hash_coco.py" + - "scripts/evaluate_onnx_coco.py" + - "complexity/deploy/onnx_detector/**" + - "tests/test_coco_release_evaluation.py" + - "tests/test_onnx_detector_core.py" + - "tests/test_vision_v8_coco_accuracy_gate.py" + workflow_dispatch: + inputs: + checkpoint: + description: "Local path to the Vision v8 checkpoint on the runner" + required: false + onnx_model: + description: "Local path to the Vision v8 ONNX model on the runner" + required: false + onnx_metadata: + description: "Local path to the Vision v8 ONNX metadata sidecar on the runner" + required: false + annotations: + description: "Local path to instances_val2017.json on the runner" + required: true + images: + description: "Local path to COCO val2017 images on the runner" + required: true + runner: + description: "Runner label with dataset/checkpoint access" + required: true + default: "self-hosted" + branch: + description: "Detector branch to evaluate" + required: true + default: "both" + type: choice + options: + - both + - o2m-nms + - nms-free + backend: + description: "Evaluator backend" + required: true + default: "pytorch" + type: choice + options: + - pytorch + - onnx + provider: + description: "ONNX Runtime provider alias for backend=onnx" + required: true + default: "cpu" + release_tag: + description: "Optional existing GitHub Release tag to receive gated report assets" + required: false + device: + description: "PyTorch device" + required: true + default: "cuda" + precision: + description: "Evaluation precision" + required: true + default: "bf16" + type: choice + options: + - bf16 + - fp32 + +permissions: + contents: read + +jobs: + fixture-gate: + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + cache: pip + - name: Install CPU test environment + run: | + python -m pip install --upgrade pip + python -m pip install torch --index-url https://download.pytorch.org/whl/cpu + python -m pip install -e ".[dev,detection]" + - name: Lint COCO accuracy gate + run: >- + ruff check + scripts/check_vision_v8_coco_report.py + scripts/evaluate_onnx_coco.py + scripts/evaluate_tr_hash_coco.py + tests/test_coco_release_evaluation.py + tests/test_onnx_detector_core.py + tests/test_vision_v8_coco_accuracy_gate.py + - name: Test report validation and release metadata helpers + run: >- + pytest -q + tests/test_coco_release_evaluation.py + tests/test_onnx_detector_core.py + tests/test_vision_v8_coco_accuracy_gate.py + + full-coco-eval: + if: github.event_name == 'workflow_dispatch' + runs-on: ${{ inputs.runner }} + timeout-minutes: 360 + permissions: + contents: write + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + cache: pip + - name: Install evaluation environment + run: | + python -m pip install --upgrade pip + python -m pip install -e ".[detection,export]" + - name: Run deterministic COCO evaluation twice + run: | + if [ "${{ inputs.backend }}" = "pytorch" ]; then + test -n "${{ inputs.checkpoint }}" + python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 + python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 + else + test -n "${{ inputs.onnx_model }}" + test -n "${{ inputs.onnx_metadata }}" + onnx_branch="${{ inputs.branch }}" + if [ "$onnx_branch" = "both" ]; then + onnx_branch="auto" + fi + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 + fi + - name: Gate full COCO report + run: >- + python scripts/check_vision_v8_coco_report.py + artifacts/vision_v8_coco_eval/run_a/evaluation.json + --repeat-report artifacts/vision_v8_coco_eval/run_b/evaluation.json + --config configs/vision_v8_coco_accuracy_gate.json + - uses: actions/upload-artifact@v4 + with: + name: vision-v8-coco-accuracy-report + path: | + artifacts/vision_v8_coco_eval/run_a/evaluation.json + artifacts/vision_v8_coco_eval/run_a/evaluation.md + artifacts/vision_v8_coco_eval/run_b/evaluation.json + artifacts/vision_v8_coco_eval/run_b/evaluation.md + - name: Upload gated reports to GitHub Release + if: inputs.release_tag != '' + env: + GH_TOKEN: ${{ github.token }} + run: >- + gh release upload "${{ inputs.release_tag }}" + artifacts/vision_v8_coco_eval/run_a/evaluation.json#vision-v8-coco-evaluation.json + artifacts/vision_v8_coco_eval/run_a/evaluation.md#vision-v8-coco-evaluation.md + artifacts/vision_v8_coco_eval/run_b/evaluation.json#vision-v8-coco-evaluation-repeat.json + artifacts/vision_v8_coco_eval/run_b/evaluation.md#vision-v8-coco-evaluation-repeat.md + --clobber diff --git a/complexity/deploy/onnx_detector/pipeline.py b/complexity/deploy/onnx_detector/pipeline.py index d94e3887..ac844841 100644 --- a/complexity/deploy/onnx_detector/pipeline.py +++ b/complexity/deploy/onnx_detector/pipeline.py @@ -100,9 +100,19 @@ def postprocess_single_image( def create_session( model_path: str | Path, providers: Sequence[str] = ("CPUExecutionProvider",), + *, + warmup_runs: int = 1, + intra_op_num_threads: int | None = None, + inter_op_num_threads: int | None = None, ) -> OnnxDetectorSession: return OnnxDetectorSession( - OrtSessionConfig(Path(model_path), providers=tuple(providers)) + OrtSessionConfig( + Path(model_path), + providers=tuple(providers), + warmup_runs=warmup_runs, + intra_op_num_threads=intra_op_num_threads, + inter_op_num_threads=inter_op_num_threads, + ) ) def _decode_and_postprocess( diff --git a/complexity/deploy/onnx_detector/session.py b/complexity/deploy/onnx_detector/session.py index 803acfb5..d34abb53 100644 --- a/complexity/deploy/onnx_detector/session.py +++ b/complexity/deploy/onnx_detector/session.py @@ -18,6 +18,8 @@ class OrtSessionConfig: model_path: Path | str providers: tuple[str, ...] = ("CPUExecutionProvider",) warmup_runs: int = 1 + intra_op_num_threads: int | None = None + inter_op_num_threads: int | None = None class OnnxDetectorSession: @@ -36,8 +38,14 @@ def open(self) -> "OnnxDetectorSession": # Prefer NVIDIA site-package DLLs over an unrelated PyTorch CUDA build. ort.preload_dlls(directory="") self._add_tensorrt_dll_directories() + session_options = ort.SessionOptions() + if self.config.intra_op_num_threads is not None: + session_options.intra_op_num_threads = self.config.intra_op_num_threads + if self.config.inter_op_num_threads is not None: + session_options.inter_op_num_threads = self.config.inter_op_num_threads self._session = ort.InferenceSession( str(self.config.model_path), + sess_options=session_options, providers=list(self.config.providers), ) return self diff --git a/configs/vision_v8_coco_accuracy_gate.json b/configs/vision_v8_coco_accuracy_gate.json new file mode 100644 index 00000000..e3fe91ad --- /dev/null +++ b/configs/vision_v8_coco_accuracy_gate.json @@ -0,0 +1,53 @@ +{ + "schema_version": 1, + "dataset": { + "name": "coco-2017", + "split": "val2017", + "required_image_count": 5000, + "annotations_sha256": null, + "image_list_sha256": null + }, + "determinism": { + "seed": 0, + "metric_tolerance": 1e-12, + "cpu_metric_tolerance": 1e-12, + "cuda_metric_tolerance": 1e-6, + "tensorrt_metric_tolerance": 1e-5 + }, + "branches": { + "o2m-nms": { + "baseline_source": "docs/tr-hash-object-detection.md independent reproduction", + "baseline_metrics": { + "map50_95": 0.200, + "map50": 0.325, + "ar_100": 0.379 + }, + "absolute_floors": { + "map50_95": 0.190, + "map50": 0.310, + "ar_100": 0.360 + }, + "max_regressions": { + "map50_95": 0.005, + "map50": 0.010, + "ar_100": 0.010 + } + }, + "nms-free": { + "baseline_source": "docs/tr-hash-object-detection.md independent reproduction", + "baseline_metrics": { + "map50_95": 0.096, + "map50": 0.140 + }, + "absolute_floors": { + "map50_95": 0.090, + "map50": 0.130 + }, + "max_regressions": { + "map50_95": 0.005, + "map50": 0.010 + } + } + } +} + diff --git a/docs/index.md b/docs/index.md index f4b19ac6..133d9db0 100644 --- a/docs/index.md +++ b/docs/index.md @@ -91,6 +91,7 @@ GQA + TR-MoE architecture. - [TR-Hash text-to-image](tr-hash-text-to-image.md) - [TR-Hash object detection and serving](tr-hash-object-detection.md) - [TR-Hash Vision ONNX deployment](onnx_deploy.md) +- [Vision v8 COCO accuracy gates](vision-v8-coco-accuracy-gates.md) - [Detector specialization and ablations](TR_HASH_DETECTOR_SPECIALIZATION.md) - [TR-Hash sensor fusion](tr_hash_sensor_fusion.md) - [Vision dependency stack](vision-dependency-stack.md) diff --git a/docs/vision-v8-coco-accuracy-gates.md b/docs/vision-v8-coco-accuracy-gates.md new file mode 100644 index 00000000..64d1ea08 --- /dev/null +++ b/docs/vision-v8-coco-accuracy-gates.md @@ -0,0 +1,114 @@ +# Vision v8 COCO Accuracy Gates + +Vision v8 checkpoint publication must be gated by an end-to-end COCO accuracy +report, not only raw-logit ONNX parity. The gate validates that the checkpoint +is still useful as a detector and that the report was produced from pinned, +reproducible inputs. + +## Evaluation Contract + +Release reports target COCO 2017 `val2017` with all 5,000 validation images. +Reports must record both the annotation JSON SHA-256 and a sorted image-list +manifest SHA-256. The dataset name, split, image count, and hashes are checked +before any metric threshold is trusted. + +Both branches are evaluated: + +- `o2m-nms`: decode, confidence prefilter, class-aware NMS. +- `nms-free`: decode, confidence prefilter, top detections without NMS. + +Both backends must use the shared deployment preprocessing and decoded-output +contract: + +- RGB conversion; +- aspect-preserving letterbox with `(114, 114, 114)` fill; +- `[-1, 1]` normalization; +- DFL LTRB box decode; +- original-pixel box restoration. + +## Determinism + +Evaluation commands must seed `random`, `numpy`, and `torch`, set deterministic +PyTorch flags where supported, use a stable sorted image iteration order, and +fix ONNX Runtime intra-op and inter-op thread counts for CPU evaluation. CI +fixtures should run the same evaluation twice and diff the JSON metrics. CPU +reports should be identical within `1e-12`; CUDA and TensorRT may use wider +documented tolerances because provider kernels can change floating-point +accumulation order. + +## Report Metadata + +JSON reports must include: + +- framework commit; +- checkpoint revision or artifact hash; +- backend (`pytorch` or `onnx`); +- requested and actual provider for ONNX Runtime; +- Python, OS, PyTorch, ONNX Runtime, CUDA, driver, and TensorRT versions when + available; +- evaluated image count; +- confidence threshold, NMS IoU threshold, and max detections; +- AP, AP50, AP75, APs, APm, APl, AR100, precision, and recall. + +## Gate Policy + +The gate uses `configs/vision_v8_coco_accuracy_gate.json` and fails reports for +two separate metric reasons: + +- absolute floor failure: the metric is below the minimum acceptable value; +- baseline regression failure: the metric dropped more than the configured + delta from the known-good baseline. + +Reports also fail if required metadata or dataset hashes are missing. Gate +configuration must be changed explicitly in source review; CI must not relax +tolerances or thresholds at runtime. + +The `Vision v8 COCO accuracy` workflow runs lightweight gate tests on pull +requests. Its manual full-COCO job runs the selected evaluator twice with the +same seed, rejects malformed/regressed/non-deterministic reports, uploads the +JSON and Markdown reports as workflow artifacts, and can attach those reports to +an existing GitHub Release when `release_tag` is provided. + +## Reproduction Commands + +Native PyTorch full COCO evaluation: + +```bash +python scripts/evaluate_tr_hash_coco.py \ + models/TR-HASH-Vision-v8-2M-COCO-SFT \ + --annotations artifacts/COCO/annotations/instances_val2017.json \ + --images artifacts/COCO/images/val2017 \ + --output artifacts/vision_v8_coco_eval/pytorch \ + --branch both \ + --device cuda \ + --precision bf16 +``` + +ONNX Runtime full COCO evaluation for one exported branch: + +```bash +python scripts/evaluate_onnx_coco.py \ + --model artifacts/onnx/tr_hash_v8_o2m.onnx \ + --metadata artifacts/onnx/tr_hash_v8_o2m.json \ + --annotations artifacts/COCO/annotations/instances_val2017.json \ + --images artifacts/COCO/images/val2017 \ + --output artifacts/vision_v8_coco_eval/onnx_o2m \ + --branch o2m-nms \ + --provider cuda +``` + +Each ONNX sidecar describes one branch. To evaluate both ONNX branches, run the +command once for the O2M export and once for the NMS-free export, then gate both +reports. + +Report gate: + +```bash +python scripts/check_vision_v8_coco_report.py \ + artifacts/vision_v8_coco_eval/pytorch/evaluation.json \ + --config configs/vision_v8_coco_accuracy_gate.json +``` + +The ONNX Runtime evaluator emits the same report schema with `backend=onnx`, +`requested_provider`, and `actual_provider` populated from ONNX Runtime after +session creation. diff --git a/scripts/check_vision_v8_coco_report.py b/scripts/check_vision_v8_coco_report.py new file mode 100644 index 00000000..5398eceb --- /dev/null +++ b/scripts/check_vision_v8_coco_report.py @@ -0,0 +1,246 @@ +"""Validate Vision v8 COCO accuracy reports before checkpoint publication.""" + +from __future__ import annotations + +import argparse +import json +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Mapping + +REQUIRED_METRICS = ( + "map50_95", + "map50", + "map75", + "ap_small", + "ap_medium", + "ap_large", + "ar_100", +) +REQUIRED_ENVIRONMENT_KEYS = ( + "python", + "os", + "torch", + "onnxruntime", +) + + +@dataclass(frozen=True) +class GateFailure: + """One explicit accuracy-gate failure.""" + + kind: str + message: str + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("report", type=Path) + parser.add_argument( + "--repeat-report", + type=Path, + help="optional second same-seed report to compare for deterministic metrics", + ) + parser.add_argument( + "--config", + type=Path, + default=Path("configs/vision_v8_coco_accuracy_gate.json"), + ) + return parser.parse_args() + + +def load_json(path: Path) -> dict[str, Any]: + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except FileNotFoundError as error: + raise SystemExit(f"missing JSON file: {path}") from error + except json.JSONDecodeError as error: + raise SystemExit(f"malformed JSON file: {path}: {error}") from error + if not isinstance(payload, dict): + raise SystemExit(f"JSON root must be an object: {path}") + return payload + + +def check_report(report: Mapping[str, Any], config: Mapping[str, Any]) -> list[GateFailure]: + failures: list[GateFailure] = [] + branches = _branches(report) + if not branches: + failures.append(GateFailure("malformed_report", "report has no branch results")) + return failures + + for branch, branch_report in branches.items(): + branch_config = _mapping(config.get("branches", {})).get(branch) + if not isinstance(branch_config, Mapping): + failures.append( + GateFailure("malformed_report", f"branch {branch!r} has no gate config") + ) + continue + failures.extend(_check_metadata(report, branch, branch_report, config)) + failures.extend(_check_metrics(branch, branch_report, branch_config)) + return failures + + +def compare_repeated_reports( + first: Mapping[str, Any], + second: Mapping[str, Any], + tolerance: float, +) -> list[GateFailure]: + """Compare repeated same-seed evaluation reports for deterministic metrics.""" + + failures: list[GateFailure] = [] + first_branches = _branches(first) + second_branches = _branches(second) + if first_branches.keys() != second_branches.keys(): + return [ + GateFailure( + "determinism", + "repeated reports contain different branch sets", + ) + ] + for branch, first_branch in first_branches.items(): + first_metrics = _mapping(first_branch.get("metrics", first_branch)) + second_metrics = _mapping(second_branches[branch].get("metrics", second_branches[branch])) + for metric in REQUIRED_METRICS: + if metric not in first_metrics or metric not in second_metrics: + continue + delta = abs(float(first_metrics[metric]) - float(second_metrics[metric])) + if delta > tolerance: + failures.append( + GateFailure( + "determinism", + f"{branch} {metric} changed by {delta:.12f}, tolerance {tolerance:.12f}", + ) + ) + return failures + + +def _branches(report: Mapping[str, Any]) -> dict[str, Mapping[str, Any]]: + if isinstance(report.get("branches"), Mapping): + return { + str(branch): _mapping(branch_report) + for branch, branch_report in _mapping(report["branches"]).items() + } + branch = report.get("branch") + if branch is not None: + return {str(branch): report} + return {} + + +def _check_metadata( + report: Mapping[str, Any], + branch: str, + branch_report: Mapping[str, Any], + config: Mapping[str, Any], +) -> list[GateFailure]: + failures: list[GateFailure] = [] + dataset = _mapping(report.get("dataset", branch_report.get("dataset", {}))) + expected_dataset = _mapping(config.get("dataset", {})) + if dataset.get("name") != expected_dataset.get("name"): + failures.append(GateFailure("malformed_report", "dataset name mismatch")) + if dataset.get("split") != expected_dataset.get("split"): + failures.append(GateFailure("malformed_report", "dataset split mismatch")) + + expected_count = expected_dataset.get("required_image_count") + actual_count = dataset.get("evaluated_images", dataset.get("images")) + if expected_count is not None and actual_count != expected_count: + failures.append( + GateFailure( + "malformed_report", + f"{branch} evaluated image count mismatch: expected {expected_count}, got {actual_count}", + ) + ) + + for key in ("annotations_sha256", "image_list_sha256"): + expected_hash = expected_dataset.get(key) + actual_hash = dataset.get(key) + if expected_hash is None and not actual_hash: + failures.append(GateFailure("malformed_report", f"missing dataset hash: {key}")) + elif expected_hash is not None and actual_hash != expected_hash: + failures.append(GateFailure("malformed_report", f"dataset hash mismatch: {key}")) + + environment = _mapping(report.get("environment", branch_report.get("environment", {}))) + for key in REQUIRED_ENVIRONMENT_KEYS: + if key not in environment: + failures.append(GateFailure("malformed_report", f"missing environment.{key}")) + if report.get("framework_commit") is None and branch_report.get("framework_commit") is None: + failures.append(GateFailure("malformed_report", "missing framework_commit")) + return failures + + +def _check_metrics( + branch: str, + branch_report: Mapping[str, Any], + branch_config: Mapping[str, Any], +) -> list[GateFailure]: + failures: list[GateFailure] = [] + metrics = _mapping(branch_report.get("metrics", branch_report)) + for metric in REQUIRED_METRICS: + if metric not in metrics: + failures.append(GateFailure("malformed_report", f"{branch} missing metric {metric}")) + + floors = _mapping(branch_config.get("absolute_floors", {})) + for metric, floor in floors.items(): + value = _metric(metrics, metric) + if value is None: + continue + if value < float(floor): + failures.append( + GateFailure( + "absolute_floor", + f"{branch} {metric}={value:.6f} below floor {float(floor):.6f}", + ) + ) + + baselines = _mapping(branch_config.get("baseline_metrics", {})) + max_regressions = _mapping(branch_config.get("max_regressions", {})) + for metric, baseline in baselines.items(): + value = _metric(metrics, metric) + if value is None: + continue + allowed_drop = float(max_regressions.get(metric, 0.0)) + minimum = float(baseline) - allowed_drop + if value < minimum: + failures.append( + GateFailure( + "baseline_regression", + ( + f"{branch} {metric}={value:.6f} regressed beyond baseline " + f"{float(baseline):.6f} minus allowed drop {allowed_drop:.6f}" + ), + ) + ) + return failures + + +def _metric(metrics: Mapping[str, Any], name: str) -> float | None: + value = metrics.get(name) + if value is None: + return None + return float(value) + + +def _mapping(value: object) -> Mapping[str, Any]: + return value if isinstance(value, Mapping) else {} + + +def main() -> None: + args = parse_args() + report = load_json(args.report) + config = load_json(args.config) + failures = check_report(report, config) + if args.repeat_report is not None: + tolerance = float( + _mapping(config.get("determinism", {})).get("metric_tolerance", 0.0) + ) + failures.extend( + compare_repeated_reports(report, load_json(args.repeat_report), tolerance) + ) + if failures: + for failure in failures: + print(f"{failure.kind}: {failure.message}") + raise SystemExit(1) + print("Vision v8 COCO accuracy report passed") + + +if __name__ == "__main__": + main() diff --git a/scripts/evaluate_onnx_coco.py b/scripts/evaluate_onnx_coco.py new file mode 100644 index 00000000..7e709eeb --- /dev/null +++ b/scripts/evaluate_onnx_coco.py @@ -0,0 +1,252 @@ +"""Evaluate a TR-Hash Vision v8 ONNX detector export on COCO. + +The evaluator uses the deployment pipeline metadata sidecar to choose the +correct branch contract: + +- O2M exports run decode, confidence filtering, and class-aware NMS. +- NMS-free exports run decode and confidence filtering without NMS. +""" + +from __future__ import annotations + +import argparse +import json +import statistics +from dataclasses import replace +from pathlib import Path +from typing import Any + +import torch + +from complexity.deploy.onnx_detector import OnnxDetectorPipeline +from complexity.deploy.onnx_detector.metadata import load_metadata +from complexity.generative.detection.coco_evaluation import evaluate_coco_predictions +from complexity.generative.detection.hub import COCO_CLASS_NAMES +from scripts.evaluate_tr_hash_coco import ( + _branch_contract, + _checkpoint_sha256, + _configure_determinism, + _environment, + _framework_commit, + _image_list_sha256, + _percentile, + _sha256_file, + _write_markdown_report, +) +from scripts.onnx_detect import provider_names + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--metadata", type=Path, required=True) + parser.add_argument("--annotations", type=Path, required=True) + parser.add_argument("--images", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument( + "--branch", + choices=("auto", "o2m-nms", "nms-free"), + default="auto", + help="expected branch; auto trusts the metadata sidecar", + ) + parser.add_argument( + "--provider", + action="append", + default=None, + help="Provider alias or ORT provider name. May be repeated or comma-separated.", + ) + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--confidence", type=float, default=None) + parser.add_argument("--nms-iou", type=float, default=None) + parser.add_argument("--max-detections", type=int, default=None) + parser.add_argument("--ort-intra-op-threads", type=int, default=1) + parser.add_argument("--ort-inter-op-threads", type=int, default=1) + parser.add_argument( + "--eval-backend", + choices=("auto", "pycocotools", "faster"), + default="auto", + ) + parser.add_argument( + "--limit", + type=int, + default=0, + help="non-release smoke-test limit; zero evaluates all val2017 images", + ) + return parser.parse_args() + + +def _branch_name(metadata_branch: str) -> str: + return "o2m-nms" if metadata_branch == "o2m" else metadata_branch + + +def _xyxy_to_xywh(values: tuple[float, float, float, float]) -> list[float]: + x1, y1, x2, y2 = values + return [x1, y1, x2 - x1, y2 - y1] + + +def _latency_summary_ms(values: list[float]) -> dict[str, float]: + return { + "mean_ms": statistics.fmean(values) if values else 0.0, + "p50_ms": _percentile(values, 0.50), + "p95_ms": _percentile(values, 0.95), + "p99_ms": _percentile(values, 0.99), + "measured_ms": sum(values), + } + + +def _run_onnx_branch( + pipeline: OnnxDetectorPipeline, + coco: Any, + image_root: Path, + image_ids: list[int], + category_ids: list[int], +) -> tuple[list[dict[str, Any]], dict[str, Any]]: + predictions: list[dict[str, Any]] = [] + preprocess_ms: list[float] = [] + inference_ms: list[float] = [] + postprocess_ms: list[float] = [] + for image_id in image_ids: + record = coco.imgs[image_id] + result = pipeline.predict(image_root / record["file_name"]) + preprocess_ms.append(result.timing.preprocess_ms) + inference_ms.append(result.timing.inference_ms) + postprocess_ms.append(result.timing.postprocess_ms) + for detection in result.detections: + class_id = int(detection.class_id) + if not 0 <= class_id < len(category_ids): + raise ValueError(f"prediction class ID outside COCO mapping: {class_id}") + predictions.append( + { + "image_id": int(image_id), + "category_id": int(category_ids[class_id]), + "bbox": _xyxy_to_xywh(detection.box_pixel), + "score": float(detection.score), + } + ) + return predictions, { + "preprocess": _latency_summary_ms(preprocess_ms), + "inference": _latency_summary_ms(inference_ms), + "postprocess": _latency_summary_ms(postprocess_ms), + "images_per_second": ( + len(image_ids) / (sum(inference_ms) / 1000.0) if sum(inference_ms) else 0.0 + ), + } + + +def main() -> None: + args = parse_args() + if args.limit < 0: + raise ValueError("limit must be non-negative") + _configure_determinism(args.seed) + + from pycocotools.coco import COCO + + metadata = load_metadata(args.metadata) + if args.confidence is not None: + metadata = replace(metadata, confidence_threshold=args.confidence) + if args.nms_iou is not None: + metadata = replace(metadata, iou_threshold=args.nms_iou) + if args.max_detections is not None: + metadata = replace(metadata, max_detections=args.max_detections) + + branch = _branch_name(metadata.branch) + if args.branch != "auto" and args.branch != branch: + raise ValueError(f"metadata branch is {branch!r}, not requested {args.branch!r}") + + coco = COCO(str(args.annotations)) + categories = sorted(coco.loadCats(coco.getCatIds()), key=lambda item: item["id"]) + category_names = tuple(str(category["name"]) for category in categories) + if category_names != COCO_CLASS_NAMES: + raise ValueError("COCO category order does not match the detector class contract") + if metadata.class_names is not None and tuple(metadata.class_names) != COCO_CLASS_NAMES: + raise ValueError("metadata class names do not match the COCO detector contract") + category_ids = [int(category["id"]) for category in categories] + image_ids = sorted(coco.getImgIds()) + if args.limit: + image_ids = image_ids[: args.limit] + if not args.limit and len(image_ids) != 5_000: + raise ValueError(f"release evaluation requires 5000 val2017 images, got {len(image_ids)}") + + args.output.mkdir(parents=True, exist_ok=True) + session = OnnxDetectorPipeline.create_session( + args.model, + providers=provider_names(args.provider), + intra_op_num_threads=args.ort_intra_op_threads, + inter_op_num_threads=args.ort_inter_op_threads, + ).open() + pipeline = OnnxDetectorPipeline(metadata=metadata, session=session) + predictions, timing = _run_onnx_branch( + pipeline, + coco, + args.images, + image_ids, + category_ids, + ) + prediction_path = args.output / f"predictions_{branch}.json" + prediction_path.write_text(json.dumps(predictions), encoding="utf-8") + + environment = _environment(torch.device("cuda" if torch.cuda.is_available() else "cpu")) + environment.update( + { + "requested_provider": provider_names(args.provider), + "actual_provider": session.provider_used, + "ort_intra_op_threads": args.ort_intra_op_threads, + "ort_inter_op_threads": args.ort_inter_op_threads, + } + ) + report: dict[str, Any] = { + "schema_version": 1, + "format_version": 1, + "backend": "onnx", + "framework_commit": _framework_commit(), + "checkpoint": str(args.model), + "checkpoint_sha256": _checkpoint_sha256(args.model), + "model": str(args.model), + "metadata": str(args.metadata), + "metadata_sha256": _sha256_file(args.metadata), + "dataset": { + "name": "coco-2017", + "split": "val2017", + "images": len(image_ids), + "evaluated_images": len(image_ids), + "annotations": str(args.annotations), + "annotations_sha256": _sha256_file(args.annotations), + "image_list_sha256": _image_list_sha256(coco, image_ids), + }, + "environment": environment, + "protocol": { + "image_size": metadata.image_size, + "seed": args.seed, + "confidence_prefilter": metadata.confidence_threshold, + "nms_iou": metadata.iou_threshold, + "max_detections": metadata.max_detections, + "release_eligible": not bool(args.limit), + }, + "branches": { + branch: { + "branch": branch, + "contract": _branch_contract(branch), + "predictions": str(prediction_path), + "detections": len(predictions), + "metrics": evaluate_coco_predictions( + args.annotations, + predictions, + image_ids, + backend=args.eval_backend, + max_detections=metadata.max_detections, + confidence_threshold=metadata.confidence_threshold, + ), + "timing": timing, + } + }, + } + (args.output / "evaluation.json").write_text( + json.dumps(report, indent=2) + "\n", + encoding="utf-8", + ) + _write_markdown_report(report, args.output / "evaluation.md") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/evaluate_tr_hash_coco.py b/scripts/evaluate_tr_hash_coco.py index 5a0de971..63f20424 100644 --- a/scripts/evaluate_tr_hash_coco.py +++ b/scripts/evaluate_tr_hash_coco.py @@ -8,13 +8,19 @@ from __future__ import annotations import argparse +import hashlib import json +import platform +import random import statistics +import subprocess +import sys import time from contextlib import nullcontext from pathlib import Path from typing import Any, Iterable +import numpy as np import torch from PIL import Image @@ -45,6 +51,7 @@ def parse_args() -> argparse.Namespace: parser.add_argument("--precision", choices=("fp32", "bf16"), default="bf16") parser.add_argument("--batch-size", type=int, default=16) parser.add_argument("--warmup", type=int, default=10) + parser.add_argument("--seed", type=int, default=0) parser.add_argument("--confidence", type=float, default=0.001) parser.add_argument("--nms-iou", type=float, default=0.5) parser.add_argument( @@ -67,6 +74,84 @@ def parse_args() -> argparse.Namespace: return parser.parse_args() +def _sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _checkpoint_sha256(path: Path) -> str | None: + return _sha256_file(path) if path.is_file() else None + + +def _image_list_sha256(coco: Any, image_ids: list[int]) -> str: + digest = hashlib.sha256() + for image_id in image_ids: + record = coco.imgs[image_id] + line = ( + f"{int(image_id)}\t{record['file_name']}\t" + f"{int(record['width'])}\t{int(record['height'])}\n" + ) + digest.update(line.encode("utf-8")) + return digest.hexdigest() + + +def _framework_commit() -> str: + try: + return subprocess.check_output( + ["git", "rev-parse", "HEAD"], + stderr=subprocess.DEVNULL, + text=True, + ).strip() + except (OSError, subprocess.CalledProcessError): + return "unknown" + + +def _package_version(module_name: str) -> str | None: + try: + from importlib import metadata + + return metadata.version(module_name) + except metadata.PackageNotFoundError: + return None + + +def _environment(device: torch.device) -> dict[str, Any]: + cuda_available = torch.cuda.is_available() + cuda_device = None + cuda_capability = None + if cuda_available: + current = device.index if device.index is not None else torch.cuda.current_device() + cuda_device = torch.cuda.get_device_name(current) + cuda_capability = torch.cuda.get_device_capability(current) + return { + "python": sys.version.split()[0], + "os": platform.platform(), + "torch": torch.__version__, + "onnxruntime": _package_version("onnxruntime"), + "cuda_available": cuda_available, + "torch_cuda": torch.version.cuda, + "cuda_device": cuda_device, + "cuda_capability": list(cuda_capability) if cuda_capability is not None else None, + "tensorrt": _package_version("tensorrt"), + } + + +def _configure_determinism(seed: int) -> None: + random.seed(seed) + np.random.seed(seed) + torch.manual_seed(seed) + if torch.cuda.is_available(): + torch.cuda.manual_seed_all(seed) + torch.backends.cudnn.benchmark = False + try: + torch.use_deterministic_algorithms(True, warn_only=True) + except TypeError: + torch.use_deterministic_algorithms(True) + + def _synchronize(device: torch.device) -> None: if device.type == "cuda": torch.cuda.synchronize(device) @@ -114,6 +199,19 @@ def _branches_to_run(requested: str, *, has_nms_free: bool) -> tuple[str, ...]: return branches +def _branch_contract(branch: str) -> dict[str, Any]: + return { + "raw_output": "[batch, 34000, 148]", + "regression": "68 LTRB/DFL logits", + "classification": "80 quality-class logits", + "postprocess": ( + "decode + confidence filtering + class-aware NMS" + if branch == "o2m-nms" + else "decode + confidence filtering, no NMS" + ), + } + + def _run_branch( model: torch.nn.Module, coco: Any, @@ -212,10 +310,67 @@ def precision_context(): return results, _timing_summary(batch_times, len(image_ids)) +def _write_markdown_report(report: dict[str, Any], path: Path) -> None: + lines = [ + "# Vision v8 COCO Accuracy Report", + "", + f"- Backend: `{report['backend']}`", + f"- Framework commit: `{report['framework_commit']}`", + f"- Checkpoint: `{report['checkpoint']}`", + f"- Checkpoint SHA-256: `{report['checkpoint_sha256'] or 'not recorded'}`", + ( + f"- Dataset: `{report['dataset']['name']}` `{report['dataset']['split']}` " + f"({report['dataset']['evaluated_images']} images)" + ), + f"- Annotation SHA-256: `{report['dataset']['annotations_sha256']}`", + f"- Image-list SHA-256: `{report['dataset']['image_list_sha256']}`", + "", + "## Branch Metrics", + "", + "| Branch | AP | AP50 | AP75 | APs | APm | APl | AR100 |", + "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |", + ] + for branch, branch_report in report["branches"].items(): + metrics = branch_report["metrics"] + lines.append( + "| " + + " | ".join( + [ + branch, + f"{float(metrics['map50_95']):.6f}", + f"{float(metrics['map50']):.6f}", + f"{float(metrics['map75']):.6f}", + f"{float(metrics['ap_small']):.6f}", + f"{float(metrics['ap_medium']):.6f}", + f"{float(metrics['ap_large']):.6f}", + f"{float(metrics['ar_100']):.6f}", + ] + ) + + " |" + ) + lines.extend( + [ + "", + "## Runtime", + "", + f"- Python: `{report['environment']['python']}`", + f"- OS: `{report['environment']['os']}`", + f"- PyTorch: `{report['environment']['torch']}`", + f"- ONNX Runtime: `{report['environment']['onnxruntime'] or 'not installed'}`", + f"- CUDA available: `{report['environment']['cuda_available']}`", + f"- CUDA runtime: `{report['environment']['torch_cuda'] or 'not available'}`", + f"- TensorRT: `{report['environment']['tensorrt'] or 'not installed'}`", + "", + ] + ) + path.write_text("\n".join(lines), encoding="utf-8") + + def main() -> None: args = parse_args() if args.batch_size <= 0 or args.warmup < 0 or args.max_detections <= 0: raise ValueError("batch size/max detections must be positive and warmup non-negative") + _configure_determinism(args.seed) provenance = read_detector_provenance(args.checkpoint) validate_native_random_init_provenance(provenance, dataset=NATIVE_COCO_DATASET) from pycocotools.coco import COCO @@ -241,20 +396,29 @@ def main() -> None: branches = _branches_to_run(args.branch, has_nms_free=model.one_to_one_head is not None) args.output.mkdir(parents=True, exist_ok=True) report: dict[str, Any] = { + "schema_version": 1, "format_version": 1, + "backend": "pytorch", + "framework_commit": _framework_commit(), "checkpoint": str(args.checkpoint), + "checkpoint_sha256": _checkpoint_sha256(args.checkpoint), "provenance": provenance, "dataset": { "name": NATIVE_COCO_DATASET, "split": "val2017", "images": len(image_ids), + "evaluated_images": len(image_ids), "annotations": str(args.annotations), + "annotations_sha256": _sha256_file(args.annotations), + "image_list_sha256": _image_list_sha256(coco, image_ids), }, + "environment": _environment(device), "protocol": { "image_size": model.config.image_size, "precision": args.precision, "batch_size": args.batch_size, "warmup_batches": args.warmup, + "seed": args.seed, "confidence_prefilter": args.confidence, "nms_iou": args.nms_iou, "max_detections": args.max_detections, @@ -289,6 +453,8 @@ def main() -> None: prediction_path = args.output / f"predictions_{branch}.json" prediction_path.write_text(json.dumps(predictions)) report["branches"][branch] = { + "branch": branch, + "contract": _branch_contract(branch), "predictions": str(prediction_path), "detections": len(predictions), "metrics": evaluate_coco_predictions( @@ -302,6 +468,7 @@ def main() -> None: "timing": timing, } (args.output / "evaluation.json").write_text(json.dumps(report, indent=2) + "\n") + _write_markdown_report(report, args.output / "evaluation.md") print(json.dumps(report, indent=2)) diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index 632275d5..d74fe86a 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -2,9 +2,12 @@ from scripts.evaluate_tr_hash_coco import ( BRANCHES, + _branch_contract, _branches_to_run, + _image_list_sha256, _percentile, _timing_summary, + _write_markdown_report, ) @@ -33,3 +36,68 @@ def test_timing_summary_uses_measured_batches_and_image_count(): def test_percentile_handles_empty_measurements(): assert _percentile([], 0.95) == 0.0 + + +def test_image_list_hash_uses_sorted_manifest_contract(): + class Coco: + imgs = { + 7: {"file_name": "000000000007.jpg", "width": 640, "height": 427}, + 3: {"file_name": "000000000003.jpg", "width": 500, "height": 375}, + } + + first = _image_list_sha256(Coco(), [3, 7]) + second = _image_list_sha256(Coco(), [3, 7]) + different_order = _image_list_sha256(Coco(), [7, 3]) + + assert first == second + assert first != different_order + + +def test_branch_contract_distinguishes_nms_requirements(): + assert "class-aware NMS" in _branch_contract("o2m-nms")["postprocess"] + assert "no NMS" in _branch_contract("nms-free")["postprocess"] + + +def test_markdown_report_contains_release_metrics(tmp_path): + report = { + "backend": "pytorch", + "framework_commit": "abc123", + "checkpoint": "checkpoint.pt", + "checkpoint_sha256": "hash", + "dataset": { + "name": "coco-2017", + "split": "val2017", + "evaluated_images": 5000, + "annotations_sha256": "annotations", + "image_list_sha256": "images", + }, + "environment": { + "python": "3.11", + "os": "linux", + "torch": "2.6.0", + "onnxruntime": "1.23.2", + "cuda_available": False, + "torch_cuda": None, + "tensorrt": None, + }, + "branches": { + "o2m-nms": { + "metrics": { + "map50_95": 0.2, + "map50": 0.3, + "map75": 0.1, + "ap_small": 0.01, + "ap_medium": 0.2, + "ap_large": 0.3, + "ar_100": 0.4, + } + } + }, + } + output = tmp_path / "evaluation.md" + + _write_markdown_report(report, output) + + text = output.read_text(encoding="utf-8") + assert "Vision v8 COCO Accuracy Report" in text + assert "| o2m-nms | 0.200000 | 0.300000" in text diff --git a/tests/test_onnx_detector_core.py b/tests/test_onnx_detector_core.py index 5dc1ee71..063e6316 100644 --- a/tests/test_onnx_detector_core.py +++ b/tests/test_onnx_detector_core.py @@ -67,6 +67,20 @@ def test_cuda_dll_preload_is_limited_to_nvidia_providers() -> None: assert not _needs_cuda_dlls(("DmlExecutionProvider",)) +def test_pipeline_session_factory_forwards_deterministic_thread_settings() -> None: + session = OnnxDetectorPipeline.create_session( + Path("model.onnx"), + providers=("CPUExecutionProvider",), + warmup_runs=0, + intra_op_num_threads=1, + inter_op_num_threads=1, + ) + + assert session.config.warmup_runs == 0 + assert session.config.intra_op_num_threads == 1 + assert session.config.inter_op_num_threads == 1 + + def test_dfl_decode_matches_hand_computed_expectation() -> None: metadata = _metadata() geometry = generate_grid_geometry(metadata.image_size, metadata.grid_sizes) diff --git a/tests/test_vision_v8_coco_accuracy_gate.py b/tests/test_vision_v8_coco_accuracy_gate.py new file mode 100644 index 00000000..4a33311a --- /dev/null +++ b/tests/test_vision_v8_coco_accuracy_gate.py @@ -0,0 +1,109 @@ +from scripts.check_vision_v8_coco_report import ( + check_report, + compare_repeated_reports, +) + + +def _config() -> dict: + return { + "dataset": { + "name": "coco-2017", + "split": "val2017", + "required_image_count": 2, + "annotations_sha256": "annotations", + "image_list_sha256": "image-list", + }, + "branches": { + "o2m-nms": { + "baseline_metrics": { + "map50_95": 0.200, + "map50": 0.325, + "ar_100": 0.379, + }, + "absolute_floors": { + "map50_95": 0.190, + "map50": 0.310, + "ar_100": 0.360, + }, + "max_regressions": { + "map50_95": 0.005, + "map50": 0.010, + "ar_100": 0.010, + }, + } + }, + } + + +def _metrics(**overrides) -> dict: + metrics = { + "map50_95": 0.200, + "map50": 0.325, + "map75": 0.190, + "ap_small": 0.050, + "ap_medium": 0.190, + "ap_large": 0.310, + "ar_100": 0.379, + } + metrics.update(overrides) + return metrics + + +def _report(metrics: dict | None = None) -> dict: + return { + "schema_version": 1, + "framework_commit": "abc123", + "dataset": { + "name": "coco-2017", + "split": "val2017", + "evaluated_images": 2, + "annotations_sha256": "annotations", + "image_list_sha256": "image-list", + }, + "environment": { + "python": "3.11", + "os": "linux", + "torch": "2.6.0", + "onnxruntime": "1.23.2", + }, + "branches": { + "o2m-nms": { + "metrics": metrics if metrics is not None else _metrics(), + } + }, + } + + +def test_accuracy_gate_accepts_valid_report() -> None: + assert check_report(_report(), _config()) == [] + + +def test_accuracy_gate_rejects_missing_dataset_hash() -> None: + report = _report() + del report["dataset"]["annotations_sha256"] + + failures = check_report(report, _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("annotations_sha256" in failure.message for failure in failures) + + +def test_accuracy_gate_separates_absolute_floor_from_baseline_regression() -> None: + floor_failures = check_report(_report(_metrics(map50_95=0.180)), _config()) + regression_failures = check_report(_report(_metrics(map50_95=0.194)), _config()) + + assert any(failure.kind == "absolute_floor" for failure in floor_failures) + assert any(failure.kind == "baseline_regression" for failure in floor_failures) + assert not any(failure.kind == "absolute_floor" for failure in regression_failures) + assert any(failure.kind == "baseline_regression" for failure in regression_failures) + + +def test_repeated_report_comparison_flags_metric_drift() -> None: + first = _report() + second = _report(_metrics(map50=0.3250002)) + + failures = compare_repeated_reports(first, second, tolerance=1e-7) + + assert len(failures) == 1 + assert failures[0].kind == "determinism" + assert "map50" in failures[0].message From bbac89aaab140f0a7e5734290c0ed29946402a3d Mon Sep 17 00:00:00 2001 From: Illiyin Date: Tue, 1 Sep 2026 08:56:18 +0500 Subject: [PATCH 2/4] Tighten Vision v8 COCO accuracy gate --- .github/workflows/vision-v8-coco-accuracy.yml | 35 +- configs/vision_v8_coco_accuracy_gate.json | 6 +- docs/vision-v8-coco-accuracy-gates.md | 26 +- scripts/check_vision_v8_coco_report.py | 300 ++++++++++++++++-- scripts/merge_vision_v8_coco_reports.py | 77 +++++ tests/test_coco_release_evaluation.py | 51 +++ tests/test_vision_v8_coco_accuracy_gate.py | 151 ++++++++- 7 files changed, 602 insertions(+), 44 deletions(-) create mode 100644 scripts/merge_vision_v8_coco_reports.py diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 37d1f264..03b28365 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -10,6 +10,7 @@ on: - "scripts/check_vision_v8_coco_report.py" - "scripts/evaluate_tr_hash_coco.py" - "scripts/evaluate_onnx_coco.py" + - "scripts/merge_vision_v8_coco_reports.py" - "complexity/deploy/onnx_detector/**" - "tests/test_coco_release_evaluation.py" - "tests/test_onnx_detector_core.py" @@ -24,6 +25,7 @@ on: - "scripts/check_vision_v8_coco_report.py" - "scripts/evaluate_tr_hash_coco.py" - "scripts/evaluate_onnx_coco.py" + - "scripts/merge_vision_v8_coco_reports.py" - "complexity/deploy/onnx_detector/**" - "tests/test_coco_release_evaluation.py" - "tests/test_onnx_detector_core.py" @@ -39,6 +41,18 @@ on: onnx_metadata: description: "Local path to the Vision v8 ONNX metadata sidecar on the runner" required: false + onnx_o2m_model: + description: "Local path to the O2M Vision v8 ONNX model on the runner" + required: false + onnx_o2m_metadata: + description: "Local path to the O2M Vision v8 ONNX metadata sidecar on the runner" + required: false + onnx_nms_free_model: + description: "Local path to the NMS-free Vision v8 ONNX model on the runner" + required: false + onnx_nms_free_metadata: + description: "Local path to the NMS-free Vision v8 ONNX metadata sidecar on the runner" + required: false annotations: description: "Local path to instances_val2017.json on the runner" required: true @@ -110,6 +124,7 @@ jobs: scripts/check_vision_v8_coco_report.py scripts/evaluate_onnx_coco.py scripts/evaluate_tr_hash_coco.py + scripts/merge_vision_v8_coco_reports.py tests/test_coco_release_evaluation.py tests/test_onnx_detector_core.py tests/test_vision_v8_coco_accuracy_gate.py @@ -143,14 +158,24 @@ jobs: python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 else - test -n "${{ inputs.onnx_model }}" - test -n "${{ inputs.onnx_metadata }}" onnx_branch="${{ inputs.branch }}" if [ "$onnx_branch" = "both" ]; then - onnx_branch="auto" + test -n "${{ inputs.onnx_o2m_model }}" + test -n "${{ inputs.onnx_o2m_metadata }}" + test -n "${{ inputs.onnx_nms_free_model }}" + test -n "${{ inputs.onnx_nms_free_metadata }}" + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b + else + test -n "${{ inputs.onnx_model }}" + test -n "${{ inputs.onnx_metadata }}" + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 fi - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 fi - name: Gate full COCO report run: >- diff --git a/configs/vision_v8_coco_accuracy_gate.json b/configs/vision_v8_coco_accuracy_gate.json index e3fe91ad..e1502a7a 100644 --- a/configs/vision_v8_coco_accuracy_gate.json +++ b/configs/vision_v8_coco_accuracy_gate.json @@ -4,8 +4,8 @@ "name": "coco-2017", "split": "val2017", "required_image_count": 5000, - "annotations_sha256": null, - "image_list_sha256": null + "annotations_sha256": "e8c7f7908f1d7278341fae127d0da654f102f11bd7b21d8aeefa635b8c810b6f", + "image_list_sha256": "0e508226935f3b7b96afd98661e82d6b53a02a6d4a4a7ed1dec777768b1d3315" }, "determinism": { "seed": 0, @@ -14,6 +14,7 @@ "cuda_metric_tolerance": 1e-6, "tensorrt_metric_tolerance": 1e-5 }, + "required_branches": ["o2m-nms", "nms-free"], "branches": { "o2m-nms": { "baseline_source": "docs/tr-hash-object-detection.md independent reproduction", @@ -50,4 +51,3 @@ } } } - diff --git a/docs/vision-v8-coco-accuracy-gates.md b/docs/vision-v8-coco-accuracy-gates.md index 64d1ea08..6e5b18b9 100644 --- a/docs/vision-v8-coco-accuracy-gates.md +++ b/docs/vision-v8-coco-accuracy-gates.md @@ -10,7 +10,11 @@ reproducible inputs. Release reports target COCO 2017 `val2017` with all 5,000 validation images. Reports must record both the annotation JSON SHA-256 and a sorted image-list manifest SHA-256. The dataset name, split, image count, and hashes are checked -before any metric threshold is trusted. +before any metric threshold is trusted. The release gate pins COCO val2017 to +annotation SHA-256 +`e8c7f7908f1d7278341fae127d0da654f102f11bd7b21d8aeefa635b8c810b6f` and +image-list SHA-256 +`0e508226935f3b7b96afd98661e82d6b53a02a6d4a4a7ed1dec777768b1d3315`. Both branches are evaluated: @@ -59,9 +63,12 @@ two separate metric reasons: - baseline regression failure: the metric dropped more than the configured delta from the known-good baseline. -Reports also fail if required metadata or dataset hashes are missing. Gate -configuration must be changed explicitly in source review; CI must not relax -tolerances or thresholds at runtime. +Reports also fail if required metadata or dataset hashes are missing, mismatched, +non-canonical, or if any gated metric is non-finite or outside `[0, 1]`. ONNX +reports must record the requested provider list and the actual provider selected +by ONNX Runtime; the actual provider must match the first requested provider, so +provider fallback fails the gate. Gate configuration must be changed explicitly +in source review; CI must not relax tolerances or thresholds at runtime. The `Vision v8 COCO accuracy` workflow runs lightweight gate tests on pull requests. Its manual full-COCO job runs the selected evaluator twice with the @@ -98,8 +105,15 @@ python scripts/evaluate_onnx_coco.py \ ``` Each ONNX sidecar describes one branch. To evaluate both ONNX branches, run the -command once for the O2M export and once for the NMS-free export, then gate both -reports. +command once for the O2M export and once for the NMS-free export, merge the +per-branch reports, then gate the combined report: + +```bash +python scripts/merge_vision_v8_coco_reports.py \ + artifacts/vision_v8_coco_eval/onnx_o2m/evaluation.json \ + artifacts/vision_v8_coco_eval/onnx_nms_free/evaluation.json \ + --output artifacts/vision_v8_coco_eval/onnx_both +``` Report gate: diff --git a/scripts/check_vision_v8_coco_report.py b/scripts/check_vision_v8_coco_report.py index 5398eceb..702de992 100644 --- a/scripts/check_vision_v8_coco_report.py +++ b/scripts/check_vision_v8_coco_report.py @@ -4,6 +4,8 @@ import argparse import json +import math +import string from dataclasses import dataclass from pathlib import Path from typing import Any, Mapping @@ -23,6 +25,9 @@ "torch", "onnxruntime", ) +REQUIRED_DATASET_HASHES = ("annotations_sha256", "image_list_sha256") +REQUIRED_PROTOCOL_KEYS = ("seed", "release_eligible") +VALID_BACKENDS = {"pytorch", "onnx"} @dataclass(frozen=True) @@ -61,13 +66,22 @@ def load_json(path: Path) -> dict[str, Any]: return payload -def check_report(report: Mapping[str, Any], config: Mapping[str, Any]) -> list[GateFailure]: - failures: list[GateFailure] = [] +def check_report( + report: Mapping[str, Any], config: Mapping[str, Any] +) -> list[GateFailure]: + failures: list[GateFailure] = _check_config(config) branches = _branches(report) if not branches: failures.append(GateFailure("malformed_report", "report has no branch results")) return failures + required_branches = _required_branches(config) + for branch in required_branches: + if branch not in branches: + failures.append( + GateFailure("malformed_report", f"missing required branch: {branch}") + ) + for branch, branch_report in branches.items(): branch_config = _mapping(config.get("branches", {})).get(branch) if not isinstance(branch_config, Mapping): @@ -83,11 +97,17 @@ def check_report(report: Mapping[str, Any], config: Mapping[str, Any]) -> list[G def compare_repeated_reports( first: Mapping[str, Any], second: Mapping[str, Any], - tolerance: float, + tolerance: float | None = None, + config: Mapping[str, Any] | None = None, ) -> list[GateFailure]: """Compare repeated same-seed evaluation reports for deterministic metrics.""" failures: list[GateFailure] = [] + tolerance = ( + determinism_tolerance(first, config) + if config is not None + else float(0.0 if tolerance is None else tolerance) + ) first_branches = _branches(first) second_branches = _branches(second) if first_branches.keys() != second_branches.keys(): @@ -99,21 +119,75 @@ def compare_repeated_reports( ] for branch, first_branch in first_branches.items(): first_metrics = _mapping(first_branch.get("metrics", first_branch)) - second_metrics = _mapping(second_branches[branch].get("metrics", second_branches[branch])) + second_metrics = _mapping( + second_branches[branch].get("metrics", second_branches[branch]) + ) for metric in REQUIRED_METRICS: if metric not in first_metrics or metric not in second_metrics: continue - delta = abs(float(first_metrics[metric]) - float(second_metrics[metric])) + first_value = _metric( + first_metrics, + metric, + branch=branch, + failures=failures, + kind="determinism", + ) + second_value = _metric( + second_metrics, + metric, + branch=branch, + failures=failures, + kind="determinism", + ) + if first_value is None or second_value is None: + continue + delta = abs(first_value - second_value) if delta > tolerance: failures.append( GateFailure( "determinism", - f"{branch} {metric} changed by {delta:.12f}, tolerance {tolerance:.12f}", + ( + f"{branch} {metric} changed by {delta:.12f}, " + f"tolerance {tolerance:.12f}" + ), ) ) return failures +def determinism_tolerance( + report: Mapping[str, Any], config: Mapping[str, Any] +) -> float: + """Select the same-seed metric tolerance for the report backend/provider.""" + + determinism = _mapping(config.get("determinism", {})) + default = _float_config(determinism.get("metric_tolerance", 0.0), default=0.0) + environment = _mapping(report.get("environment", {})) + + if report.get("backend") == "onnx": + provider = str(environment.get("actual_provider", "")).lower() + if "tensorrt" in provider: + return _float_config( + determinism.get("tensorrt_metric_tolerance", default), default=default + ) + if "cuda" in provider: + return _float_config( + determinism.get("cuda_metric_tolerance", default), default=default + ) + if "cpu" in provider: + return _float_config( + determinism.get("cpu_metric_tolerance", default), default=default + ) + + if environment.get("cuda_available") is True: + return _float_config( + determinism.get("cuda_metric_tolerance", default), default=default + ) + return _float_config( + determinism.get("cpu_metric_tolerance", default), default=default + ) + + def _branches(report: Mapping[str, Any]) -> dict[str, Mapping[str, Any]]: if isinstance(report.get("branches"), Mapping): return { @@ -146,24 +220,82 @@ def _check_metadata( failures.append( GateFailure( "malformed_report", - f"{branch} evaluated image count mismatch: expected {expected_count}, got {actual_count}", + ( + f"{branch} evaluated image count mismatch: " + f"expected {expected_count}, got {actual_count}" + ), ) ) - for key in ("annotations_sha256", "image_list_sha256"): + for key in REQUIRED_DATASET_HASHES: expected_hash = expected_dataset.get(key) actual_hash = dataset.get(key) - if expected_hash is None and not actual_hash: - failures.append(GateFailure("malformed_report", f"missing dataset hash: {key}")) - elif expected_hash is not None and actual_hash != expected_hash: - failures.append(GateFailure("malformed_report", f"dataset hash mismatch: {key}")) + if not _valid_sha256(actual_hash): + failures.append( + GateFailure( + "malformed_report", + f"missing or invalid dataset hash: {key}", + ) + ) + elif actual_hash != expected_hash: + failures.append( + GateFailure("malformed_report", f"dataset hash mismatch: {key}") + ) - environment = _mapping(report.get("environment", branch_report.get("environment", {}))) + environment = _mapping( + report.get("environment", branch_report.get("environment", {})) + ) for key in REQUIRED_ENVIRONMENT_KEYS: if key not in environment: - failures.append(GateFailure("malformed_report", f"missing environment.{key}")) - if report.get("framework_commit") is None and branch_report.get("framework_commit") is None: + failures.append( + GateFailure("malformed_report", f"missing environment.{key}") + ) + backend = report.get("backend", branch_report.get("backend")) + if backend not in VALID_BACKENDS: + failures.append(GateFailure("malformed_report", "missing or invalid backend")) + if backend == "onnx": + failures.extend(_check_onnx_provider(environment)) + for key in ("model", "metadata"): + if _field(report, branch_report, key) is None: + failures.append(GateFailure("malformed_report", f"missing {key}")) + metadata_hash = _field(report, branch_report, "metadata_sha256") + if not _valid_sha256(metadata_hash): + failures.append( + GateFailure("malformed_report", "missing or invalid metadata_sha256") + ) + + protocol = _mapping(report.get("protocol", branch_report.get("protocol", {}))) + for key in REQUIRED_PROTOCOL_KEYS: + if key not in protocol: + failures.append(GateFailure("malformed_report", f"missing protocol.{key}")) + expected_seed = _mapping(config.get("determinism", {})).get("seed") + if expected_seed is not None and protocol.get("seed") != expected_seed: + failures.append( + GateFailure( + "malformed_report", + ( + f"protocol.seed mismatch: expected {expected_seed}, " + f"got {protocol.get('seed')}" + ), + ) + ) + if protocol.get("release_eligible") is not True: + failures.append( + GateFailure("malformed_report", "protocol.release_eligible must be true") + ) + + if ( + report.get("framework_commit") is None + and branch_report.get("framework_commit") is None + ): failures.append(GateFailure("malformed_report", "missing framework_commit")) + checkpoint_hash = _field(report, branch_report, "checkpoint_sha256") + if _field(report, branch_report, "checkpoint") is None: + failures.append(GateFailure("malformed_report", "missing checkpoint")) + if not _valid_sha256(checkpoint_hash): + failures.append( + GateFailure("malformed_report", "missing or invalid checkpoint_sha256") + ) return failures @@ -174,13 +306,20 @@ def _check_metrics( ) -> list[GateFailure]: failures: list[GateFailure] = [] metrics = _mapping(branch_report.get("metrics", branch_report)) + metric_values: dict[str, float] = {} for metric in REQUIRED_METRICS: if metric not in metrics: - failures.append(GateFailure("malformed_report", f"{branch} missing metric {metric}")) + failures.append( + GateFailure("malformed_report", f"{branch} missing metric {metric}") + ) + continue + value = _metric(metrics, metric, branch=branch, failures=failures) + if value is not None: + metric_values[metric] = value floors = _mapping(branch_config.get("absolute_floors", {})) for metric, floor in floors.items(): - value = _metric(metrics, metric) + value = metric_values.get(metric) if value is None: continue if value < float(floor): @@ -194,7 +333,7 @@ def _check_metrics( baselines = _mapping(branch_config.get("baseline_metrics", {})) max_regressions = _mapping(branch_config.get("max_regressions", {})) for metric, baseline in baselines.items(): - value = _metric(metrics, metric) + value = metric_values.get(metric) if value is None: continue allowed_drop = float(max_regressions.get(metric, 0.0)) @@ -212,28 +351,141 @@ def _check_metrics( return failures -def _metric(metrics: Mapping[str, Any], name: str) -> float | None: +def _metric( + metrics: Mapping[str, Any], + name: str, + *, + branch: str, + failures: list[GateFailure], + kind: str = "malformed_report", +) -> float | None: value = metrics.get(name) if value is None: return None - return float(value) + if isinstance(value, bool): + failures.append(GateFailure(kind, f"{branch} {name} is not numeric")) + return None + try: + numeric = float(value) + except (TypeError, ValueError): + failures.append(GateFailure(kind, f"{branch} {name} is not numeric")) + return None + if not math.isfinite(numeric): + failures.append(GateFailure(kind, f"{branch} {name} is non-finite")) + return None + if numeric < 0.0 or numeric > 1.0: + failures.append( + GateFailure(kind, f"{branch} {name}={numeric:.6f} outside [0, 1]") + ) + return None + return numeric def _mapping(value: object) -> Mapping[str, Any]: return value if isinstance(value, Mapping) else {} +def _field( + report: Mapping[str, Any], branch_report: Mapping[str, Any], key: str +) -> object: + return branch_report.get(key, report.get(key)) + + +def _check_config(config: Mapping[str, Any]) -> list[GateFailure]: + failures: list[GateFailure] = [] + dataset = _mapping(config.get("dataset", {})) + for key in REQUIRED_DATASET_HASHES: + if not _valid_sha256(dataset.get(key)): + failures.append( + GateFailure( + "config", + f"dataset.{key} must be pinned to a 64-character SHA-256", + ) + ) + + branches = _mapping(config.get("branches", {})) + if not branches: + failures.append( + GateFailure("config", "branches must configure at least one branch") + ) + for branch in _required_branches(config): + if branch not in branches: + failures.append( + GateFailure( + "config", + f"required branch {branch!r} has no branch config", + ) + ) + if "seed" not in _mapping(config.get("determinism", {})): + failures.append(GateFailure("config", "determinism.seed must be pinned")) + return failures + + +def _required_branches(config: Mapping[str, Any]) -> list[str]: + configured = list(_mapping(config.get("branches", {})).keys()) + required = config.get("required_branches", configured) + if not isinstance(required, list): + return configured + return [str(branch) for branch in required] + + +def _check_onnx_provider(environment: Mapping[str, Any]) -> list[GateFailure]: + failures: list[GateFailure] = [] + requested = environment.get("requested_provider") + actual = environment.get("actual_provider") + if not isinstance(requested, list) or not all( + isinstance(provider, str) and provider for provider in requested + ): + failures.append( + GateFailure( + "malformed_report", + "missing or invalid environment.requested_provider", + ) + ) + if not isinstance(actual, str) or not actual: + failures.append( + GateFailure( + "malformed_report", + "missing or invalid environment.actual_provider", + ) + ) + elif isinstance(requested, list) and requested and actual != requested[0]: + failures.append( + GateFailure( + "malformed_report", + ( + f"environment.actual_provider {actual!r} does not match " + f"requested provider {requested[0]!r}" + ), + ) + ) + return failures + + +def _valid_sha256(value: object) -> bool: + if not isinstance(value, str) or len(value) != 64: + return False + return all(character in string.hexdigits for character in value) + + +def _float_config(value: object, *, default: float) -> float: + try: + numeric = float(value) + except (TypeError, ValueError): + return default + return numeric if math.isfinite(numeric) else default + + def main() -> None: args = parse_args() report = load_json(args.report) config = load_json(args.config) failures = check_report(report, config) if args.repeat_report is not None: - tolerance = float( - _mapping(config.get("determinism", {})).get("metric_tolerance", 0.0) - ) + repeat_report = load_json(args.repeat_report) + failures.extend(check_report(repeat_report, config)) failures.extend( - compare_repeated_reports(report, load_json(args.repeat_report), tolerance) + compare_repeated_reports(report, repeat_report, config=config) ) if failures: for failure in failures: diff --git a/scripts/merge_vision_v8_coco_reports.py b/scripts/merge_vision_v8_coco_reports.py new file mode 100644 index 00000000..90a6a47f --- /dev/null +++ b/scripts/merge_vision_v8_coco_reports.py @@ -0,0 +1,77 @@ +"""Merge per-branch Vision v8 COCO reports into one gated release report.""" + +from __future__ import annotations + +import argparse +import json +from copy import deepcopy +from pathlib import Path +from typing import Any, Mapping + +from scripts.evaluate_tr_hash_coco import _write_markdown_report + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("reports", nargs="+", type=Path) + parser.add_argument("--output", required=True, type=Path) + return parser.parse_args() + + +def load_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"report root must be an object: {path}") + return payload + + +def merge_reports(reports: list[Mapping[str, Any]]) -> dict[str, Any]: + if not reports: + raise ValueError("at least one report is required") + + merged = deepcopy(dict(reports[0])) + merged["checkpoint"] = "multiple ONNX branch artifacts" + merged["model"] = "multiple ONNX branch artifacts" + merged["metadata"] = "multiple ONNX branch sidecars" + merged["branches"] = {} + shared_keys = ("backend", "framework_commit", "dataset", "protocol") + for report in reports: + for key in shared_keys: + if report.get(key) != reports[0].get(key): + raise ValueError(f"cannot merge reports with different {key}") + + branches = report.get("branches") + if not isinstance(branches, Mapping): + raise ValueError("each report must contain a branches object") + for branch, branch_report in branches.items(): + if branch in merged["branches"]: + raise ValueError(f"duplicate branch in merged reports: {branch}") + if not isinstance(branch_report, Mapping): + raise ValueError(f"branch report must be an object: {branch}") + merged_branch = deepcopy(dict(branch_report)) + for key in ( + "checkpoint", + "checkpoint_sha256", + "model", + "metadata", + "metadata_sha256", + ): + merged_branch.setdefault(key, report.get(key)) + merged["branches"][branch] = merged_branch + + return merged + + +def main() -> None: + args = parse_args() + report = merge_reports([load_json(path) for path in args.reports]) + args.output.mkdir(parents=True, exist_ok=True) + (args.output / "evaluation.json").write_text( + json.dumps(report, indent=2) + "\n", + encoding="utf-8", + ) + _write_markdown_report(report, args.output / "evaluation.md") + + +if __name__ == "__main__": + main() diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index d74fe86a..501a20e8 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -9,6 +9,7 @@ _timing_summary, _write_markdown_report, ) +from scripts.merge_vision_v8_coco_reports import merge_reports def test_both_branches_require_end_to_end_head(): @@ -101,3 +102,53 @@ def test_markdown_report_contains_release_metrics(tmp_path): text = output.read_text(encoding="utf-8") assert "Vision v8 COCO Accuracy Report" in text assert "| o2m-nms | 0.200000 | 0.300000" in text + + +def test_merge_onnx_reports_keeps_both_branch_artifact_hashes(): + report = { + "backend": "onnx", + "framework_commit": "abc123", + "checkpoint": "o2m.onnx", + "checkpoint_sha256": "a" * 64, + "model": "o2m.onnx", + "metadata": "o2m.json", + "metadata_sha256": "b" * 64, + "dataset": {"name": "coco-2017"}, + "environment": {}, + "protocol": {"seed": 0}, + "branches": {"o2m-nms": {"metrics": {}}}, + } + nms_free = { + **report, + "checkpoint": "nms_free.onnx", + "checkpoint_sha256": "c" * 64, + "model": "nms_free.onnx", + "metadata": "nms_free.json", + "metadata_sha256": "d" * 64, + "branches": {"nms-free": {"metrics": {}}}, + } + + merged = merge_reports([report, nms_free]) + + assert set(merged["branches"]) == {"o2m-nms", "nms-free"} + assert merged["branches"]["o2m-nms"]["checkpoint_sha256"] == "a" * 64 + assert merged["branches"]["nms-free"]["checkpoint_sha256"] == "c" * 64 + assert merged["branches"]["nms-free"]["metadata_sha256"] == "d" * 64 + + +def test_merge_onnx_reports_rejects_mismatched_protocol(): + first = { + "backend": "onnx", + "framework_commit": "abc123", + "dataset": {"name": "coco-2017"}, + "protocol": {"seed": 0}, + "branches": {"o2m-nms": {"metrics": {}}}, + } + second = { + **first, + "protocol": {"seed": 1}, + "branches": {"nms-free": {"metrics": {}}}, + } + + with pytest.raises(ValueError, match="protocol"): + merge_reports([first, second]) diff --git a/tests/test_vision_v8_coco_accuracy_gate.py b/tests/test_vision_v8_coco_accuracy_gate.py index 4a33311a..ca053c29 100644 --- a/tests/test_vision_v8_coco_accuracy_gate.py +++ b/tests/test_vision_v8_coco_accuracy_gate.py @@ -1,8 +1,12 @@ from scripts.check_vision_v8_coco_report import ( check_report, compare_repeated_reports, + determinism_tolerance, ) +HASH = "a" * 64 +OTHER_HASH = "b" * 64 + def _config() -> dict: return { @@ -10,9 +14,17 @@ def _config() -> dict: "name": "coco-2017", "split": "val2017", "required_image_count": 2, - "annotations_sha256": "annotations", - "image_list_sha256": "image-list", + "annotations_sha256": HASH, + "image_list_sha256": OTHER_HASH, + }, + "determinism": { + "seed": 0, + "metric_tolerance": 1e-12, + "cpu_metric_tolerance": 1e-12, + "cuda_metric_tolerance": 1e-6, + "tensorrt_metric_tolerance": 1e-5, }, + "required_branches": ["o2m-nms", "nms-free"], "branches": { "o2m-nms": { "baseline_metrics": { @@ -30,7 +42,21 @@ def _config() -> dict: "map50": 0.010, "ar_100": 0.010, }, - } + }, + "nms-free": { + "baseline_metrics": { + "map50_95": 0.096, + "map50": 0.140, + }, + "absolute_floors": { + "map50_95": 0.090, + "map50": 0.130, + }, + "max_regressions": { + "map50_95": 0.005, + "map50": 0.010, + }, + }, }, } @@ -52,13 +78,16 @@ def _metrics(**overrides) -> dict: def _report(metrics: dict | None = None) -> dict: return { "schema_version": 1, + "backend": "pytorch", "framework_commit": "abc123", + "checkpoint": "checkpoint.pt", + "checkpoint_sha256": HASH, "dataset": { "name": "coco-2017", "split": "val2017", "evaluated_images": 2, - "annotations_sha256": "annotations", - "image_list_sha256": "image-list", + "annotations_sha256": HASH, + "image_list_sha256": OTHER_HASH, }, "environment": { "python": "3.11", @@ -66,10 +95,17 @@ def _report(metrics: dict | None = None) -> dict: "torch": "2.6.0", "onnxruntime": "1.23.2", }, + "protocol": { + "seed": 0, + "release_eligible": True, + }, "branches": { "o2m-nms": { "metrics": metrics if metrics is not None else _metrics(), - } + }, + "nms-free": { + "metrics": _metrics(map50_95=0.096, map50=0.140), + }, }, } @@ -88,6 +124,90 @@ def test_accuracy_gate_rejects_missing_dataset_hash() -> None: assert any("annotations_sha256" in failure.message for failure in failures) +def test_accuracy_gate_rejects_unpinned_config_dataset_hashes() -> None: + config = _config() + config["dataset"]["annotations_sha256"] = None + + failures = check_report(_report(), config) + + assert any(failure.kind == "config" for failure in failures) + assert any("annotations_sha256" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_mismatched_canonical_dataset_hash() -> None: + report = _report() + report["dataset"]["image_list_sha256"] = HASH + + failures = check_report(report, _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("image_list_sha256" in failure.message for failure in failures) + + +def test_accuracy_gate_requires_complete_configured_branch_set() -> None: + report = _report() + del report["branches"]["nms-free"] + + failures = check_report(report, _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("missing required branch" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_non_finite_metrics() -> None: + failures = check_report(_report(_metrics(map50_95=float("nan"))), _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("non-finite" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_out_of_range_metrics() -> None: + failures = check_report(_report(_metrics(map50=1.1)), _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("outside [0, 1]" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_missing_release_metadata() -> None: + report = _report() + del report["checkpoint_sha256"] + report["protocol"]["release_eligible"] = False + + failures = check_report(report, _config()) + + assert any("checkpoint_sha256" in failure.message for failure in failures) + assert any("release_eligible" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_missing_backend_and_seed() -> None: + report = _report() + del report["backend"] + del report["protocol"]["seed"] + + failures = check_report(report, _config()) + + assert any("backend" in failure.message for failure in failures) + assert any("protocol.seed" in failure.message for failure in failures) + + +def test_accuracy_gate_rejects_onnx_provider_fallback() -> None: + report = _report() + report["backend"] = "onnx" + report["model"] = "model.onnx" + report["metadata"] = "model.json" + report["metadata_sha256"] = OTHER_HASH + report["environment"]["requested_provider"] = [ + "CUDAExecutionProvider", + "CPUExecutionProvider", + ] + report["environment"]["actual_provider"] = "CPUExecutionProvider" + + failures = check_report(report, _config()) + + assert any(failure.kind == "malformed_report" for failure in failures) + assert any("actual_provider" in failure.message for failure in failures) + + def test_accuracy_gate_separates_absolute_floor_from_baseline_regression() -> None: floor_failures = check_report(_report(_metrics(map50_95=0.180)), _config()) regression_failures = check_report(_report(_metrics(map50_95=0.194)), _config()) @@ -107,3 +227,22 @@ def test_repeated_report_comparison_flags_metric_drift() -> None: assert len(failures) == 1 assert failures[0].kind == "determinism" assert "map50" in failures[0].message + + +def test_repeated_report_comparison_rejects_non_finite_metric() -> None: + failures = compare_repeated_reports( + _report(), + _report(_metrics(map50_95=float("nan"))), + tolerance=1e-7, + ) + + assert any(failure.kind == "determinism" for failure in failures) + assert any("non-finite" in failure.message for failure in failures) + + +def test_determinism_tolerance_uses_actual_onnx_provider() -> None: + report = _report() + report["backend"] = "onnx" + report["environment"]["actual_provider"] = "CUDAExecutionProvider" + + assert determinism_tolerance(report, _config()) == 1e-6 From 001c1abc8f8a9a92ebc1524d9d499fcba1d58131 Mon Sep 17 00:00:00 2001 From: Illiyin Date: Tue, 1 Sep 2026 15:22:59 +0500 Subject: [PATCH 3/4] Fix COCO gate release path blockers --- .github/workflows/vision-v8-coco-accuracy.yml | 45 +++++-------------- docs/vision-v8-coco-accuracy-gates.md | 13 ++++-- scripts/evaluate_tr_hash_coco.py | 9 +++- tests/test_coco_release_evaluation.py | 20 +++++++++ 4 files changed, 49 insertions(+), 38 deletions(-) diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 03b28365..00ffb28d 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -35,12 +35,6 @@ on: checkpoint: description: "Local path to the Vision v8 checkpoint on the runner" required: false - onnx_model: - description: "Local path to the Vision v8 ONNX model on the runner" - required: false - onnx_metadata: - description: "Local path to the Vision v8 ONNX metadata sidecar on the runner" - required: false onnx_o2m_model: description: "Local path to the O2M Vision v8 ONNX model on the runner" required: false @@ -63,13 +57,6 @@ on: description: "Runner label with dataset/checkpoint access" required: true default: "self-hosted" - branch: - description: "Detector branch to evaluate" - required: true - default: "both" - type: choice - options: - - both - o2m-nms - nms-free backend: @@ -155,27 +142,19 @@ jobs: run: | if [ "${{ inputs.backend }}" = "pytorch" ]; then test -n "${{ inputs.checkpoint }}" - python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 - python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "${{ inputs.branch }}" --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 + python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch both --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 + python scripts/evaluate_tr_hash_coco.py "${{ inputs.checkpoint }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch both --device "${{ inputs.device }}" --precision "${{ inputs.precision }}" --seed 0 else - onnx_branch="${{ inputs.branch }}" - if [ "$onnx_branch" = "both" ]; then - test -n "${{ inputs.onnx_o2m_model }}" - test -n "${{ inputs.onnx_o2m_metadata }}" - test -n "${{ inputs.onnx_nms_free_model }}" - test -n "${{ inputs.onnx_nms_free_metadata }}" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 - python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 - python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b - else - test -n "${{ inputs.onnx_model }}" - test -n "${{ inputs.onnx_metadata }}" - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 - python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_model }}" --metadata "${{ inputs.onnx_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b --branch "$onnx_branch" --provider "${{ inputs.provider }}" --seed 0 - fi + test -n "${{ inputs.onnx_o2m_model }}" + test -n "${{ inputs.onnx_o2m_metadata }}" + test -n "${{ inputs.onnx_nms_free_model }}" + test -n "${{ inputs.onnx_nms_free_metadata }}" + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_a_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_a_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_a_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_a + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_o2m_model }}" --metadata "${{ inputs.onnx_o2m_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_o2m --branch o2m-nms --provider "${{ inputs.provider }}" --seed 0 + python scripts/evaluate_onnx_coco.py --model "${{ inputs.onnx_nms_free_model }}" --metadata "${{ inputs.onnx_nms_free_metadata }}" --annotations "${{ inputs.annotations }}" --images "${{ inputs.images }}" --output artifacts/vision_v8_coco_eval/run_b_nms_free --branch nms-free --provider "${{ inputs.provider }}" --seed 0 + python scripts/merge_vision_v8_coco_reports.py artifacts/vision_v8_coco_eval/run_b_o2m/evaluation.json artifacts/vision_v8_coco_eval/run_b_nms_free/evaluation.json --output artifacts/vision_v8_coco_eval/run_b fi - name: Gate full COCO report run: >- diff --git a/docs/vision-v8-coco-accuracy-gates.md b/docs/vision-v8-coco-accuracy-gates.md index 6e5b18b9..12baf9a3 100644 --- a/docs/vision-v8-coco-accuracy-gates.md +++ b/docs/vision-v8-coco-accuracy-gates.md @@ -71,10 +71,11 @@ provider fallback fails the gate. Gate configuration must be changed explicitly in source review; CI must not relax tolerances or thresholds at runtime. The `Vision v8 COCO accuracy` workflow runs lightweight gate tests on pull -requests. Its manual full-COCO job runs the selected evaluator twice with the -same seed, rejects malformed/regressed/non-deterministic reports, uploads the -JSON and Markdown reports as workflow artifacts, and can attach those reports to -an existing GitHub Release when `release_tag` is provided. +requests. Its manual full-COCO publication job always evaluates both configured +branches twice with the same seed, rejects malformed/regressed/non-deterministic +reports, uploads the JSON and Markdown reports as workflow artifacts, and can +attach those reports to an existing GitHub Release when `release_tag` is +provided. ## Reproduction Commands @@ -91,6 +92,10 @@ python scripts/evaluate_tr_hash_coco.py \ --precision bf16 ``` +For directory checkpoints, `checkpoint_sha256` is the SHA-256 of the weights file +that the native loader uses: `ema.safetensors` when present, otherwise +`model.safetensors`. + ONNX Runtime full COCO evaluation for one exported branch: ```bash diff --git a/scripts/evaluate_tr_hash_coco.py b/scripts/evaluate_tr_hash_coco.py index 63f20424..a4904410 100644 --- a/scripts/evaluate_tr_hash_coco.py +++ b/scripts/evaluate_tr_hash_coco.py @@ -83,7 +83,14 @@ def _sha256_file(path: Path) -> str: def _checkpoint_sha256(path: Path) -> str | None: - return _sha256_file(path) if path.is_file() else None + if path.is_file(): + return _sha256_file(path) + if path.is_dir(): + for weights_name in ("ema.safetensors", "model.safetensors"): + weights_path = path / weights_name + if weights_path.is_file(): + return _sha256_file(weights_path) + return None def _image_list_sha256(coco: Any, image_ids: list[int]) -> str: diff --git a/tests/test_coco_release_evaluation.py b/tests/test_coco_release_evaluation.py index 501a20e8..64647fbb 100644 --- a/tests/test_coco_release_evaluation.py +++ b/tests/test_coco_release_evaluation.py @@ -1,9 +1,12 @@ +import hashlib + import pytest from scripts.evaluate_tr_hash_coco import ( BRANCHES, _branch_contract, _branches_to_run, + _checkpoint_sha256, _image_list_sha256, _percentile, _timing_summary, @@ -54,6 +57,23 @@ class Coco: assert first != different_order +def test_checkpoint_sha256_hashes_loaded_directory_weights(tmp_path): + checkpoint = tmp_path / "checkpoint" + checkpoint.mkdir() + model_weights = checkpoint / "model.safetensors" + ema_weights = checkpoint / "ema.safetensors" + model_weights.write_bytes(b"model weights") + ema_weights.write_bytes(b"ema weights") + + assert _checkpoint_sha256(checkpoint) == hashlib.sha256(b"ema weights").hexdigest() + + ema_weights.unlink() + + assert _checkpoint_sha256(checkpoint) == hashlib.sha256( + b"model weights" + ).hexdigest() + + def test_branch_contract_distinguishes_nms_requirements(): assert "class-aware NMS" in _branch_contract("o2m-nms")["postprocess"] assert "no NMS" in _branch_contract("nms-free")["postprocess"] From 342dd6cc9646c6821750c0107f3dcfa548452e6e Mon Sep 17 00:00:00 2001 From: Illiyin Date: Tue, 1 Sep 2026 16:19:12 +0500 Subject: [PATCH 4/4] Fix Vision v8 COCO workflow syntax --- .github/workflows/vision-v8-coco-accuracy.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/workflows/vision-v8-coco-accuracy.yml b/.github/workflows/vision-v8-coco-accuracy.yml index 00ffb28d..6adbb621 100644 --- a/.github/workflows/vision-v8-coco-accuracy.yml +++ b/.github/workflows/vision-v8-coco-accuracy.yml @@ -57,8 +57,6 @@ on: description: "Runner label with dataset/checkpoint access" required: true default: "self-hosted" - - o2m-nms - - nms-free backend: description: "Evaluator backend" required: true