diff --git a/benchmark/README.md b/benchmark/README.md index 2b8f9e696..f29b0db82 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -16,6 +16,9 @@ Harbor directly at the upstream provider. For a small automated MMLU-Redux example using NeMo Gym instead of Harbor, see [Evaluate Switchyard routing with NeMo Gym](nemo_gym/README.md). +To replay recorded TypeSafe/Jev routing evidence without credentials or provider calls, see +[Offline TypeSafe/Jev Replay](typesafe/README.md). + ## Prerequisites From the repo root: diff --git a/benchmark/typesafe/README.md b/benchmark/typesafe/README.md new file mode 100644 index 000000000..2a935675d --- /dev/null +++ b/benchmark/typesafe/README.md @@ -0,0 +1,71 @@ + + + +# Offline TypeSafe/Jev Replay + +Use the replay tool to evaluate recorded TypeSafe routing evidence without credentials, network +access, or new model calls: + +```bash +uv run --no-sync python benchmark/typesafe_replay.py \ + benchmark/typesafe/fixtures/synthetic-routing.json +``` + +Pass `--output report.json` to save the stable, machine-readable report. Replaying the same fixture +always produces the same bytes. + +To compare several confidence thresholds on the same recorded cases, run: + +```bash +uv run --no-sync python benchmark/typesafe_replay.py \ + benchmark/typesafe/fixtures/synthetic-routing.json \ + --thresholds 0 0.25 0.5 1 +``` + +The comparison sorts the supplied thresholds and reports selected-target counts, fallback count, +quality, cost, and regret against the same best fixed target for each one. Thresholds must be +unique numbers in `[0, 1]`. A case falls back only when its unrounded confidence is *below* the +threshold. Without `--thresholds`, the original single-threshold output is unchanged. + +Choose a threshold using development cases, then check the frozen policy on separate held-out +cases. Repeatedly choosing from held-out outcomes turns them into tuning data. A comparison is +descriptive: it cannot predict results for tasks or target versions absent from the fixture. + +## Fixture format + +Every fixture is one JSON document with numeric `schema_version: 1`; JSON Booleans are not valid +versions. It records: + +- a suite name and pinned Jev model identifier; +- two or more candidate labels and descriptions; +- the confidence threshold and fallback target; +- one or more unique candidate orders per case, with a complete probability distribution for each; +- measured quality in `[0, 1]` for every candidate outcome; +- optional nonnegative cost for each candidate outcome. + +The replay averages the recorded distributions, normalizes the average, and chooses the largest +probability. Configured candidate order breaks an exact tie. Confidence is the selected +probability's improvement over a uniform distribution, scaled to `[0, 1]`. A result below +`base_threshold` selects `default_target`. + +`best_fixed_target` is the candidate with the highest total quality when used for every case. When +every quality-tied candidate has complete costs, lower cost breaks the tie. If any tied candidate +has incomplete costs, configured candidate order breaks the tie instead of treating an unknown +cost as zero. The report compares the routed totals with that fixed baseline. Positive +`quality_regret_vs_best_fixed` means routing lost quality; a negative value means it beat every +fixed target. A cost total is `null` if any outcome contributing to it has no cost, and +`cost_delta_vs_best_fixed` is `null` unless both compared totals are available. Otherwise the cost +delta uses the same sign convention. `order_sensitive_case_count` counts cases whose per-order top +candidate changes, while `maximum_probability_movement` reports the largest probability shift +seen across orders. + +## Sanitization boundary + +The schema deliberately excludes request text, credentials, response bodies, and provider error +details. Use opaque case IDs. Candidate descriptions and model identifiers can still contain +deployment information, so replace them with non-sensitive equivalents before committing a +fixture. The checked-in fixture is entirely synthetic and does not claim production routing +quality. + +This is an offline evaluator, not a provider collector. Collection of live evidence must remain an +explicit, separately reviewed workflow. diff --git a/benchmark/typesafe/fixtures/synthetic-routing.json b/benchmark/typesafe/fixtures/synthetic-routing.json new file mode 100644 index 000000000..15c78a242 --- /dev/null +++ b/benchmark/typesafe/fixtures/synthetic-routing.json @@ -0,0 +1,70 @@ +{ + "schema_version": 1, + "suite": "synthetic-two-target-routing", + "model": "jev-pinned-example", + "candidates": [ + { + "label": "capable", + "description": "Best for complex, multi-step work." + }, + { + "label": "efficient", + "description": "Best for short, well-specified work." + } + ], + "base_threshold": 0.25, + "default_target": "efficient", + "cases": [ + { + "id": "simple-request", + "orders": [ + { + "candidate_order": ["capable", "efficient"], + "probabilities": {"capable": 0.1, "efficient": 0.9} + }, + { + "candidate_order": ["efficient", "capable"], + "probabilities": {"capable": 0.2, "efficient": 0.8} + } + ], + "outcomes": { + "capable": {"quality": 1.0, "cost": 10.0}, + "efficient": {"quality": 1.0, "cost": 2.0} + } + }, + { + "id": "complex-request", + "orders": [ + { + "candidate_order": ["capable", "efficient"], + "probabilities": {"capable": 0.8, "efficient": 0.2} + }, + { + "candidate_order": ["efficient", "capable"], + "probabilities": {"capable": 0.6, "efficient": 0.4} + } + ], + "outcomes": { + "capable": {"quality": 1.0, "cost": 10.0}, + "efficient": {"quality": 0.0, "cost": 2.0} + } + }, + { + "id": "order-sensitive-request", + "orders": [ + { + "candidate_order": ["capable", "efficient"], + "probabilities": {"capable": 0.7, "efficient": 0.3} + }, + { + "candidate_order": ["efficient", "capable"], + "probabilities": {"capable": 0.3, "efficient": 0.7} + } + ], + "outcomes": { + "capable": {"quality": 1.0, "cost": 10.0}, + "efficient": {"quality": 0.0, "cost": 2.0} + } + } + ] +} diff --git a/benchmark/typesafe_replay.py b/benchmark/typesafe_replay.py new file mode 100644 index 000000000..094e5dc66 --- /dev/null +++ b/benchmark/typesafe_replay.py @@ -0,0 +1,339 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Replay recorded TypeSafe/Jev routing evidence without provider calls.""" + +from __future__ import annotations + +import argparse +import json +import math +import sys +from pathlib import Path +from typing import Any, cast + +SCHEMA_VERSION = 1 +PROBABILITY_SUM_TOLERANCE = 0.02 + + +class FixtureError(ValueError): + """Report invalid or unsupported replay evidence.""" + + +def require(condition: bool, message: str) -> None: + """Reject incomplete evidence before calculating metrics.""" + if not condition: + raise FixtureError(message) + + +def object_value(value: Any, name: str) -> dict[str, Any]: + """Read one JSON object with a useful field diagnostic.""" + require(isinstance(value, dict), f"{name} must be an object") + return cast(dict[str, Any], value) + + +def string_value(value: Any, name: str) -> str: + """Read one non-empty string field.""" + require(isinstance(value, str) and bool(value.strip()), f"{name} must be a non-empty string") + return value + + +def number_value(value: Any, name: str, *, maximum: float | None = None) -> float: + """Read one finite, nonnegative numeric field.""" + require( + type(value) in (int, float) and math.isfinite(value) and value >= 0, + f"{name} must be a finite, nonnegative number", + ) + number = float(value) + if maximum is not None: + require(number <= maximum, f"{name} must be at most {maximum}") + return number + + +def string_list(value: Any, name: str) -> list[str]: + """Read one non-empty list of unique strings.""" + require(isinstance(value, list) and bool(value), f"{name} must be a non-empty list") + values = [string_value(item, f"{name}[]") for item in value] + require(len(values) == len(set(values)), f"{name} must not contain duplicates") + return values + + +def probabilities(value: Any, labels: list[str], name: str) -> dict[str, float]: + """Validate one complete probability distribution.""" + document = object_value(value, name) + require(set(document) == set(labels), f"{name} must contain every candidate exactly once") + parsed = { + label: number_value(document[label], f"{name}.{label}", maximum=1.0) for label in labels + } + total = sum(parsed.values()) + require( + abs(total - 1.0) <= PROBABILITY_SUM_TOLERANCE, + f"{name} must sum to one within {PROBABILITY_SUM_TOLERANCE}", + ) + return parsed + + +def selected_label(labels: list[str], values: dict[str, float]) -> str: + """Select the highest probability, breaking ties by configured candidate order.""" + return max(labels, key=values.__getitem__) + + +def rounded(value: float) -> float: + """Keep reports readable and stable across harmless floating-point noise.""" + return round(value, 12) + + +def replay(document: dict[str, Any]) -> dict[str, Any]: + """Validate one fixture document and return its deterministic replay report.""" + schema_version = document.get("schema_version") + require( + type(schema_version) in (int, float) and schema_version == SCHEMA_VERSION, + f"schema_version must be {SCHEMA_VERSION}", + ) + suite = string_value(document.get("suite"), "suite") + model = string_value(document.get("model"), "model") + candidate_rows = document.get("candidates") + require( + isinstance(candidate_rows, list) and len(candidate_rows) >= 2, + "candidates needs at least two entries", + ) + + candidates = [object_value(row, "candidates[]") for row in candidate_rows] + labels = [string_value(row.get("label"), "candidates[].label") for row in candidates] + require(len(labels) == len(set(labels)), "candidate labels must be unique") + for row in candidates: + string_value(row.get("description"), "candidates[].description") + + default_target = string_value(document.get("default_target"), "default_target") + require(default_target in labels, "default_target must name a candidate") + threshold = number_value(document.get("base_threshold"), "base_threshold", maximum=1.0) + case_rows = document.get("cases") + require(isinstance(case_rows, list) and bool(case_rows), "cases must be a non-empty list") + + fixed_quality = dict.fromkeys(labels, 0.0) + fixed_cost: dict[str, float | None] = dict.fromkeys(labels, 0.0) + selected_counts = dict.fromkeys(labels, 0) + routed_quality = 0.0 + routed_cost: float | None = 0.0 + fallback_count = 0 + order_sensitive_count = 0 + maximum_movement = 0.0 + case_ids: set[str] = set() + case_reports: list[dict[str, Any]] = [] + + for case_index, case_value in enumerate(case_rows): + case = object_value(case_value, f"cases[{case_index}]") + case_id = string_value(case.get("id"), f"cases[{case_index}].id") + require(case_id not in case_ids, f"duplicate case id {case_id!r}") + case_ids.add(case_id) + + order_rows = case.get("orders") + require( + isinstance(order_rows, list) and bool(order_rows), + f"case {case_id!r} needs at least one candidate order", + ) + seen_orders: set[tuple[str, ...]] = set() + distributions: list[dict[str, float]] = [] + order_winners: list[str] = [] + for order_index, order_value in enumerate(order_rows): + order = object_value(order_value, f"case {case_id!r} order {order_index}") + order_labels = string_list( + order.get("candidate_order"), + f"case {case_id!r} order {order_index}.candidate_order", + ) + require( + set(order_labels) == set(labels) and len(order_labels) == len(labels), + f"case {case_id!r} order {order_index} must contain every candidate", + ) + order_key = tuple(order_labels) + require(order_key not in seen_orders, f"case {case_id!r} repeats a candidate order") + seen_orders.add(order_key) + distribution = probabilities( + order.get("probabilities"), + labels, + f"case {case_id!r} order {order_index}.probabilities", + ) + distributions.append(distribution) + order_winners.append(selected_label(order_labels, distribution)) + + averaged = { + label: sum(distribution[label] for distribution in distributions) / len(distributions) + for label in labels + } + average_total = sum(averaged.values()) + averaged = {label: value / average_total for label, value in averaged.items()} + classifier_target = selected_label(labels, averaged) + uniform = 1.0 / len(labels) + unbounded_confidence = (averaged[classifier_target] - uniform) / (1.0 - uniform) + confidence = max(0.0, min(1.0, unbounded_confidence)) + + fallback = confidence < threshold + final_target = default_target if fallback else classifier_target + fallback_count += int(fallback) + selected_counts[final_target] += 1 + + order_sensitive = len(set(order_winners)) > 1 + order_sensitive_count += int(order_sensitive) + movement = max( + max(distribution[label] for distribution in distributions) + - min(distribution[label] for distribution in distributions) + for label in labels + ) + maximum_movement = max(maximum_movement, movement) + + outcomes = object_value(case.get("outcomes"), f"case {case_id!r}.outcomes") + require( + set(outcomes) == set(labels), + f"case {case_id!r}.outcomes must contain every candidate exactly once", + ) + parsed_quality: dict[str, float] = {} + parsed_cost: dict[str, float | None] = {} + for label in labels: + outcome = object_value(outcomes[label], f"case {case_id!r}.outcomes.{label}") + quality = number_value( + outcome.get("quality"), f"case {case_id!r}.outcomes.{label}.quality", maximum=1.0 + ) + cost_value = outcome.get("cost") + cost = ( + None + if cost_value is None + else number_value(cost_value, f"case {case_id!r}.outcomes.{label}.cost") + ) + parsed_quality[label] = quality + parsed_cost[label] = cost + fixed_quality[label] += quality + if fixed_cost[label] is not None: + fixed_cost[label] = None if cost is None else fixed_cost[label] + cost + + routed_quality += parsed_quality[final_target] + selected_cost = parsed_cost[final_target] + if routed_cost is not None: + routed_cost = None if selected_cost is None else routed_cost + selected_cost + case_reports.append( + { + "id": case_id, + "average_probabilities": {label: rounded(averaged[label]) for label in labels}, + "classifier_target": classifier_target, + "confidence": rounded(confidence), + "fallback": fallback, + "selected_target": final_target, + "order_sensitive": order_sensitive, + "maximum_probability_movement": rounded(movement), + } + ) + + best_quality = max(fixed_quality.values()) + quality_ties = [label for label in labels if fixed_quality[label] == best_quality] + if all(fixed_cost[label] is not None for label in quality_ties): + best_fixed = min(quality_ties, key=lambda label: fixed_cost[label]) + else: + best_fixed = quality_ties[0] + fixed_report = { + label: { + "quality": rounded(fixed_quality[label]), + "cost": None if fixed_cost[label] is None else rounded(fixed_cost[label]), + } + for label in labels + } + best_fixed_cost = fixed_cost[best_fixed] + cost_delta = ( + None + if routed_cost is None or best_fixed_cost is None + else rounded(routed_cost - best_fixed_cost) + ) + return { + "schema_version": SCHEMA_VERSION, + "suite": suite, + "model": model, + "base_threshold": threshold, + "default_target": default_target, + "case_count": len(case_reports), + "selected_counts": selected_counts, + "fallback_count": fallback_count, + "order_sensitive_case_count": order_sensitive_count, + "maximum_probability_movement": rounded(maximum_movement), + "routed": { + "quality": rounded(routed_quality), + "cost": None if routed_cost is None else rounded(routed_cost), + }, + "fixed_targets": fixed_report, + "best_fixed_target": best_fixed, + "quality_regret_vs_best_fixed": rounded(fixed_quality[best_fixed] - routed_quality), + "cost_delta_vs_best_fixed": cost_delta, + "cases": case_reports, + } + + +def compare_thresholds(document: dict[str, Any], thresholds: list[float]) -> dict[str, Any]: + """Compare explicit thresholds using the same recorded evidence and outcomes.""" + require(bool(thresholds), "thresholds must not be empty") + values = [number_value(value, "thresholds[]", maximum=1.0) for value in thresholds] + require(len(values) == len(set(values)), "thresholds must not contain duplicates") + + baseline = replay(document) + comparisons = [] + for value in sorted(values): + report = ( + baseline + if value == baseline["base_threshold"] + else replay({**document, "base_threshold": value}) + ) + comparisons.append( + { + "base_threshold": value, + "selected_counts": report["selected_counts"], + "fallback_count": report["fallback_count"], + "routed": report["routed"], + "quality_regret_vs_best_fixed": report["quality_regret_vs_best_fixed"], + "cost_delta_vs_best_fixed": report["cost_delta_vs_best_fixed"], + } + ) + return { + "schema_version": baseline["schema_version"], + "suite": baseline["suite"], + "model": baseline["model"], + "default_target": baseline["default_target"], + "case_count": baseline["case_count"], + "fixed_targets": baseline["fixed_targets"], + "best_fixed_target": baseline["best_fixed_target"], + "thresholds": comparisons, + } + + +def read_fixture(path: Path) -> dict[str, Any]: + """Read one replay fixture from disk.""" + return object_value(json.loads(path.read_text(encoding="utf-8")), str(path)) + + +def main(argv: list[str] | None = None) -> int: + """Replay one fixture and write stable JSON to stdout or a file.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("fixture", type=Path, help="Versioned TypeSafe replay fixture") + parser.add_argument("--output", type=Path, help="Write the JSON report to this path") + parser.add_argument( + "--thresholds", + nargs="+", + type=float, + help="Compare these confidence thresholds from 0 to 1 instead of replaying one", + ) + args = parser.parse_args(argv) + try: + fixture = read_fixture(args.fixture) + report = ( + replay(fixture) + if args.thresholds is None + else compare_thresholds(fixture, args.thresholds) + ) + encoded = json.dumps(report, indent=2, sort_keys=True, allow_nan=False) + "\n" + if args.output is None: + sys.stdout.write(encoded) + else: + args.output.write_text(encoded, encoding="utf-8") + except (FixtureError, OSError, UnicodeError, json.JSONDecodeError) as error: + print(f"Cannot replay fixture: {error}", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_typesafe_replay.py b/tests/test_typesafe_replay.py new file mode 100644 index 000000000..598652d0a --- /dev/null +++ b/tests/test_typesafe_replay.py @@ -0,0 +1,257 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import importlib.util +import json +import math +import subprocess +import sys +from copy import deepcopy +from pathlib import Path +from types import ModuleType + +import pytest + +REPO = Path(__file__).resolve().parents[1] +SCRIPT = REPO / "benchmark/typesafe_replay.py" +FIXTURE = REPO / "benchmark/typesafe/fixtures/synthetic-routing.json" + + +@pytest.fixture +def replay_module() -> ModuleType: + """Load the replay tool without installing Switchyard or provider dependencies.""" + spec = importlib.util.spec_from_file_location("switchyard_typesafe_replay", SCRIPT) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +@pytest.fixture +def fixture_document() -> dict[str, object]: + """Return an isolated copy of the checked-in synthetic fixture.""" + return json.loads(FIXTURE.read_text(encoding="utf-8")) + + +def test_synthetic_fixture_reports_routing_quality_cost_and_order_sensitivity( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Replay aggregation, fallback, and fixed-baseline comparisons together.""" + report = replay_module.replay(fixture_document) + + assert report["case_count"] == 3 + assert report["selected_counts"] == {"capable": 1, "efficient": 2} + assert report["fallback_count"] == 1 + assert report["order_sensitive_case_count"] == 1 + assert report["maximum_probability_movement"] == 0.4 + assert report["routed"] == {"quality": 2.0, "cost": 14.0} + assert report["fixed_targets"] == { + "capable": {"quality": 3.0, "cost": 30.0}, + "efficient": {"quality": 1.0, "cost": 6.0}, + } + assert report["best_fixed_target"] == "capable" + assert report["quality_regret_vs_best_fixed"] == 1.0 + assert report["cost_delta_vs_best_fixed"] == -16.0 + assert report["cases"][2] == { + "id": "order-sensitive-request", + "average_probabilities": {"capable": 0.5, "efficient": 0.5}, + "classifier_target": "capable", + "confidence": 0.0, + "fallback": True, + "selected_target": "efficient", + "order_sensitive": True, + "maximum_probability_movement": 0.4, + } + + +def test_cli_output_is_byte_stable_and_offline() -> None: + """Run twice with an empty environment and require byte-identical JSON.""" + command = [sys.executable, str(SCRIPT), str(FIXTURE)] + first = subprocess.run(command, check=False, capture_output=True, env={}) + second = subprocess.run(command, check=False, capture_output=True, env={}) + + assert first.returncode == second.returncode == 0 + assert first.stderr == second.stderr == b"" + assert first.stdout == second.stdout + assert json.loads(first.stdout)["suite"] == "synthetic-two-target-routing" + + +def test_threshold_comparison_uses_the_same_fixed_outcomes( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Compare fallback, quality, and cost without changing the original fixture.""" + original = deepcopy(fixture_document) + report = replay_module.compare_thresholds(fixture_document, [1.0, 0.25, 0.0]) + + assert fixture_document == original + assert report["best_fixed_target"] == "capable" + assert report["fixed_targets"] == replay_module.replay(fixture_document)["fixed_targets"] + assert report["thresholds"] == [ + { + "base_threshold": 0.0, + "selected_counts": {"capable": 2, "efficient": 1}, + "fallback_count": 0, + "routed": {"quality": 3.0, "cost": 22.0}, + "quality_regret_vs_best_fixed": 0.0, + "cost_delta_vs_best_fixed": -8.0, + }, + { + "base_threshold": 0.25, + "selected_counts": {"capable": 1, "efficient": 2}, + "fallback_count": 1, + "routed": {"quality": 2.0, "cost": 14.0}, + "quality_regret_vs_best_fixed": 1.0, + "cost_delta_vs_best_fixed": -16.0, + }, + { + "base_threshold": 1.0, + "selected_counts": {"capable": 0, "efficient": 3}, + "fallback_count": 3, + "routed": {"quality": 1.0, "cost": 6.0}, + "quality_regret_vs_best_fixed": 2.0, + "cost_delta_vs_best_fixed": -24.0, + }, + ] + + +def test_threshold_comparison_uses_strict_boundary( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """A case falls back only when its unrounded confidence is below the threshold.""" + document = deepcopy(fixture_document) + for order in document["cases"][1]["orders"]: + order["probabilities"] = {"capable": 0.75, "efficient": 0.25} + + report = replay_module.compare_thresholds(document, [0.5, math.nextafter(0.5, 1.0)]) + assert [row["fallback_count"] for row in report["thresholds"]] == [1, 2] + assert [row["routed"]["quality"] for row in report["thresholds"]] == [2.0, 1.0] + + +def test_threshold_comparison_preserves_unknown_costs( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Never treat an unknown selected or fixed-target cost as zero.""" + document = deepcopy(fixture_document) + for case in document["cases"]: + for outcome in case["outcomes"].values(): + outcome.pop("cost") + + report = replay_module.compare_thresholds(document, [0.0, 1.0]) + assert all(row["routed"]["cost"] is None for row in report["thresholds"]) + assert all(row["cost_delta_vs_best_fixed"] is None for row in report["thresholds"]) + + +@pytest.mark.parametrize( + "thresholds", [[], [0.1, 0.1], [-0.1], [1.1], [math.nan], [math.inf], [True]] +) +def test_threshold_comparison_rejects_invalid_values( + replay_module: ModuleType, fixture_document: dict[str, object], thresholds: list[float] +) -> None: + """Reject comparisons whose thresholds cannot describe a routing policy.""" + with pytest.raises(replay_module.FixtureError, match="thresholds"): + replay_module.compare_thresholds(fixture_document, thresholds) + + +def test_threshold_cli_is_byte_stable_and_offline() -> None: + """Render one ordered comparison without credentials or network calls.""" + command = [ + sys.executable, + str(SCRIPT), + str(FIXTURE), + "--thresholds", + "1", + "0", + "0.25", + ] + first = subprocess.run(command, check=False, capture_output=True, env={}) + second = subprocess.run(command, check=False, capture_output=True, env={}) + + assert first.returncode == second.returncode == 0 + assert first.stderr == second.stderr == b"" + assert first.stdout == second.stdout + assert [row["base_threshold"] for row in json.loads(first.stdout)["thresholds"]] == [ + 0.0, + 0.25, + 1.0, + ] + + +def test_unsupported_schema_version_is_rejected( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Reject fixture semantics the evaluator does not understand.""" + fixture_document["schema_version"] = 2 + + with pytest.raises(replay_module.FixtureError, match="schema_version must be 1"): + replay_module.replay(fixture_document) + + +def test_boolean_schema_version_is_rejected_but_integral_float_is_accepted( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Keep JSON Booleans out of the numeric schema-version contract.""" + fixture_document["schema_version"] = True + with pytest.raises(replay_module.FixtureError, match="schema_version must be 1"): + replay_module.replay(fixture_document) + + fixture_document["schema_version"] = 1.0 + assert replay_module.replay(fixture_document)["schema_version"] == 1 + + +def test_missing_costs_keep_quality_metrics_and_deterministic_output( + tmp_path: Path, replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Evaluate quality without inventing cost totals or cost-based tie breaks.""" + document = deepcopy(fixture_document) + document["candidates"].reverse() + for case in document["cases"]: + for outcome in case["outcomes"].values(): + outcome["quality"] = 1.0 + outcome.pop("cost") + + report = replay_module.replay(document) + assert report["routed"] == {"quality": 3.0, "cost": None} + assert report["fixed_targets"] == { + "efficient": {"quality": 3.0, "cost": None}, + "capable": {"quality": 3.0, "cost": None}, + } + assert report["best_fixed_target"] == "efficient" + assert report["quality_regret_vs_best_fixed"] == 0.0 + assert report["cost_delta_vs_best_fixed"] is None + + path = tmp_path / "cost-free.json" + path.write_text(json.dumps(document), encoding="utf-8") + command = [sys.executable, str(SCRIPT), str(path)] + first = subprocess.run(command, check=False, capture_output=True, env={}) + second = subprocess.run(command, check=False, capture_output=True, env={}) + assert first.returncode == second.returncode == 0 + assert first.stderr == second.stderr == b"" + assert first.stdout == second.stdout + + +def test_invalid_probability_distribution_is_rejected( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Do not calculate metrics from incomplete provider evidence.""" + document = deepcopy(fixture_document) + document["cases"][0]["orders"][0]["probabilities"] = { + "capable": 0.8, + "efficient": 0.8, + } + + with pytest.raises(replay_module.FixtureError, match="must sum to one"): + replay_module.replay(document) + + +def test_duplicate_candidate_order_is_rejected( + replay_module: ModuleType, fixture_document: dict[str, object] +) -> None: + """Prevent one repeated ordering from receiving accidental extra weight.""" + document = deepcopy(fixture_document) + first_order = document["cases"][0]["orders"][0] + document["cases"][0]["orders"][1] = deepcopy(first_order) + + with pytest.raises(replay_module.FixtureError, match="repeats a candidate order"): + replay_module.replay(document)