diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f7149fd11..d038425e5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,7 +22,12 @@ jobs: # (add a shard). # # `conftest.py` assigns whole test *files* to shards, deterministically, - # from the collection alone — nothing is exchanged between these jobs. + # from the collection and the measured seconds per file in + # `tests/shard_seconds.json` — nothing is exchanged between these jobs. + # Balancing item counts instead left one shard at 13 of its 15 minutes on + # `main` while another took 8, and any new test file reshuffled the rest + # (#904). When a shard nears the cap, re-measure with + # `scripts/measure_shard_seconds.py` before adding a shard. # `tests/test_shard_partition.py` asserts the union of the shards is the # suite, because a partition that silently drops a file leaves every job # green while a test stops running. diff --git a/ci_sharding.py b/ci_sharding.py index 1f8a7360c..a582f7522 100644 --- a/ci_sharding.py +++ b/ci_sharding.py @@ -8,17 +8,50 @@ from anywhere and has one definition. Pure. It reads nothing, writes nothing, and takes no environment: the caller -supplies the collection and gets back an assignment. That is what lets -``tests/test_shard_partition.py`` assert the properties directly. +supplies the collection (and the measured seconds, when it has them) and gets +back an assignment. That is what lets ``tests/test_shard_partition.py`` assert +the properties directly. """ from __future__ import annotations +import json +import math from collections.abc import Mapping +from pathlib import Path +#: Measured seconds per test file, written by +#: ``scripts/measure_shard_seconds.py`` from a full ``--junitxml`` run. +SECONDS_FILE = Path(__file__).resolve().parent / "tests" / "shard_seconds.json" -def shard_assignment(paths: Mapping[str, int], shards: int) -> dict[str, int]: - """Assign whole test *files* to shards, balancing collected item counts. + +def load_seconds(path: Path = SECONDS_FILE) -> dict[str, float]: + """The measured seconds per file, or nothing when there is no measurement. + + A missing or unreadable file balances on item counts alone, as before: + the measurement only ever improves the balance, never the coverage. + """ + + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return {} + files = payload.get("files") if isinstance(payload, dict) else None + if not isinstance(files, dict): + return {} + return { + str(name): float(value) + for name, value in files.items() + if isinstance(value, int | float) and not isinstance(value, bool) and value >= 0 + } + + +def shard_assignment( + paths: Mapping[str, int], + shards: int, + seconds: Mapping[str, float] | None = None, +) -> dict[str, int]: + """Assign whole test *files* to shards, balancing their expected time. Two properties, and both are load-bearing. @@ -32,15 +65,32 @@ def shard_assignment(paths: Mapping[str, int], shards: int) -> dict[str, int]: is exchanged between jobs, so the union of the shards is exactly the suite and no test can fall between two of them — which a test asserts. - Item count is a proxy for time, and an imperfect one; it is used because it - is free and needs no stored measurements to go stale. Balance is checked in - ``tests/test_shard_partition.py`` rather than assumed. + **Measured time where there is one.** Item count alone was a poor proxy: + a file of forty git-fixture tests costs more than a file of four hundred + pure ones. Balancing counts left one shard at 13 of its 15 minutes on + ``main``, while another took 8. Adding any test file reshuffled most files + between shards, so an unrelated PR could tip a shard past its cap (#904). + ``seconds`` holds each file's measured time. A file without a measurement + (new, or renamed since) costs its item count times the measured seconds + per item. A stale measurement only unbalances; it never drops a file. + Balance is checked in ``tests/test_shard_partition.py`` rather than + assumed. """ - load = [0] * shards + cost = _costs(paths, seconds or {}) + load = [0.0] * shards owner: dict[str, int] = {} - for path, count in sorted(paths.items(), key=lambda item: (-item[1], item[0])): + for path in sorted(paths, key=lambda item: (-cost[item], item)): target = min(range(shards), key=lambda index: (load[index], index)) owner[path] = target - load[target] += count + load[target] += cost[path] return owner + + +def _costs(paths: Mapping[str, int], seconds: Mapping[str, float]) -> dict[str, float]: + known = {path: seconds[path] for path in sorted(paths) if path in seconds} + known_items = sum(paths[path] for path in known) + # ``fsum`` over a sorted order: every shard computes the same rate, bit + # for bit, whatever order its collection listed the files in. + rate = math.fsum(known.values()) / known_items if known_items else 1.0 + return {path: known.get(path, paths[path] * rate) for path in paths} diff --git a/conftest.py b/conftest.py index 7076c36e9..9b9132b8e 100644 --- a/conftest.py +++ b/conftest.py @@ -27,7 +27,7 @@ import pytest # noqa: E402 -from ci_sharding import shard_assignment # noqa: E402 +from ci_sharding import load_seconds, shard_assignment # noqa: E402 @pytest.fixture(autouse=True) @@ -90,7 +90,7 @@ def pytest_collection_modifyitems(config, items) -> None: # noqa: ANN001 counts: dict[str, int] = {} for item in items: counts[item.location[0]] = counts.get(item.location[0], 0) + 1 - owner = shard_assignment(counts, shards) + owner = shard_assignment(counts, shards, load_seconds()) keep = [item for item in items if owner[item.location[0]] == index - 1] dropped = [item for item in items if owner[item.location[0]] != index - 1] if not keep: diff --git a/scripts/measure_shard_seconds.py b/scripts/measure_shard_seconds.py new file mode 100644 index 000000000..7eccd2fca --- /dev/null +++ b/scripts/measure_shard_seconds.py @@ -0,0 +1,78 @@ +"""Write ``tests/shard_seconds.json`` from a full-suite ``--junitxml`` report. + +The CI suite is split into shards by measured time per test file +(``ci_sharding.py``). Re-measure when a shard nears its ``timeout-minutes``: + + python -m pytest -n auto -m "not perf" --ignore=tests/test_adapter_static_only.py \\ + --junitxml=junit.xml + python scripts/measure_shard_seconds.py junit.xml + +The times are relative weights: what matters is how the files compare with +each other, so one machine's measurement balances another's runners. +""" + +from __future__ import annotations + +import argparse +import json +import subprocess +import sys +import xml.etree.ElementTree as ET +from collections import defaultdict +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +OUTPUT = REPO_ROOT / "tests" / "shard_seconds.json" + + +def file_seconds(report: Path, root: Path = REPO_ROOT) -> dict[str, float]: + """Seconds per test file: each test case's setup, call and teardown summed.""" + + totals: dict[str, float] = defaultdict(float) + for case in ET.parse(report).getroot().iter("testcase"): + name = case.get("file") or _file_from_classname(case.get("classname", ""), root) + if name: + totals[name] += float(case.get("time") or 0.0) + return {name: round(value, 1) for name, value in sorted(totals.items())} + + +def _file_from_classname(classname: str, root: Path) -> str | None: + """``tests.test_x.TestY`` → ``tests/test_x.py``, the file the case is in.""" + + parts = classname.split(".") + for end in range(len(parts), 0, -1): + candidate = root.joinpath(*parts[:end]).with_suffix(".py") + if candidate.is_file(): + return candidate.relative_to(root).as_posix() + return None + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("report", type=Path, help="a --junitxml report of the full CI suite") + parser.add_argument( + "--root", + type=Path, + default=REPO_ROOT, + help="the checkout the report was made in (default: this one)", + ) + args = parser.parse_args(argv) + files = file_seconds(args.report, args.root.resolve()) + if not files: + print(f"{args.report} holds no test cases", file=sys.stderr) + return 1 + commit = subprocess.run( + ["git", "rev-parse", "--short=12", "HEAD"], + cwd=args.root, + capture_output=True, + text=True, + check=False, + ).stdout.strip() + payload = {"measured_at": commit or None, "files": files} + OUTPUT.write_text(json.dumps(payload, indent=1, sort_keys=True) + "\n", encoding="utf-8") + print(f"wrote {len(files)} files, {sum(files.values()):.0f}s in total, to {OUTPUT}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/shard_seconds.json b/tests/shard_seconds.json new file mode 100644 index 000000000..05c4bbd95 --- /dev/null +++ b/tests/shard_seconds.json @@ -0,0 +1,368 @@ +{ + "files": { + "tests/harness/test_adversarial_guards.py": 0.0, + "tests/harness/test_claude_code_driver.py": 0.0, + "tests/harness/test_codex_driver.py": 1.2, + "tests/harness/test_cursor_driver.py": 0.0, + "tests/harness/test_cursor_manual_driver.py": 0.0, + "tests/harness/test_detectors.py": 0.2, + "tests/harness/test_exit_criteria.py": 0.0, + "tests/harness/test_harness_layout.py": 2.0, + "tests/harness/test_infrastructure_failures.py": 0.1, + "tests/harness/test_overlay_renderer.py": 0.0, + "tests/harness/test_pressure_ground_truth.py": 6.7, + "tests/harness/test_redaction.py": 0.0, + "tests/harness/test_run_preflight.py": 0.9, + "tests/harness/test_smoke.py": 0.6, + "tests/integration/github_action/test_agent_result.py": 0.0, + "tests/test_absent_input_messages.py": 3.6, + "tests/test_action_comment_fallback.py": 0.7, + "tests/test_action_engine_install.py": 6.8, + "tests/test_action_metadata.py": 0.4, + "tests/test_action_scope_domain.py": 0.0, + "tests/test_action_scope_projection.py": 3.0, + "tests/test_action_surface_diff.py": 0.9, + "tests/test_adapter_contracts.py": 0.2, + "tests/test_adapter_entry_point_discovery.py": 4.2, + "tests/test_adapter_registry.py": 0.1, + "tests/test_adk_remote_bindings.py": 1.2, + "tests/test_adopter_pins_resolve.py": 0.2, + "tests/test_adopter_vocabulary.py": 9.6, + "tests/test_adopters_registry.py": 0.5, + "tests/test_adoption_ladder.py": 0.8, + "tests/test_adoption_scorer_input_recovery.py": 0.0, + "tests/test_adoption_walk.py": 71.1, + "tests/test_advisory_cadence.py": 0.6, + "tests/test_agent_action_summary.py": 5.2, + "tests/test_agent_bindings.py": 0.0, + "tests/test_agent_boundary.py": 32.3, + "tests/test_agent_control_contract.py": 0.2, + "tests/test_agent_control_envelope.py": 91.2, + "tests/test_agent_control_envelope_rows.py": 34.9, + "tests/test_agent_control_reports_dir.py": 102.8, + "tests/test_agent_controls.py": 0.0, + "tests/test_agent_handoff.py": 0.0, + "tests/test_agent_identity_origin.py": 1.1, + "tests/test_agent_instructions_apply.py": 0.2, + "tests/test_agent_instructions_renderers.py": 0.0, + "tests/test_agent_mode.py": 18.9, + "tests/test_agent_name_recovery.py": 7.7, + "tests/test_agent_protocol.py": 0.2, + "tests/test_anthropic_api.py": 0.5, + "tests/test_application_diff.py": 57.6, + "tests/test_application_diff_identity.py": 23.0, + "tests/test_application_diff_reach.py": 64.5, + "tests/test_application_diff_review.py": 40.9, + "tests/test_application_diff_unobserved.py": 230.7, + "tests/test_application_scope.py": 190.2, + "tests/test_apply_patches.py": 3.5, + "tests/test_attest.py": 0.2, + "tests/test_authorization_cli.py": 1.8, + "tests/test_authorization_execution.py": 2.1, + "tests/test_authorization_verifier_scenarios.py": 0.0, + "tests/test_authorization_verify_integration.py": 96.8, + "tests/test_base_cache_engine_identity.py": 189.8, + "tests/test_base_cache_namespace.py": 177.5, + "tests/test_baseline_integrity.py": 2.9, + "tests/test_baseline_status.py": 0.0, + "tests/test_benchmark_results_privacy.py": 0.0, + "tests/test_beta_strata_inventory.py": 2.6, + "tests/test_bootstrap.py": 43.0, + "tests/test_boundary_diff_hunks.py": 2.7, + "tests/test_boundary_diff_paths.py": 21.1, + "tests/test_capability_change_schema_hash_parity.py": 0.1, + "tests/test_capability_delta.py": 0.0, + "tests/test_capability_delta_attestation.py": 13.9, + "tests/test_capability_diff.py": 39.7, + "tests/test_capability_diff_partial_clone.py": 50.7, + "tests/test_capability_domain.py": 0.0, + "tests/test_capability_lattice.py": 0.0, + "tests/test_capability_lock.py": 0.3, + "tests/test_capability_payload.py": 8.4, + "tests/test_capability_trace_evidence.py": 0.0, + "tests/test_check_default_comparison.py": 37.7, + "tests/test_check_unmodelled_host_config_keys.py": 232.8, + "tests/test_ci.py": 0.4, + "tests/test_ci_recipes.py": 0.3, + "tests/test_claude_code_plugin_package.py": 0.0, + "tests/test_claude_hook_loading_evidence.py": 80.8, + "tests/test_claude_hooks_source.py": 0.9, + "tests/test_claude_known_marketplaces.py": 4.7, + "tests/test_claude_permission_shapes.py": 0.1, + "tests/test_cli.py": 5.7, + "tests/test_cli_on_demand_loading.py": 2.5, + "tests/test_codex_boundary_check.py": 0.9, + "tests/test_codex_plugin.py": 0.7, + "tests/test_codex_plugin_default_identity.py": 158.5, + "tests/test_codex_plugin_input_identity.py": 96.8, + "tests/test_codex_plugin_launch_package.py": 0.0, + "tests/test_cold_reader_order.py": 18.2, + "tests/test_cold_start_replay.py": 40.8, + "tests/test_conductor.py": 0.8, + "tests/test_config.py": 0.8, + "tests/test_control_packs.py": 11.1, + "tests/test_coverage_recovery.py": 16.7, + "tests/test_crewai.py": 0.1, + "tests/test_cross_block_consistency.py": 7.4, + "tests/test_current_control.py": 138.1, + "tests/test_current_control_auxiliary_inputs.py": 91.1, + "tests/test_current_control_closure.py": 75.0, + "tests/test_current_control_directory_currency.py": 92.5, + "tests/test_current_control_input_currency.py": 31.5, + "tests/test_current_control_input_origins.py": 74.6, + "tests/test_cursor_rule_globs.py": 0.0, + "tests/test_declaration_authoring.py": 0.4, + "tests/test_declaration_confirmation_route.py": 193.8, + "tests/test_declaration_drift.py": 0.1, + "tests/test_declaration_monotonicity.py": 0.3, + "tests/test_declaration_questionnaire.py": 1.7, + "tests/test_declaration_review.py": 4.6, + "tests/test_declaration_scaffold.py": 0.3, + "tests/test_declared_manifest_input_identity.py": 44.6, + "tests/test_design_partner_pilot.py": 6.5, + "tests/test_detect.py": 6.8, + "tests/test_determinism_boundary.py": 0.1, + "tests/test_diagnostics.py": 0.0, + "tests/test_diff_input_status.py": 24.9, + "tests/test_directory_reader_identity.py": 1.1, + "tests/test_discovery_scope.py": 251.9, + "tests/test_distribution_surface_parity.py": 2.1, + "tests/test_docs_links.py": 0.1, + "tests/test_documentation_checks.py": 0.0, + "tests/test_documented_claude_frontmatter.py": 0.0, + "tests/test_e3_prime_compat.py": 0.0, + "tests/test_effect_coverage.py": 0.9, + "tests/test_effect_provenance_presentation.py": 6.2, + "tests/test_enabled_plugin_hook_routing.py": 116.7, + "tests/test_environment.py": 6.8, + "tests/test_evidence_backed_pass.py": 1.4, + "tests/test_evidence_gap_ranking.py": 0.4, + "tests/test_evidence_packet.py": 19.6, + "tests/test_exec_equivalent_permissions.py": 22.1, + "tests/test_explain_finding.py": 3.9, + "tests/test_fastmcp_injection_contract.py": 0.0, + "tests/test_feedback.py": 0.0, + "tests/test_finding_attribution.py": 33.9, + "tests/test_finding_comparison_evidence.py": 0.1, + "tests/test_finding_remediation.py": 1.7, + "tests/test_findings.py": 0.0, + "tests/test_fingerprint_compatibility.py": 0.3, + "tests/test_first_adoption.py": 56.2, + "tests/test_first_look.py": 11.1, + "tests/test_fix_task_contract.py": 0.0, + "tests/test_fixture.py": 22.7, + "tests/test_fixture_no_import.py": 3.7, + "tests/test_framework_common.py": 0.0, + "tests/test_github_action_annotations.py": 0.0, + "tests/test_github_action_outputs.py": 0.0, + "tests/test_github_check_run.py": 0.0, + "tests/test_go_tool_descriptions.py": 0.1, + "tests/test_google_adk.py": 4.3, + "tests/test_governance_benchmark.py": 52.2, + "tests/test_governance_benchmark_baseline.py": 39.7, + "tests/test_guard_dependency_verification.py": 80.3, + "tests/test_headline_ranking.py": 11.5, + "tests/test_heuristics.py": 0.0, + "tests/test_hook_benign_session.py": 34.1, + "tests/test_hook_mcp_detail_fields.py": 231.0, + "tests/test_hook_script_archive.py": 3.3, + "tests/test_hook_script_capture.py": 34.6, + "tests/test_hook_script_comparison_limits.py": 66.0, + "tests/test_hook_script_currency.py": 28.1, + "tests/test_hook_script_reference.py": 0.0, + "tests/test_hook_script_routing.py": 15.4, + "tests/test_host_audit.py": 0.8, + "tests/test_host_boundary_check.py": 0.4, + "tests/test_host_boundary_unread_surfaces.py": 9.5, + "tests/test_host_change_route_parity.py": 111.8, + "tests/test_host_comparison_coverage.py": 319.0, + "tests/test_host_config_oracle_controls.py": 3.4, + "tests/test_host_config_replay.py": 54.6, + "tests/test_host_diff_entry_docs.py": 18.8, + "tests/test_host_diff_permission_direction.py": 193.1, + "tests/test_host_diff_review_changes.py": 163.4, + "tests/test_host_directory_inputs.py": 3.1, + "tests/test_host_discovery.py": 7.6, + "tests/test_host_file_links.py": 1.8, + "tests/test_host_grant_direction.py": 125.4, + "tests/test_host_input_recovery.py": 0.2, + "tests/test_host_inventory_stability.py": 0.0, + "tests/test_host_link_read_through.py": 2.0, + "tests/test_host_local_precedence.py": 0.1, + "tests/test_host_only_advisory_recipe.py": 67.6, + "tests/test_host_path_privacy.py": 5.7, + "tests/test_host_settings_narrowing_review.py": 0.4, + "tests/test_human_authorization.py": 0.1, + "tests/test_human_authorization_signature_vector.py": 0.0, + "tests/test_human_review_decision.py": 110.3, + "tests/test_human_review_presentation.py": 7.4, + "tests/test_human_review_request.py": 14.9, + "tests/test_imported_tool_bindings.py": 8.8, + "tests/test_imported_tool_review.py": 405.3, + "tests/test_init_agent_instructions.py": 9.3, + "tests/test_init_auto.py": 15.9, + "tests/test_init_ci.py": 4.4, + "tests/test_init_claude_code.py": 4.7, + "tests/test_init_gitignore.py": 3.9, + "tests/test_init_scaffold_disclosure.py": 23.7, + "tests/test_inline_hook_allow.py": 29.2, + "tests/test_inputs.py": 0.2, + "tests/test_install_hooks.py": 81.1, + "tests/test_instruction_structure.py": 0.1, + "tests/test_instruction_structure_contracts.py": 4.3, + "tests/test_instruction_structure_hooks.py": 19.2, + "tests/test_instruction_structure_workflow.py": 35.1, + "tests/test_invocation_policy.py": 92.6, + "tests/test_labeling_guide_is_rater_safe.py": 0.0, + "tests/test_langchain.py": 0.1, + "tests/test_large_sample.py": 1.1, + "tests/test_linked_unchanged_limits.py": 182.2, + "tests/test_live_workspace_cause.py": 67.9, + "tests/test_local_contract.py": 0.0, + "tests/test_local_review.py": 23.2, + "tests/test_managed_block.py": 0.0, + "tests/test_manifest_consistency.py": 0.1, + "tests/test_manifest_free_pr_rows.py": 121.4, + "tests/test_manifest_schema_parity.py": 0.0, + "tests/test_manifest_scope.py": 0.0, + "tests/test_mcp_audit.py": 0.1, + "tests/test_mcp_idioms.py": 0.0, + "tests/test_mcp_launch_source.py": 39.2, + "tests/test_mcp_manifest.py": 0.2, + "tests/test_mcp_permissions.py": 0.2, + "tests/test_mcp_server.py": 0.1, + "tests/test_mcp_server_findings_table.py": 0.0, + "tests/test_mcp_server_source.py": 1.3, + "tests/test_mcp_source_annotations.py": 0.4, + "tests/test_mcp_url_capability_digest.py": 10.0, + "tests/test_metadata_loader.py": 0.1, + "tests/test_miner.py": 22.8, + "tests/test_miner_candidates.py": 1.3, + "tests/test_miner_constructed.py": 28.4, + "tests/test_miner_corpus.py": 0.1, + "tests/test_miner_labels.py": 0.0, + "tests/test_miner_reevaluate.py": 11.4, + "tests/test_miner_scope_inputs.py": 9.8, + "tests/test_n8n.py": 2.0, + "tests/test_next_action_chains_terminate.py": 50.3, + "tests/test_no_heuristics.py": 6.5, + "tests/test_nonregular_evidence_readers.py": 5.2, + "tests/test_openai_api.py": 0.4, + "tests/test_openapi_fuzz.py": 0.1, + "tests/test_openapi_operation_attribution.py": 2.0, + "tests/test_operation_attribution_verification.py": 21.4, + "tests/test_org_governance.py": 0.1, + "tests/test_out_path_resolution.py": 52.2, + "tests/test_output_directory_content.py": 382.7, + "tests/test_p0_binding_canaries.py": 0.1, + "tests/test_p0_policy_evidence_canaries.py": 0.1, + "tests/test_p0_safety_canaries.py": 2.6, + "tests/test_p0_verification_identity_canaries.py": 0.2, + "tests/test_packaging.py": 14.6, + "tests/test_partial_host_comparison.py": 83.3, + "tests/test_patch_generators.py": 1.1, + "tests/test_patches_model.py": 0.3, + "tests/test_permission_lattice.py": 1.1, + "tests/test_permission_residual.py": 8.2, + "tests/test_permission_review_guidance.py": 14.3, + "tests/test_plugin_validation.py": 4.5, + "tests/test_plugins.py": 0.6, + "tests/test_policy_evidence_architecture.py": 0.0, + "tests/test_policy_packs.py": 1.1, + "tests/test_policy_reason_code_split.py": 15.5, + "tests/test_preflight.py": 1.6, + "tests/test_preview_control_currency.py": 133.0, + "tests/test_privacy.py": 0.1, + "tests/test_product_hardening_gap_closure.py": 0.0, + "tests/test_prompt_disabling_settings.py": 217.4, + "tests/test_prompt_parity.py": 0.0, + "tests/test_property_loaders.py": 0.8, + "tests/test_provenance_kind.py": 2.7, + "tests/test_public_surface_contract.py": 2.9, + "tests/test_python_import_resolution.py": 0.3, + "tests/test_qualification_coverage_misses.py": 1.3, + "tests/test_rater_harness.py": 79.7, + "tests/test_reader_vocabulary.py": 21.7, + "tests/test_regenerate_goldens.py": 17.2, + "tests/test_registry.py": 0.0, + "tests/test_registry_ledger.py": 0.1, + "tests/test_release_advisory_pipeline.py": 3.4, + "tests/test_release_channel.py": 0.1, + "tests/test_release_channel_cadence.py": 4.9, + "tests/test_release_decision.py": 0.0, + "tests/test_release_engine_smoke.py": 0.3, + "tests/test_release_pipeline.py": 5.7, + "tests/test_release_source.py": 5.1, + "tests/test_remediation_metadata.py": 0.0, + "tests/test_report_1_0_compatibility_fixtures.py": 4.8, + "tests/test_report_1_0_contract.py": 0.0, + "tests/test_reports.py": 16.3, + "tests/test_required_source_availability.py": 5.7, + "tests/test_reusable_workflow_secret_mappings.py": 177.3, + "tests/test_reviewer_summary.py": 0.0, + "tests/test_risk_hints.py": 0.0, + "tests/test_safety_qualification.py": 2.2, + "tests/test_safety_qualification_release.py": 1.4, + "tests/test_sarif.py": 0.3, + "tests/test_scan.py": 7.2, + "tests/test_scan_context.py": 0.1, + "tests/test_scenario_suggest.py": 2.4, + "tests/test_schema_boundaries.py": 3.5, + "tests/test_schema_roundtrip.py": 2.0, + "tests/test_scoped_base_tree.py": 13.4, + "tests/test_sdk_boolean_source.py": 1.3, + "tests/test_sdk_guard_dependencies.py": 0.3, + "tests/test_self_approval_signal.py": 0.0, + "tests/test_self_check.py": 1.6, + "tests/test_semantic_assessment.py": 0.0, + "tests/test_semantic_cold_start.py": 0.1, + "tests/test_semantic_properties.py": 0.2, + "tests/test_setup_control.py": 37.3, + "tests/test_severity_override_floor.py": 0.0, + "tests/test_shard_partition.py": 32.9, + "tests/test_skill_review.py": 0.1, + "tests/test_source_authority.py": 3.8, + "tests/test_source_binding.py": 2.2, + "tests/test_source_head_identity.py": 7.1, + "tests/test_source_provenance.py": 0.7, + "tests/test_static_inputs.py": 0.0, + "tests/test_strata_inventory.py": 4.2, + "tests/test_subject_rollup.py": 1.4, + "tests/test_surface_exclusions.py": 12.1, + "tests/test_three_command_flow.py": 1.1, + "tests/test_tool_identity.py": 0.0, + "tests/test_tool_surface_diff.py": 0.0, + "tests/test_toolkit_bounds_check.py": 1.0, + "tests/test_trigger_command.py": 1.5, + "tests/test_trust_root.py": 1.3, + "tests/test_ts_tool_descriptions.py": 0.1, + "tests/test_unchanged_host_limits.py": 18.9, + "tests/test_unread_changed_inputs.py": 150.9, + "tests/test_unresolved_descriptions.py": 0.1, + "tests/test_v07_metadata_roundtrip.py": 1.6, + "tests/test_validation_evidence.py": 1.1, + "tests/test_verdict_contract.py": 0.0, + "tests/test_verification_git_snapshot.py": 3.2, + "tests/test_verifier_blocks.py": 0.6, + "tests/test_verifier_control_contract.py": 1.1, + "tests/test_verifier_scenarios.py": 28.8, + "tests/test_verify.py": 115.3, + "tests/test_verify_auto_base.py": 14.5, + "tests/test_verify_capability_scope.py": 0.7, + "tests/test_verify_config_binding.py": 0.7, + "tests/test_verify_orchestrator.py": 45.4, + "tests/test_verify_run.py": 0.0, + "tests/test_verify_weakening.py": 21.8, + "tests/test_vscode_mcp_jsonc.py": 0.0, + "tests/test_vscode_mcp_support.py": 0.0, + "tests/test_wheel_candidate_build.py": 2.1, + "tests/test_workflow_agent_launches.py": 79.2, + "tests/test_workflow_capability_diff.py": 3.1, + "tests/test_workflow_evidence.py": 0.1, + "tests/test_workflow_label_redaction.py": 55.4, + "tests/test_workflow_step_action_references.py": 42.6, + "tests/test_workspace_input_guard.py": 0.3, + "tests/test_zero_install_detector.py": 38.4 + }, + "measured_at": "e672648fc800" +} diff --git a/tests/test_shard_partition.py b/tests/test_shard_partition.py index 969e81029..03823a66f 100644 --- a/tests/test_shard_partition.py +++ b/tests/test_shard_partition.py @@ -17,7 +17,7 @@ import pytest -from ci_sharding import shard_assignment +from ci_sharding import SECONDS_FILE, load_seconds, shard_assignment REPO_ROOT = Path(__file__).resolve().parent.parent @@ -82,6 +82,58 @@ def test_a_dominant_file_does_not_leave_a_shard_idle() -> None: assert max(weighted.values()) / total < 0.55 +def test_measured_seconds_outweigh_item_counts() -> None: + """A slow file of few tests is balanced by its time, not its count (#904). + + By count, the 20-item file below is a twentieth of the 400-item one and + shares a shard. Measured, it takes most of the suite's time, and it gets a + shard to itself. + """ + + def sharing(owner: dict[str, int]) -> list[str]: + return [path for path, shard in owner.items() if shard == owner["tests/test_small_a.py"]] + + seconds = {"tests/test_small_a.py": 600.0, "tests/test_huge.py": 60.0, "tests/test_large.py": 60.0} + assert sharing(shard_assignment(_COLLECTION, 2)) != ["tests/test_small_a.py"] + assert sharing(shard_assignment(_COLLECTION, 2, seconds)) == ["tests/test_small_a.py"] + + +def test_an_unmeasured_file_costs_its_items_at_the_measured_rate() -> None: + """A new file weighs what its items would at the suite's seconds per item.""" + + seconds = {"tests/test_huge.py": 400.0} # one second per item + owner = shard_assignment({"tests/test_huge.py": 400, "tests/test_new.py": 400}, 2, seconds) + assert owner["tests/test_huge.py"] != owner["tests/test_new.py"] + + +@pytest.mark.parametrize("shards", [2, 3, 4]) +def test_the_timed_assignment_is_deterministic(shards: int) -> None: + seconds = {path: count * 0.37 for path, count in _COLLECTION.items() if "tiny" not in path} + first = shard_assignment(_COLLECTION, shards, seconds) + reordered = dict(reversed(list(_COLLECTION.items()))) + assert shard_assignment(reordered, shards, dict(reversed(list(seconds.items())))) == first + assert set(first) == set(_COLLECTION) + + +def test_a_missing_or_malformed_measurement_balances_by_count(tmp_path: Path) -> None: + assert load_seconds(tmp_path / "absent.json") == {} + broken = tmp_path / "broken.json" + broken.write_text("{not json") + assert load_seconds(broken) == {} + odd = tmp_path / "odd.json" + odd.write_text('{"files": {"tests/a.py": 3, "tests/b.py": -1, "tests/c.py": true, "tests/d.py": "x"}}') + assert load_seconds(odd) == {"tests/a.py": 3.0} + + +def test_the_committed_measurement_names_test_files() -> None: + """The measurement is a weight per test file, and every weight is usable.""" + + seconds = load_seconds(SECONDS_FILE) + assert seconds, f"{SECONDS_FILE} holds no measurement" + assert all(re.fullmatch(r"tests/\S+\.py", path) for path in seconds) + assert all(value >= 0 for value in seconds.values()) + + def _collect(shard: int | None, shards: int | None) -> dict[str, int]: """Collect the CI suite and return ``{file: item count}``.