From 9b78e525319dad969cc27d27ea493712d5dfc0a9 Mon Sep 17 00:00:00 2001 From: "agents-workflows-bot[bot]" <251285186+agents-workflows-bot[bot]@users.noreply.github.com> Date: Tue, 25 Aug 2026 17:35:44 +0000 Subject: [PATCH] chore: sync workflow templates from Workflows repo Automated sync from stranske/Workflows Template hash: 25782af9864e Changes synced from sync-manifest.yml Consumer-sync plan ID: sha256:25782af9864ed8fea0dbe923cf303e0cc14e72268afa8e83a47f7ac5c85954e1 Plan scope: source-delta Scope base SHA: 1f060bc6bd7aea61dd974e261506e529ff0f7aa3 Source commit: 2ca9251a5e17a7e69e5f41e2d439aa8b189d8927 Canary evidence JSON (base64): eyJzY2hlbWEiOiJ3b3JrZmxvd3MuY29uc3VtZXItc3luYy1jYW5hcnktZXZpZGVuY2UvdjEiLCJ2ZXJzaW9uIjoxLCJyZXN1bHRzIjpbeyJyZXBvIjoic3RyYW5za2UvVHJhdmVsLVBsYW4tUGVybWlzc2lvbiIsInBsYW5faWQiOiJzaGEyNTY6MjU3ODJhZjk4NjRlZDhmZWEwZGJlOTIzY2YzMDNlMGNjMTRlNzIyNjhhZmE4ZTgzYTQ3ZjdhYzVjODU5NTRlMSIsInBsYW5fc2NvcGUiOiJzb3VyY2UtZGVsdGEiLCJzY29wZV9iYXNlX3NoYSI6IjFmMDYwYmM2YmQ3YWVhNjFkZDk3NGUyNjE1MDZlNTI5ZmYwZjdhYTMiLCJzb3VyY2VfY29tbWl0IjoiMmNhOTI1MWE1ZTE3YTdlNjllNWY0MWUyZDQzOWFhOGIxODlkODkyNyIsInByIjoxNDc4LCJoZWFkX3NoYSI6IjJhODY4OWFkOTBhMmNjYzQ5OWIwNDdjYjYxOTA4OGQxMzYzNDg4NDAiLCJldmlkZW5jZV9zb3VyY2UiOiJvcGVuLWNhbmRpZGF0ZSIsInJlcXVpcmVkX2NoZWNrX3N0YXRlIjoic3VjY2VzcyIsImFjdGl2ZV9yZXZpZXdfdGhyZWFkX2NvdW50IjowfSx7InJlcG8iOiJzdHJhbnNrZS90cmlwLXBsYW5uZXIiLCJwbGFuX2lkIjoic2hhMjU2OjI1NzgyYWY5ODY0ZWQ4ZmVhMGRiZTkyM2NmMzAzZTBjYzE0ZTcyMjY4YWZhOGU4M2E0N2Y3YWM1Yzg1OTU0ZTEiLCJwbGFuX3Njb3BlIjoic291cmNlLWRlbHRhIiwic2NvcGVfYmFzZV9zaGEiOiIxZjA2MGJjNmJkN2FlYTYxZGQ5NzRlMjYxNTA2ZTUyOWZmMGY3YWEzIiwic291cmNlX2NvbW1pdCI6IjJjYTkyNTFhNWUxN2E3ZTY5ZTVmNDFlMmQ0MzlhYThiMTg5ZDg5MjciLCJwciI6MTc2NiwiaGVhZF9zaGEiOiI0MDZlMGFjNzQ5MTIyMmEzZGVhZmE5YjBkYjNmYmMwNzM4NTU3ZWJlIiwiZXZpZGVuY2Vfc291cmNlIjoib3Blbi1jYW5kaWRhdGUiLCJyZXF1aXJlZF9jaGVja19zdGF0ZSI6InN1Y2Nlc3MiLCJhY3RpdmVfcmV2aWV3X3RocmVhZF9jb3VudCI6MH0seyJyZXBvIjoic3RyYW5za2UvUG9ydGFibGUtQWxwaGEtRXh0ZW5zaW9uLU1vZGVsIiwicGxhbl9pZCI6InNoYTI1NjoyNTc4MmFmOTg2NGVkOGZlYTBkYmU5MjNjZjMwM2UwY2MxNGU3MjI2OGFmYThlODNhNDdmN2FjNWM4NTk1NGUxIiwicGxhbl9zY29wZSI6InNvdXJjZS1kZWx0YSIsInNjb3BlX2Jhc2Vfc2hhIjoiMWYwNjBiYzZiZDdhZWE2MWRkOTc0ZTI2MTUwNmU1MjlmZjBmN2FhMyIsInNvdXJjZV9jb21taXQiOiIyY2E5MjUxYTVlMTdhN2U2OWU1ZjQxZTJkNDM5YWE4YjE4OWQ4OTI3IiwicHIiOjIyNDQsImhlYWRfc2hhIjoiYmE0ZDk5NDM3NTk1YTc4MjAwYjNlNGRhYmRiMzlmMWVjMDg4MzI0MCIsImV2aWRlbmNlX3NvdXJjZSI6Im9wZW4tY2FuZGlkYXRlIiwicmVxdWlyZWRfY2hlY2tfc3RhdGUiOiJzdWNjZXNzIiwiYWN0aXZlX3Jldmlld190aHJlYWRfY291bnQiOjB9XX0= --- tools/coverage_trend.py | 215 +++++++++++++++++++++++++++++++++++++--- 1 file changed, 199 insertions(+), 16 deletions(-) diff --git a/tools/coverage_trend.py b/tools/coverage_trend.py index 523fa213f4..2d4727fc54 100755 --- a/tools/coverage_trend.py +++ b/tools/coverage_trend.py @@ -25,6 +25,107 @@ def _load_json(path: Path) -> dict[str, Any]: return {} +# The keys a `config/coverage-baseline.json` may use for the percentage. BOTH are real and +# both are in service today: stranske/Workflows and stranske/learning-management-system write +# `coverage`, stranske/Trend_Model_Project writes `line`. `tools/coverage_guard.py` -- the OTHER +# consumer of the very same file -- has always read `payload.get("line", payload.get("coverage"))`, +# so the two scripts disagreed about their shared config and only one of them said so. +BASELINE_KEYS = ("line", "coverage") + + +# Why `baseline` is Optional and never 0.0: this function used to return a bare float, defaulting +# to 0.0 whenever the file was missing, unparseable, or keyed differently. That made ONE sentinel +# mean two opposite things -- "there is no baseline to compare against" and "the baseline is zero" +# -- and only the second is a measurement. The visible result was a summary reading +# `Baseline 0.00% | Delta +83.32% | Status Pass` on a repo whose baseline file says 85.0 and whose +# real coverage of 83.32% is a BREACH. A comparison against a silently-absent baseline cannot fail, +# so it was reporting success at the exact moment it had nothing to report. +def _resolve_baseline(path: Path | None) -> tuple[float | None, str]: + """Return (baseline, status). `None` means NOT CONFIGURED -- never the number zero. + + status is one of: ok, unset, absent, unreadable, no_recognised_key. Each names a different + fix, which is the whole point of not collapsing them: `absent` means write the file, + `no_recognised_key` means it is there and this script cannot read it. + """ + if path is None: + return None, "unset" + if not path.exists(): + return None, "absent" + payload = _load_json(path) + if not payload: + return None, "unreadable" + for key in BASELINE_KEYS: + if key in payload: + try: + return float(payload[key]), "ok" + except (TypeError, ValueError): + return None, "unreadable" + return None, "no_recognised_key" + + +def _partition_files( + files: dict[str, Any], project_root: Path | None +) -> tuple[dict[str, Any], list[str]]: + """Split coverage rows into rows inside the project and rows measured somewhere else. + + A test that copies the source tree into a tmpdir and exercises the copy makes coverage.py + record BOTH: the real module under its relative path, and the copy under an absolute + `/tmp/...` path. The copy is barely executed, so it sorts to the top of every hotspot table + and is counted a second time in the total. Observed on stranske/Trend_Model_Project, where 13 + of the 15 reported worst files were + `/tmp/pytest-of-runner/pytest-0/popen-gw0/test_autofix_pipeline_repairs_0/workspace/src/...` + -- paths that do not exist in the repository, so the one actionable output of this script + pointed at files nobody could open. + + Relative paths are always in-project: coverage.py records repo files relative to the run root, + so an ABSOLUTE path outside that root is the tell. Reported, never silently dropped from the + headline number -- see `current_project_only`. + """ + if project_root is None: + return files, [] + root = project_root.resolve() + inside: dict[str, Any] = {} + foreign: list[str] = [] + for filepath, data in files.items(): + candidate = Path(filepath) + if candidate.is_absolute() and not candidate.resolve().is_relative_to(root): + foreign.append(filepath) + else: + inside[filepath] = data + return inside, foreign + + +def _percent_from_rows(files: dict[str, Any]) -> float | None: + """Recompute a coverage percentage from per-file rows, or None when there are none. + + ON THE SAME BASIS AS `current`, which is coverage.py's `totals.percent_covered`. With branch + coverage enabled that is (covered_lines + covered_branches) over (statements + branches), NOT + statements alone -- coverage.py reports the statements-only figure separately, as + `percent_statements_covered`. + + The first version summed lines only, and the failure did not look like a failure: it produced + a plausible number about three points away, so `current_project_only` read as though + contamination had cost three points on repos where NOTHING was contaminated. Measured on + stranske/Pension-Data (foreign_file_count 0): statements-only 90.96%, line+branch 87.78%, and + coverage.py's own percent_covered 87.78%. The invariant that catches it is cheap and is now a + test -- with no foreign rows this MUST equal `current`. + """ + covered = 0 + missing = 0 + covered_branches = 0 + num_branches = 0 + for data in files.values(): + summary = data.get("summary", {}) if isinstance(data, dict) else {} + covered += int(summary.get("covered_lines", 0) or 0) + missing += int(summary.get("missing_lines", 0) or 0) + covered_branches += int(summary.get("covered_branches", 0) or 0) + num_branches += int(summary.get("num_branches", 0) or 0) + denominator = covered + missing + num_branches + if denominator <= 0: + return None + return (covered + covered_branches) / denominator * 100.0 + + def _extract_coverage_percent(coverage_json: dict[str, Any]) -> float: """Extract overall coverage percentage from coverage.json.""" totals = coverage_json.get("totals", {}) @@ -35,13 +136,19 @@ def _get_hotspots( coverage_json: dict[str, Any], limit: int = 15, low_threshold: float = 50.0, + project_root: Path | None = None, ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: """Extract hotspot files from coverage.json. + `project_root` defaults to None, which keeps every row -- the behaviour callers had before + contamination filtering existed. Pass a root to drop rows measured outside the project. + The two-value return shape is deliberately preserved because consumer repositories call this + helper directly; callers that need the excluded paths can use ``_partition_files``. + Returns: Tuple of (all_hotspots sorted by coverage, low_coverage_files below threshold) """ - files = coverage_json.get("files", {}) + files, _foreign = _partition_files(coverage_json.get("files", {}), project_root) all_files = [] for filepath, data in files.items(): @@ -93,6 +200,15 @@ def main(args: list[str] | None = None) -> int: parser.add_argument("--coverage-xml", type=Path, help="Path to coverage.xml") parser.add_argument("--coverage-json", type=Path, help="Path to coverage.json") parser.add_argument("--baseline", type=Path, help="Path to baseline JSON") + parser.add_argument( + "--project-root", + type=Path, + default=Path.cwd(), + help=( + "Project root used to tell repository files from files a test copied elsewhere " + "(coverage rows with an absolute path outside this root). Defaults to the cwd." + ), + ) parser.add_argument("--summary-path", type=Path, help="Path to output summary markdown") parser.add_argument("--job-summary", type=Path, help="Path to GITHUB_STEP_SUMMARY") parser.add_argument("--artifact-path", type=Path, help="Path to output trend artifact") @@ -112,14 +228,12 @@ def main(args: list[str] | None = None) -> int: coverage_data = _load_json(parsed.coverage_json) current_coverage = _extract_coverage_percent(coverage_data) - # Load baseline - baseline_coverage = 0.0 - if parsed.baseline and parsed.baseline.exists(): - baseline_data = _load_json(parsed.baseline) - baseline_coverage = float(baseline_data.get("coverage", 0.0)) + # Load baseline. None means not configured; it is NEVER swapped for 0.0 (see _resolve_baseline). + baseline_coverage, baseline_status = _resolve_baseline(parsed.baseline) - # Calculate delta - delta = current_coverage - baseline_coverage + # Calculate delta. Absent baseline -> absent delta, rather than a delta against zero that + # renders as a large improvement on every run. + delta = None if baseline_coverage is None else current_coverage - baseline_coverage passes_minimum = current_coverage >= parsed.minimum # Get hotspots @@ -127,17 +241,29 @@ def main(args: list[str] | None = None) -> int: coverage_data, limit=parsed.hotspot_limit, low_threshold=parsed.low_threshold, + project_root=parsed.project_root, + ) + project_files, foreign_files = _partition_files( + coverage_data.get("files", {}), parsed.project_root ) + project_only = _percent_from_rows(project_files) # Generate trend record trend_record = { "current": current_coverage, "baseline": baseline_coverage, + "baseline_status": baseline_status, "delta": delta, "minimum": parsed.minimum, "passes_minimum": passes_minimum, "hotspot_count": len(hotspots), "low_coverage_count": len(low_coverage), + # Contamination reporting. `current` stays exactly as coverage.py computed it, so this + # record still agrees with coverage.xml and with the delta job; the project-only figure + # sits BESIDE it, and the gap between the two is the size of the problem. + "foreign_file_count": len(foreign_files), + "foreign_files": foreign_files[:10], + "current_project_only": project_only, } # Write outputs @@ -152,18 +278,55 @@ def main(args: list[str] | None = None) -> int: parsed.artifact_path.write_text(json.dumps(artifact_data, indent=2), encoding="utf-8") status = "✅ Pass" if passes_minimum else "❌ Below minimum" + if baseline_coverage is None: + baseline_cell = f"⚠️ not configured ({baseline_status})" + delta_cell = "n/a — nothing to compare against" + else: + baseline_cell = f"{baseline_coverage:.2f}%" + delta_cell = f"{delta:+.2f}%" summary = f"""## Coverage Trend | Metric | Value | |--------|-------| | Current | {current_coverage:.2f}% | -| Baseline | {baseline_coverage:.2f}% | -| Delta | {delta:+.2f}% | +| Baseline | {baseline_cell} | +| Delta | {delta_cell} | | Minimum | {parsed.minimum:.2f}% | | Status | {status} | """ + if baseline_coverage is None: + summary += ( + f"> **No baseline was read ({baseline_status}), so the delta above is not a " + "measurement.** `Status` reflects only the `--minimum` floor. Write " + "`config/coverage-baseline.json` with a `line` or `coverage` percentage to enable " + "the comparison; that file is deliberately not synced from Workflows, so each repo " + "owns its own.\n\n" + ) + + if foreign_files and not hotspots: + summary += ( + f"> **⚠️ EVERY coverage row ({len(foreign_files)}) is outside `{parsed.project_root}`, " + "so nothing could be attributed to this project.** That is almost certainly a wrong " + "`--project-root`, not a contaminated test run — a real fixture copy leaves the " + "genuine rows behind. Check the working directory the reporter runs in before reading " + "anything below.\n\n" + ) + elif foreign_files: + shown = "\n".join(f"> - `{path}`" for path in foreign_files[:5]) + more = f"\n> - …and {len(foreign_files) - 5} more" if len(foreign_files) > 5 else "" + project_cell = f"{project_only:.2f}%" if project_only is not None else "not computable" + summary += ( + f"> **⚠️ {len(foreign_files)} file(s) were measured OUTSIDE the project root, so " + f"`Current` above is contaminated.** A test is copying the source tree somewhere " + f"else and exercising the copy, so those modules are counted twice — once as " + f"themselves and once as a barely-executed duplicate. Project-only coverage is " + f"**{project_cell}**. They are excluded from the tables below because their paths do " + f"not exist in the repository. Fix by adding the copy location to " + f"`[tool.coverage.run] omit`.\n{shown}{more}\n\n" + ) + # Add hotspot tables if we have coverage data if hotspots: summary += _format_hotspot_table(hotspots, "Top Coverage Hotspots (lowest coverage)") @@ -185,17 +348,37 @@ def main(args: list[str] | None = None) -> int: if parsed.github_output: parsed.github_output.parent.mkdir(parents=True, exist_ok=True) with parsed.github_output.open("w", encoding="utf-8") as f: + # Plain locals rather than f-strings nested inside f-strings: the nested form is + # valid only from 3.12 (PEP 701) and this file ships to 3.12 AND 3.13 runners. + baseline_out = "" if baseline_coverage is None else f"{baseline_coverage:.2f}" + delta_out = "" if delta is None else f"{delta:.2f}" f.write(f"coverage={current_coverage:.2f}\n") - f.write(f"baseline={baseline_coverage:.2f}\n") - f.write(f"delta={delta:.2f}\n") + # Empty, not 0.00, when there is no baseline: a reader testing `-n "$baseline"` then + # sees the difference, where 0.00 is indistinguishable from a real measurement. + f.write(f"baseline={baseline_out}\n") + f.write(f"baseline_status={baseline_status}\n") + f.write(f"delta={delta_out}\n") f.write(f"passes_minimum={'true' if passes_minimum else 'false'}\n") f.write(f"hotspot_count={len(hotspots)}\n") f.write(f"low_coverage_count={len(low_coverage)}\n") + f.write(f"foreign_file_count={len(foreign_files)}\n") - print( - f"Coverage: {current_coverage:.2f}% " - f"(baseline: {baseline_coverage:.2f}%, delta: {delta:+.2f}%)" - ) + if baseline_coverage is None: + print( + f"Coverage: {current_coverage:.2f}% " + f"(no baseline: {baseline_status} — delta not computed)" + ) + else: + print( + f"Coverage: {current_coverage:.2f}% " + f"(baseline: {baseline_coverage:.2f}%, delta: {delta:+.2f}%)" + ) + if foreign_files: + project_only_out = "n/a" if project_only is None else f"{project_only:.2f}%" + print( + f"WARNING: {len(foreign_files)} file(s) measured outside the project root; " + f"`Current` is contaminated (project-only: {project_only_out})" + ) if hotspots: print(f"Hotspots: {len(hotspots)} files with lowest coverage")