From 302f160ac6fdaa26a6e9dad8a9dec3a50a2db95d Mon Sep 17 00:00:00 2001 From: Babissimo Date: Wed, 23 Sep 2026 17:41:16 +0100 Subject: [PATCH] Add a ci-status skill that reports a PR's real CI state gh pr checks and gh run watch have each reported green for a PR that was untested, conflicting, failed or unreviewed. On retina-server that happened at least eight ways, each kept as a separate memory note that worktree sessions never load, so every session rediscovered the trap it hit. ci_status.py judges the runs for the PR's recorded head instead of the checks table. It sets aside runs whose jobs all skipped (a title or body edit), cancelled duplicates (a stack push) and non-gating events; lets a newer run supersede an older one job by job, so an edit's run cannot hide tests that failed before it; takes each run's latest attempt while listing failed earlier ones (a rerun overwrites the conclusion); decides by job conclusions rather than the run's; reports a conflicting PR as never going to run; checks the PR's head against its branch on GitHub and the local branch; counts other apps' checks and commit statuses; and counts the Claude review only when a bot comment from the latest attempt of this head's review run links back to it with its checklist ticked. It also says when the head's copy of the review workflow differs from the default branch's, since the action then skips itself without a word until the branch is rebased. What may yet arrive (a run, the review run, the head following its branch) is 'not settled' in one reading, because GitHub records no push time to judge the wait against; --watch calls it final once it has seen it last a few minutes. The exit status separates failed (1) from a wrong question (2), not settled (3) and GitHub being unreadable (4), so neither a 502 nor a typo reads as a red build. It lives here rather than in one repo because every repo scaffolded by setup-repo carries the same review workflow and the same gh habits. Its tests run in a workflow of their own so a failure is labelled as theirs. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ci-status-tests.yml | 32 + .gitignore | 3 + README.md | 1 + plugins/core/.claude-plugin/plugin.json | 2 +- plugins/core/skills/ci-status/SKILL.md | 64 + .../skills/ci-status/scripts/ci_status.py | 883 +++++++++++ tests/ci-status/test_ci_status.py | 1341 +++++++++++++++++ 7 files changed, 2325 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/ci-status-tests.yml create mode 100644 plugins/core/skills/ci-status/SKILL.md create mode 100755 plugins/core/skills/ci-status/scripts/ci_status.py create mode 100644 tests/ci-status/test_ci_status.py diff --git a/.github/workflows/ci-status-tests.yml b/.github/workflows/ci-status-tests.yml new file mode 100644 index 0000000..4c218d6 --- /dev/null +++ b/.github/workflows/ci-status-tests.yml @@ -0,0 +1,32 @@ +# Runs the ci-status skill's tests. They need only pytest: the script under test +# calls gh, and the tests stand a fake in its place. +name: ci-status tests + +on: + push: + branches: [main] + pull_request: + +permissions: + contents: read + +jobs: + ci-status: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Install uv + uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5.4.2 + + - name: Install pytest + run: uv pip install --system pytest + + - name: Run the tests + run: python -m pytest tests/ci-status diff --git a/.gitignore b/.gitignore index 5000dfc..49df947 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,6 @@ # Superpowers planning artifacts and subagent-driven-development scratch docs/superpowers/ .superpowers/ + +# Bytecode from running the Python skills and their tests +__pycache__/ diff --git a/README.md b/README.md index 31bcbe4..20c0be4 100644 --- a/README.md +++ b/README.md @@ -4,6 +4,7 @@ Offworld Labs' org-wide Claude Code resource: a **plugin marketplace** (`offworl plus **shared reference docs** used across every repo in the organisation. - `plugins/core` — the `core` plugin; its `setup-repo` skill bundles the shared rules, `.claude/settings.json`, `CLAUDE.md`, and CI workflow templates used to scaffold new repos. +- `plugins/core/skills/ci-status`: reports a pull request's real CI state, where `gh pr checks` and `gh run watch` misreport it. - `docs/` — on-demand org-wide reference docs (see [Documentation](#documentation)). ## Install diff --git a/plugins/core/.claude-plugin/plugin.json b/plugins/core/.claude-plugin/plugin.json index 1a1a95a..2a1dedf 100644 --- a/plugins/core/.claude-plugin/plugin.json +++ b/plugins/core/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "core", - "version": "0.6.0", + "version": "0.7.0", "description": "Core Offworld Labs skills, commands, agents, and hooks shared across all repos.", "author": { "name": "Offworld Labs" diff --git a/plugins/core/skills/ci-status/SKILL.md b/plugins/core/skills/ci-status/SKILL.md new file mode 100644 index 0000000..b2521c0 --- /dev/null +++ b/plugins/core/skills/ci-status/SKILL.md @@ -0,0 +1,64 @@ +--- +name: ci-status +description: Use whenever you need to know whether a pull request's CI passed, to wait for CI after a push, or to judge the run a merge to main started. Replaces `gh pr checks`, `gh pr checks --watch` and `gh run watch`, whose exit codes and tables report skipped, superseded and not-yet-started runs as passes. +--- + +# ci-status + +Reports a PR's real CI state from the runs for its head commit, and exits +non-zero unless every gate passed. The script's docstring lists the traps it +handles; the short version is that `gh pr checks` and `gh run watch` have each +reported green for a PR that was untested, conflicting, failed or unreviewed. + +`CI_STATUS="${CLAUDE_PLUGIN_ROOT}/skills/ci-status/scripts/ci_status.py"` + +## Use + +``` +python3 "$CI_STATUS" -R / # one reading +python3 "$CI_STATUS" -R / --watch # poll until it settles +python3 "$CI_STATUS" --commit -R / # a merge's push run +``` + +- Inside a clone of the repo, `-R` and the PR number can be left out; the PR + defaults to the current branch's. A fork's clone resolves to whichever repo + `gh` is set to there (often the upstream), so pass `-R` in one. +- `--watch` polls every 30 s for up to 60 min. Run it in the background (it + outlives a foreground command's limit) and act on its exit status. +- `-v` lists every job rather than only those not passing. +- `--no-review` drops the Claude review from the gate. Use it only when a human + has reviewed the PR instead. + +## Reading the answer + +| Exit | Meaning | What to do | +|------|---------|------------| +| 0 | Every gate passed on the PR's head | Read the review it links, if any: green is not "no findings" | +| 1 | Failed, or will not run until someone acts | Act on the reason it prints (below) | +| 2 | The question was wrong: no such PR, no PR for this branch, not a GitHub repo, or no access | Fix the arguments | +| 3 | Not settled: runs still going, or something not arrived yet (a run, the review, the head following its branch) | Ask again, or `--watch`, which calls a wait final after a few minutes | +| 4 | GitHub could not be read, so there is no verdict | Ask again; it says nothing about CI | + +Checks from other apps and commit statuses count as gates beside the Actions +runs. + +The reasons that need action: + +- **A job failed.** Fix it, or report it as a flake. Do not rerun silently: the + rerun overwrites the run's conclusion (ci-status still lists the failed + attempt). +- **The PR conflicts.** GitHub makes no merge ref, so no CI will run. Rebase. +- **Every run skipped all its jobs.** Nothing was tested, usually because the + PR is a draft or a filter excluded the change. +- **The PR's head has not followed its branch.** No CI runs for the new commit. + Amend to a new SHA (`git commit --amend --no-edit`) and push with + `--force-with-lease`. +- **The review bot did not review.** The PR edits the review workflow (review it + by hand), the branch carries an older copy of that workflow than the default + branch (rebase onto it), or the workflow did not trigger for this head. Report + any of them as "not reviewed", never as a clean review. +- **The review stopped partway.** Rerun it with the command it prints, + `gh run rerun `. `--failed` reruns nothing when the job itself passed, + which it does unless the repo's workflow checks the review finished. + +Skipped jobs inside a run that ran are neutral: deploy jobs skip on every PR. diff --git a/plugins/core/skills/ci-status/scripts/ci_status.py b/plugins/core/skills/ci-status/scripts/ci_status.py new file mode 100755 index 0000000..6f90c1c --- /dev/null +++ b/plugins/core/skills/ci-status/scripts/ci_status.py @@ -0,0 +1,883 @@ +#!/usr/bin/env python3 +"""Report a pull request's real CI state, and exit non-zero unless every gate passed. + + ci_status.py [PR] [-R owner/repo] [--watch] [--no-review] [-v] + ci_status.py --commit SHA [-R owner/repo] [--watch] [-v] + +`gh pr checks` and `gh run watch` answer a narrower question than they appear to, +and each of these has passed for green: + +* Just after a push, no run is registered yet, and `gh pr checks --watch` prints + "no checks reported" and exits 0. Here that is "not settled", never a pass. +* A PR whose branch conflicts has no merge ref, so pull_request CI never runs. + That is reported as a failure, since nothing will change until a rebase. +* GitHub can move a branch without moving the PR's head, so no CI runs for the + new commit. While the head lags its branch nothing about the old head is + judged. A local branch with unpushed work (new commits, or an amend or rebase + of the PR's head) fails the reading, as none of it describes that work, and so + does one that has diverged from the head with commits the head has no copy of. + A head this clone has not fetched cannot be compared, and is only noted. +* A title or body edit fires an `edited` run in which every job skips, and + `gh pr checks` then shows those skips in place of the real results. Runs whose + jobs all skipped are set aside. +* After a force-push the watchers can report the runs the push superseded. Only + runs for the head SHA count, and of several for one workflow (a stack + force-push queues two, and cancels one), the newest that was not cancelled. +* `gh run watch --exit-status` has exited 0 on a failed run, and a run can read + completed while its jobs are still going. Each job's own conclusion decides. +* A rerun replaces a run's conclusion, so a failure that was rerun reads as a + pass. The latest attempt decides, and earlier failed attempts are listed. +* The Claude review skips itself on a PR that edits its own workflow, and on a + branch carrying an older copy of that workflow than the default branch, and + says nothing either way. A review counts only when a bot comment posted during + the latest attempt of this head's review run links back to it, carrying a + progress checklist with every item ticked. Where the review workflow does not + track progress it posts nothing to link, so there a finished run is all there + is to read, and ci-status says so rather than calling it reviewed. +* One commit can head two branches, each with its own PR, and their runs share + its SHA. Only this PR's pull_request runs, and push runs on its branch, count. + GitHub ties a run to every PR open from the same branch, though, so two PRs + from one branch cannot be told apart. + +Something that may yet arrive (a run, a review run, the PR's head following its +branch) is "not settled" in a single reading, because GitHub records no push +time to judge the wait against. --watch calls it final once it has seen it last +a few minutes. + +A job allowed to fail (continue-on-error) is listed but does not gate, which +GitHub shows as a failed job in a run that succeeded; until its run finishes it +cannot be told from a real failure, and gates. pull_request_target runs +are filed under the base branch's commit, so they are not read. + +Checks posted by other apps, and commit statuses, count as gates too. Check runs +the github-actions app posts are taken to mirror jobs and are not read twice, so a +check a workflow posts through the Checks API (a test reporter) is not counted. A +workflow_run-triggered run counts once it has registered, but nothing waits for +one that has not. + +Exit status: 0 green; 1 failed, or will not run until someone acts; 2 the +question was wrong (bad arguments, or GitHub refused them); 3 not settled yet; +4 GitHub could not be read. +""" + +from __future__ import annotations + +import argparse +import base64 +import json +import re +import subprocess +import sys +import time +import urllib.parse +from collections.abc import Callable +from dataclasses import dataclass, field + +GREEN, FAILED, PENDING = "green", "failed", "pending" +EXIT = {GREEN: 0, FAILED: 1, PENDING: 3} +EXIT_UNREADABLE = 4 + +# Events whose runs gate a change. workflow_dispatch and schedule runs share the +# SHA but test something else; issue_comment runs carry the default branch's SHA. +GATE_EVENTS = {"pull_request", "pull_request_target", "push", "merge_group", "workflow_run"} +PR_EVENTS = {"pull_request", "pull_request_target"} +BAD_CONCLUSIONS = {"failure", "cancelled", "timed_out", "action_required", "startup_failure", "stale"} + +REVIEW_WORKFLOW = "claude-code-review.yml" +REVIEW_PATH = f".github/workflows/{REVIEW_WORKFLOW}" + +# How long --watch waits on a condition before calling it final. Timed from +# when the watch first saw it, since GitHub records no push time to time it from. +STUCK_GRACE_S = 180 # a branch ahead of its PR's head +NO_RUNS_GRACE_S = 600 # no run for the commit at all +ALL_SKIPPED_GRACE_S = 180 # only runs that skipped every job +REVIEW_GRACE_S = 300 # no review run for the head +# Consecutive unreadable polls --watch rides out before giving up. +WATCH_ERRORS = 5 + + +class GhError(Exception): + def __init__(self, args: list[str], stderr: str): + super().__init__(f"gh {' '.join(args)}: {stderr.strip()}") + self.not_found = "HTTP 404" in stderr + # An outage, a rate limit or a timeout, as against a question with no answer. + self.transient = bool( + re.search( + r"HTTP (5\d\d|429)\b|timed out|rate limit|error connecting|connection (refused|reset)|unexpected EOF" + r"|i/o timeout|handshake timeout|network is unreachable|deadline exceeded|broken pipe", + stderr, + re.IGNORECASE, + ) + ) + + +def _gh(args: list[str]) -> str: + try: + done = subprocess.run(["gh", *args], capture_output=True, text=True, timeout=120, check=False) + except FileNotFoundError: + print("ci-status needs the GitHub CLI, gh, on PATH, so there is no verdict", file=sys.stderr) + raise SystemExit(EXIT_UNREADABLE) from None + except subprocess.TimeoutExpired as exc: + raise GhError(args, "timed out after 120s") from exc + if done.returncode != 0: + raise GhError(args, done.stderr) + return done.stdout + + +class GitHub: + """The REST calls ci-status makes, through gh so its authentication applies. + + `fixed` marks an answer that cannot change (a finished attempt's jobs, a file + at a commit), which --watch then reads once rather than on every poll. It is + True, or a test the answer must pass before it is kept. + """ + + def __init__(self): + self._fixed: dict[tuple[str, str], object] = {} + + def get(self, path: str, *, fixed: bool | Callable[[object], bool] = False): + return self._read(path, "", fixed, lambda: json.loads(_gh(["api", path]))) + + def items( + self, path: str, key: str | None = None, *, fixed: bool | Callable[[object], bool] = False, version: str = "" + ) -> list: + def read(): + jq = f".{key}[] | @json" if key else ".[] | @json" + out = _gh(["api", "--paginate", path, "--jq", jq]) + return [json.loads(line) for line in out.splitlines() if line.strip()] + + return self._read(path, f"{key}@{version}", fixed, read) + + def _read(self, path, key, fixed, read): + if (path, key) in self._fixed: + return self._fixed[(path, key)] + answer = read() + if fixed is True or (callable(fixed) and fixed(answer)): + self._fixed[(path, key)] = answer + return answer + + +@dataclass +class Job: + name: str + status: str + conclusion: str | None + # The older run this job's result comes from, when the deciding run skipped it. + source: int | None = None + # A lent job's failure that its own run finished as a success, so allowed it. + allowed: bool = False + # Whether its `if:` let it queue. A run cancelled early lists the jobs GitHub had not + # yet evaluated as cancelled with no runner at all; a queued job has runner_id 0. + queued: bool = True + + @property + def state(self) -> str: + return _state(self.status, self.conclusion) + + +def _state(status: str | None, conclusion: str | None) -> str: + return (conclusion if status == "completed" else status) or "?" + + +@dataclass +class Run: + id: int + name: str + path: str + event: str + status: str + conclusion: str | None + attempt: int + created_at: str + started_at: str + jobs: list[Job] = field(default_factory=list) + # Older runs of the same workflow whose jobs this one carries, having skipped them. + lenders: list[Run] = field(default_factory=list) + + @property + def is_review(self) -> bool: + return self.path.rsplit("/", 1)[-1] == REVIEW_WORKFLOW + + @property + def all_skipped(self) -> bool: + if self.status != "completed": + return False + if not self.jobs: + return self.conclusion == "skipped" + return all(job.state == "skipped" for job in self.jobs) + + +@dataclass +class Assessment: + lines: list[str] = field(default_factory=list) + failures: list[str] = field(default_factory=list) + pending: list[str] = field(default_factory=list) + # The commit judged. A wait belongs to it, so a push restarts every wait. + head: str = "" + # Pending conditions that stop being worth waiting for: key -> (seconds, failure). + waits: dict[str, tuple[float, str]] = field(default_factory=dict) + + def wait(self, key: str, grace: float, pending: str, failure: str) -> None: + """Pending now; --watch turns it into `failure` once it has lasted `grace` seconds.""" + self.pending.append(pending) + self.waits[f"{key}@{self.head}"] = (grace, failure) + + @property + def state(self) -> str: + if self.failures: + return FAILED + return PENDING if self.pending else GREEN + + def say(self, label: str, text: str) -> None: + first, *rest = text.splitlines() or [""] + self.lines.append(f" {label:<10}{first}") + self.lines.extend(f" {'':<10}{line}" for line in rest) + + def verdict(self) -> str: + if self.failures: + return "FAILED: " + "; ".join(self.failures) + if self.pending: + return "PENDING: " + "; ".join(self.pending) + return "GREEN: every gate passed" + + +def fetch_runs( + gh: GitHub, repo: str, sha: str, *, pr: int | None = None, branch: str | None = None, review: bool = True +) -> list[Run]: + runs = [] + for raw in gh.items(f"repos/{repo}/actions/runs?head_sha={sha}&per_page=100", "workflow_runs"): + # A fork's pull_request run lists no PRs at all, so an empty list is kept. + prs = [p.get("number") for p in raw.get("pull_requests") or []] + if pr is not None and raw["event"] in PR_EVENTS and prs and pr not in prs: + continue + if branch is not None and raw["event"] == "push" and raw.get("head_branch") != branch: + continue + run = Run( + id=raw["id"], + name=raw.get("name") or raw.get("path", "?"), + path=raw.get("path") or "", + event=raw["event"], + status=raw["status"], + conclusion=raw.get("conclusion"), + attempt=raw.get("run_attempt") or 1, + created_at=raw.get("created_at") or "", + started_at=raw.get("run_started_at") or raw.get("created_at") or "", + ) + if run.is_review and not review: + continue + if run.event in GATE_EVENTS: + run.jobs = attempt_jobs(gh, repo, run.id, run.attempt, finished=run.status == "completed") + runs.append(run) + return runs + + +def attempt_jobs(gh: GitHub, repo: str, run_id: int, attempt: int, *, finished: bool) -> list[Job]: + # A run can read completed while its jobs are still going, and a queued run + # lists none yet, so a list is kept only once the run and every job in it + # have finished. + jobs = gh.items( + f"repos/{repo}/actions/runs/{run_id}/attempts/{attempt}/jobs?per_page=100", + "jobs", + fixed=lambda answer: finished and bool(answer) and all(j.get("status") == "completed" for j in answer), + ) + return [ + Job(j["name"], j["status"], j.get("conclusion"), queued=bool(j.get("steps")) or j.get("runner_id") is not None) + for j in jobs + ] + + +def choose_runs(runs: list[Run]) -> tuple[list[Run], list[tuple[Run, str]]]: + """The run that decides each workflow, and every other run with why it was set aside.""" + ignored: list[tuple[Run, str]] = [] + groups: dict[tuple[str, str], list[Run]] = {} + for run in runs: + if run.event not in GATE_EVENTS: + ignored.append((run, f"a {run.event} run, not a gate")) + elif run.all_skipped: + ignored.append((run, "every job skipped (the run a title or body edit fires)")) + else: + groups.setdefault((run.path or run.name, run.event), []).append(run) + chosen = [] + for group in groups.values(): + group.sort(key=lambda r: (r.created_at, r.id), reverse=True) + live = [r for r in group if r.conclusion != "cancelled"] + pick = live[0] if live else group[0] + # A cancelled run lends only what no live run ran: a job it cut short (a + # timeout, a hand cancel) must not vanish behind an edit's run that skipped it. + cancelled = [r for r in group if r.conclusion == "cancelled" and r is not pick] + donors = _fill_skipped(pick, [r for r in live if r is not pick] + cancelled) + pick.lenders = [r for r in group if r.id in donors] + chosen.append(pick) + for other in group: + if other is not pick: + lent = ", which lends it the jobs it skipped" if other.id in donors else "" + ignored.append((other, f"{other.conclusion or other.status}, superseded by run {pick.id}{lent}")) + chosen.sort(key=lambda r: (r.is_review, r.name)) + return chosen, ignored + + +def _fill_skipped(pick: Run, others: list[Run]) -> set[int]: + """Give `pick` each job it skipped from the first run in `others` that ran it. + + A newer run supersedes an older one job by job, not wholesale: an edit's run + that runs one aggregate job and skips the tests must not hide the tests that + failed in the run before it. The price is that a job the newer run skipped on + purpose (its label removed, say) keeps its older result until the next push. + A cancelled job that never queued ran nothing: a stack push's cancelled + duplicate lists the jobs its survivor skips that way. + """ + donors = set() + jobs: list[Job] = [] + for job in pick.jobs: + lent = [] + if job.state == "skipped": + for run in others: + ran = [ + j + for j in run.jobs + if _same_job(job.name, j.name) and j.state != "skipped" and (j.queued or j.state != "cancelled") + ] + if ran: + allowed = run.status == "completed" and run.conclusion == "success" + lent = [Job(j.name, j.status, j.conclusion, source=run.id, allowed=allowed) for j in ran] + donors.add(run.id) + break + jobs.extend(lent or [job]) + pick.jobs = jobs + return donors + + +def _same_job(skipped: str, ran: str) -> bool: + """Whether a job that ran is the one a newer run skipped under `skipped`. + + A matrix job skipped by its `if` is listed once, unexpanded: `tests`, or + `Web (${{ matrix.label }})` where the name carries an expression. Every + expansion of it (`tests (2)`, `Web (e2e)`) is the same job. + """ + if ran == skipped: + return True + base = re.sub(r"\s*\(\$\{\{.*\}\}\)\s*$", "", skipped) + return ran.startswith(f"{base} (") and (base != skipped or "${{" not in skipped) + + +def judge_run(gh: GitHub, repo: str, run: Run, out: Assessment, verbose: bool) -> None: + unfinished = [j for j in run.jobs if j.status != "completed"] + bad = [j for j in run.jobs if j.conclusion in BAD_CONCLUSIONS] + # Only continue-on-error lets a job fail in a run that succeeded. A job lent by + # an older run is judged by that run. + succeeded = not unfinished and run.status == "completed" and run.conclusion == "success" + allowed = [j for j in bad if j.allowed or (succeeded and not j.source)] + bad = [j for j in bad if j not in allowed] + for job in allowed: + out.say("", f" {job.name} failed, which its run allows (continue-on-error)") + counts: dict[str, int] = {} + for job in run.jobs: + counts[job.state] = counts.get(job.state, 0) + 1 + tally = ", ".join(f"{n} {k}" for k, n in sorted(counts.items())) or "no jobs" + out.say("run", f"{run.name}: {run.id} {run.event} attempt {run.attempt} {run.status} ({tally})") + for job in run.jobs: + if verbose or job.state not in ("success", "skipped") or job.source: + source = f" (from run {job.source})" if job.source else "" + out.say("", f" {job.state:<11} {job.name}{source}") + + # A lender's reruns are history the verdict rests on too. + for past in [run, *run.lenders]: + whose = "" if past is run else f"run {past.id}'s " + for earlier in range(1, past.attempt): + attempt = gh.get(f"repos/{repo}/actions/runs/{past.id}/attempts/{earlier}", fixed=True) + if attempt.get("conclusion") in (None, "success"): + continue + earlier_jobs = attempt_jobs(gh, repo, past.id, earlier, finished=True) + names = [j.name for j in earlier_jobs if j.conclusion in BAD_CONCLUSIONS] + out.say("", f" {whose}attempt {earlier} was {attempt['conclusion']}: {', '.join(names) or 'no job named'}") + + if bad: + out.failures.append(f"{run.name}: {', '.join(f'{j.name} {j.conclusion}' for j in bad)}") + elif unfinished or run.status != "completed": + done = len(run.jobs) - len(unfinished) + out.pending.append(f"{run.name} still running ({done}/{len(run.jobs)} jobs done)") + elif run.conclusion in BAD_CONCLUSIONS: + out.failures.append(f"{run.name}: the run concluded {run.conclusion} with no failing job named") + + +def judge_other_checks(gh: GitHub, repo: str, sha: str, out: Assessment) -> int: + """Checks from apps other than Actions, and commit statuses: gates too, when a repo has them. + + Returns how many there were. + """ + checks = [ + c + for c in gh.items(f"repos/{repo}/commits/{sha}/check-runs?per_page=100", "check_runs") + if ((c.get("app") or {}).get("slug")) != "github-actions" + ] + statuses = gh.items(f"repos/{repo}/commits/{sha}/status?per_page=100", "statuses") + for check in checks: + state = _state(check.get("status"), check.get("conclusion")) + app = (check.get("app") or {}).get("slug", "?") + out.say("check", f"{check.get('name')} ({app}): {state}") + if check.get("status") != "completed": + out.pending.append(f"{check.get('name')} still running") + elif state in BAD_CONCLUSIONS: + out.failures.append(f"{check.get('name')} ({app}) {state}") + for status in statuses: + state = status.get("state") + out.say("status", f"{status.get('context')}: {state}") + if state == "pending": + out.pending.append(f"{status.get('context')} pending") + elif state in ("failure", "error"): + out.failures.append(f"{status.get('context')} {state}") + return len(checks) + len(statuses) + + +def _prose(body: str) -> str: + # A quoted run link or checklist inside a code block belongs to what was quoted. + return re.sub(r"```.*?```", "", body or "", flags=re.DOTALL) + + +def _by_reviewer(comment: dict) -> bool: + # claude[bot], or the same action under a repo's own app name. + user = comment.get("user") or {} + return user.get("type") == "Bot" and "claude" in (user.get("login") or "").lower() + + +def _checklist(body: str) -> list[str] | None: + """The unticked steps of a comment's progress checklist, or None when it has none. + + The checklist runs from the first task line through blank lines and indented + sub-items to the first line of ordinary text. Findings further down may be + written as task lists too, and those are advice, not steps left undone. + """ + block: list[str] = [] + for line in _prose(body).splitlines(): + task = re.match(r"^\s*- \[[ xX]\] ", line) + if task or (block and (not line.strip() or line[:1].isspace())): + if task: + block.append(line) + elif block: + break + if not block: + return None + return [re.sub(r"^\s*- \[ \] ", "", line) for line in block if re.match(r"^\s*- \[ \] ", line)] + + +def judge_review( + gh: GitHub, + repo: str, + pr: dict, + runs: list[Run], + ignored: list[Run], + changed: list[str], + out: Assessment, +) -> None: + head = pr["head"]["sha"] + default = pr["base"]["repo"]["default_branch"] + if REVIEW_PATH in changed: + out.say("review", "none: this PR edits the review workflow, and the action skips itself on such a PR") + out.failures.append("the review bot did not review (this PR edits its workflow); review it by hand") + return + # Keyed by commit, so a watch reads each copy once rather than on every poll. + default_sha = gh.get(f"repos/{repo}/git/ref/heads/{urllib.parse.quote(default, safe='/')}")["object"]["sha"] + try: + on_default = gh.get(f"repos/{repo}/contents/{REVIEW_PATH}?ref={default_sha}", fixed=True) + except GhError as exc: + if exc.not_found: + out.say("review", f"no {REVIEW_PATH} on {default}, so no bot review is expected") + return + raise + try: + on_head = gh.get(f"repos/{repo}/contents/{REVIEW_PATH}?ref={head}", fixed=True) + except GhError as exc: + if not exc.not_found: + raise + on_head = None + # The action runs only when the head's copy of its workflow is byte-identical + # to the default branch's, so a branch cut before that file last changed (or + # stacked on one) is skipped without a word until it is rebased. Only an open + # PR can be rebased, and the default branch may have moved since it merged. + stale = pr["state"] == "open" and (on_head is None or on_head["sha"] != on_default["sha"]) + rebase = f"the review bot skips this head: its {REVIEW_WORKFLOW} differs from {default}'s, so rebase" + if stale: + out.say("review", f"this head's {REVIEW_WORKFLOW} differs from {default}'s") + + review = next((r for r in runs if r.is_review), None) + if review is None: + if stale: + out.failures.append(rebase) + return + skipped = any(r.is_review and r.event in PR_EVENTS and r.all_skipped for r in ignored) + why = "its job skipped (a draft, or a filter)" if skipped else "its workflow did not trigger (a filter?)" + out.say("review", f"no review run for this head{' that ran a job' if skipped else ''} yet") + out.wait( + "no-review", + REVIEW_GRACE_S, + "no Claude review run for this head yet", + f"the review bot did not review this head: {why}", + ) + return + # A run that lent the review its job, when the newest run skipped it, is the + # one whose comment speaks for it. + lenders = {j.source for j in review.jobs if j.source} + candidates = [review] + [r for r in ignored if r.id in lenders] + # The action compared this head with the default branch's copy as it stood + # when the run began, so a head that is stale now was skipped only if that + # copy has not changed since. + since = min(r.started_at for r in candidates) + skipped_head = stale and not _landed_since(gh, repo, default, default_sha, since) + + if review.status != "completed" or any(j.status != "completed" for j in review.jobs): + # judge_run has counted it as pending already; the review is usually the last to finish. + out.say("review", f"run {review.id} still going") + if skipped_head: + out.failures.append(rebase) + return + + # A rerun keeps the run id and posts a fresh comment (or, with a sticky + # comment, edits the old one), so only a comment written during the latest + # attempt speaks for it. + marker = re.compile(rf"/actions/runs/({'|'.join(str(r.id) for r in candidates)})(?!\d)") + mine = [ + c + for c in gh.items(f"repos/{repo}/issues/{pr['number']}/comments?per_page=100") + if _by_reviewer(c) + and marker.search(_prose(c.get("body", ""))) + and max(c.get("created_at") or "", c.get("updated_at") or "") >= since + ] + if mine: + comment = mine[-1] + unticked = _checklist(comment["body"]) + inline = [ + c + for c in gh.items(f"repos/{repo}/pulls/{pr['number']}/comments?per_page=100") + if _by_reviewer(c) and c.get("original_commit_id") == head + ] + out.say( + "review", + f"{comment['html_url']}, with {len(inline)} inline comment(s) made on this head. Read the findings:\n" + f"gh api repos/{repo}/issues/comments/{comment.get('id')} --jq .body", + ) + if unticked is None: + out.say("", "its comment carries no progress checklist, so the review may never have started") + out.failures.append(f"the review left no checklist to show it ran; rerun it: gh run rerun {review.id}") + elif unticked: + out.say("", "unfinished: " + "; ".join(unticked)) + out.failures.append( + f"the review stopped partway ({len(unticked)} step(s) unticked); rerun it: gh run rerun {review.id}" + ) + return + + if skipped_head: + out.say("review", f"run {review.id} {review.conclusion}, with nothing to read: the action skipped this head") + out.failures.append(rebase) + return + text = base64.b64decode((on_head or on_default).get("content", "")).decode("utf-8", "replace") + setting = re.search(r"^\s*track_progress:\s*(\S.*?)\s*$", text, re.MULTILINE) + value = re.sub(r"\s+#.*$", "", setting.group(1)).strip("\"'") if setting else "false" + if value.lower() in ("false", "no", "off") or "${{" in value: + how = "sets track_progress by an expression" if "${{" in value else "does not track progress" + out.say( + "review", + f"run {review.id} {review.conclusion}, and this workflow {how}, so its\n" + "silence cannot be told from a clean pass. Nothing to read.", + ) + return + out.say("review", f"run {review.id} {review.conclusion}, but no bot comment from its latest attempt links to it") + out.failures.append("the review bot did not review (no comment for this head's review run)") + + +def _landed_since(gh: GitHub, repo: str, default: str, default_sha: str, stamp: str) -> bool: + """Whether a change to the review workflow reached `default` after `stamp`. + + A commit's own date says when it was written, not when it landed: a merge + commit carries its branch's older commits. So a commit that came in through a + PR is dated by that PR's merge. + """ + # Git lists by commit date, and a merged branch's older commits sort below + # newer direct edits, so the window is generous for a file that rarely changes. + for commit in gh.get(f"repos/{repo}/commits?sha={default_sha}&path={REVIEW_PATH}&per_page=10", fixed=True): + landed = [ + p["merged_at"] + for p in gh.items(f"repos/{repo}/commits/{commit['sha']}/pulls", fixed=True) + if p.get("merged_at") and (p.get("base") or {}).get("ref") == default + ] + if (min(landed) if landed else commit["commit"]["committer"]["date"]) >= stamp: + return True + return False + + +def local_tip(repo: str, branch: str, head: str) -> tuple[str | None, str]: + """The local branch's commit, when the working directory is a clone of `repo`, and how it stands to `head`. + + "behind" when `head` contains it, "ahead" when it contains `head` and more, + "rewritten" when the branch was at `head` here and has since been amended or + rebased (its reflog holds `head`), "diverged" when neither contains the other + and the branch has commits with no copy in `head` (work made on an older head, + or a remote rewrite that changed them), "superseded" when every commit it has + is in `head` in some form (a remote rebase), and "unknown" when `head` has not + been fetched here. + """ + + def git(*args: str) -> str | None: + try: + done = subprocess.run(["git", *args], capture_output=True, text=True, timeout=20, check=False) + except (OSError, subprocess.TimeoutExpired): + return None + return done.stdout.strip() if done.returncode == 0 else None + + remotes = git("remote", "-v") or "" + if not re.search(rf"[/:]{re.escape(repo)}(\.git)?\s", remotes, re.IGNORECASE): + return None, "unknown" + tip = git("rev-parse", "--verify", "--quiet", f"refs/heads/{branch}^{{commit}}") + if not tip or git("cat-file", "-e", f"{head}^{{commit}}") is None: + return tip, "unknown" + if git("merge-base", "--is-ancestor", tip, head) is not None: + return tip, "behind" + if git("merge-base", "--is-ancestor", head, tip) is not None: + return tip, "ahead" + reflog = (git("reflog", "show", "--format=%H", f"refs/heads/{branch}") or "").split() + if head in reflog: + return tip, "rewritten" + # `git cherry` marks with "+" each commit of the branch with no equivalent patch in `head`. + cherry = git("cherry", head, tip) + if cherry is None or any(line.startswith("+") for line in cherry.splitlines()): + return tip, "diverged" + return tip, "superseded" + + +def assess_pr( + gh: GitHub, + repo: str, + number: int, + *, + review: bool = True, + verbose: bool = False, + local: Callable[[str, str, str], tuple[str | None, str]] = local_tip, + sleep: Callable[[float], None] = time.sleep, +) -> Assessment: + out = Assessment() + pr = gh.get(f"repos/{repo}/pulls/{number}") + for _ in range(3): + if pr.get("mergeable") is not None or pr["state"] != "open": + break + # GitHub computes mergeability on the first read after a change. + sleep(3) + pr = gh.get(f"repos/{repo}/pulls/{number}") + head, ref = pr["head"]["sha"], pr["head"]["ref"] + out.head = head + status = "merged" if pr.get("merged") else pr["state"] + (", draft" if pr.get("draft") else "") + out.lines.append(f"{repo}#{number} {ref} -> {pr['base']['ref']} ({status})") + + notes = [] + stuck = False + on_github = None + head_repo = (pr["head"].get("repo") or {}).get("full_name") + if head_repo and pr["state"] == "open": + try: + on_github = gh.get(f"repos/{head_repo}/git/ref/heads/{urllib.parse.quote(ref, safe='/')}")["object"]["sha"] + except GhError as exc: + if not exc.not_found: + raise + on_github = None + notes.append("branch deleted on GitHub") + if on_github == head: + notes.append("matches its branch on GitHub") + elif on_github: + notes.append(f"but its branch on GitHub is at {on_github[:10]}") + stuck = True + out.wait( + "stuck-head", + STUCK_GRACE_S, + f"the PR's head is still {head[:10]} though its branch is at {on_github[:10]}, so no CI runs " + "for the new commit. If you pushed seconds ago, ask again; if it persists, amend to a new " + "SHA and push with --force-with-lease", + f"the PR's head has not followed its branch for {STUCK_GRACE_S // 60} min: amend to a new SHA " + "and push with --force-with-lease", + ) + # A fork's branch name says nothing about a local branch of the same name. + own = (head_repo or "").lower() == repo.lower() + mine, standing = local(repo, ref, head) if own else (None, "unknown") + if mine == head: + notes.append("matches the local branch") + elif mine and standing in ("ahead", "rewritten") and mine != on_github: + notes.append( + f"local {ref} is at {mine[:10]}, {'rewritten' if standing == 'rewritten' else 'ahead'} and unpushed" + ) + out.failures.append(f"local {ref} has work the PR's head does not, so none of this describes it: push") + elif mine and standing == "diverged" and mine != on_github: + notes.append(f"local {ref} is at {mine[:10]}, diverged from it with commits it has no copy of") + out.failures.append( + f"local {ref} and the PR's head have diverged, so this may not describe the local work: " + "reconcile them (fetch, then rebase or reset) before trusting it" + ) + elif mine: + how = {"behind": "behind it", "ahead": "pushed", "superseded": "superseded by it"}.get(standing, "not compared") + notes.append(f"local {ref} is at {mine[:10]} ({how})") + out.say("head", f"{head[:10]}" + (f" ({'; '.join(notes)})" if notes else "")) + if stuck: + out.say("CI", "not read: it would describe a head the branch has moved past") + return out + + if pr["state"] == "open": + if pr.get("mergeable") is False or pr.get("mergeable_state") == "dirty": + out.say("mergeable", "no: CONFLICTING") + out.failures.append( + "the PR conflicts with its base, so GitHub has no merge ref and pull_request CI will not " + "run until it is rebased. Any 'no checks' reading means nothing was validated" + ) + elif pr.get("mergeable") is None: + out.say("mergeable", "not yet computed by GitHub") + out.pending.append("GitHub has not computed mergeability yet") + else: + out.say("mergeable", f"yes ({pr.get('mergeable_state')})") + + runs = fetch_runs(gh, repo, head, pr=number, branch=ref, review=review) + other_checks = judge_other_checks(gh, repo, head, out) + chosen, ignored = _judge_runs(gh, repo, runs, out, verbose, other_checks=other_checks) + if review: + files = gh.items( + f"repos/{repo}/pulls/{number}/files?per_page=100", fixed=True, version=f"{head}..{pr['base']['sha']}" + ) + changed = [f["filename"] for f in files] + [f["previous_filename"] for f in files if f.get("previous_filename")] + judge_review(gh, repo, pr, chosen, ignored, changed, out) + return out + + +def assess_commit(gh: GitHub, repo: str, sha: str, *, verbose: bool = False) -> Assessment: + full = gh.get(f"repos/{repo}/commits/{sha}", fixed=True)["sha"] + out = Assessment(head=full) + out.lines.append(f"{repo}@{full[:10]}") + other_checks = judge_other_checks(gh, repo, full, out) + _judge_runs(gh, repo, fetch_runs(gh, repo, full), out, verbose, other_checks=other_checks) + return out + + +def _judge_runs( + gh: GitHub, repo: str, runs: list[Run], out: Assessment, verbose: bool, *, other_checks: int +) -> tuple[list[Run], list[Run]]: + chosen, ignored = choose_runs(runs) + # The review reads the change; it tests nothing, so it cannot stand in for CI. + # Another app's check or a commit status can. + tested = any(not r.is_review for r in chosen) or (other_checks > 0 and not _actions_ci(gh, repo)) + skipped = [r for r, _ in ignored if r.event in GATE_EVENTS and r.all_skipped and not r.is_review] + if not tested and skipped: + out.say("CI", "every run for this commit so far skipped all its jobs") + out.wait( + "all-skipped", + ALL_SKIPPED_GRACE_S, + "every run so far skipped all its jobs (a draft, or a filter); runs a new event starts may follow", + "nothing was tested: every run for this commit skipped all its jobs (a draft, or a filter)", + ) + elif not tested: + out.say("CI", "no gating run registered for this commit") + out.wait( + "no-runs", + NO_RUNS_GRACE_S, + "no runs yet: GitHub registers them a few seconds after a push, so ask again (or --watch)", + "no workflow ran for this commit ([skip ci], a paths filter, or no trigger for its event)", + ) + for run in chosen: + judge_run(gh, repo, run, out, verbose) + others: dict[str, int] = {} + for run, why in ignored: + if run.event in GATE_EVENTS: + out.say("ignored", f"{run.name}: {run.id}, {why}") + else: + others[run.event] = others.get(run.event, 0) + 1 + for event, n in sorted(others.items()): + out.say("ignored", f"{n} {event} run(s), which are not gates") + return chosen, [run for run, _ in ignored] + + +def _actions_ci(gh: GitHub, repo: str) -> bool: + """Whether the repo has an active workflow besides the review, so Actions runs are to be waited for.""" + workflows = gh.items(f"repos/{repo}/actions/workflows?per_page=100", "workflows", fixed=True) + return any( + w.get("state") == "active" and w.get("path", "").startswith(".github/workflows/") and w["path"] != REVIEW_PATH + for w in workflows + ) + + +def escalate(result: Assessment, seen: dict[str, float], now: float) -> None: + """Turn each wait that has lasted its grace, as far as this watch has seen, into a failure.""" + for key in list(seen): + if key not in result.waits: + del seen[key] + for key, (grace, failure) in result.waits.items(): + if now - seen.setdefault(key, now) >= grace: + result.failures.append(failure) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("pr", nargs="?", type=int, help="pull request number (default: the current branch's)") + parser.add_argument("-R", "--repo", help="owner/name (default: the current directory's repository)") + parser.add_argument("--commit", help="judge the runs for a commit instead of a PR, e.g. a merge to main") + parser.add_argument("--watch", action="store_true", help="poll until the state settles") + parser.add_argument("--interval", type=float, default=30, help="seconds between polls (default 30)") + parser.add_argument("--timeout", type=float, default=60, help="minutes --watch waits (default 60)") + parser.add_argument("--no-review", action="store_true", help="do not require a Claude review") + parser.add_argument("-v", "--verbose", action="store_true", help="list every job, not only those not passing") + args = parser.parse_args(argv) + if args.pr and args.commit: + parser.error("give a PR or --commit, not both") + if args.repo and not (args.pr or args.commit): + parser.error("with -R, name the PR (or --commit); the current branch belongs to this directory's repo") + + def say_unreadable(exc: Exception) -> int: + print(f"could not read GitHub, so there is no verdict: {exc}", file=sys.stderr) + return EXIT_UNREADABLE + + try: + repo = args.repo or _gh(["repo", "view", "--json", "nameWithOwner", "--jq", ".nameWithOwner"]).strip() + number = args.pr + if not number and not args.commit: + number = int(_gh(["pr", "view", "--json", "number", "--jq", ".number"]).strip()) + except GhError as exc: + if exc.transient: + return say_unreadable(exc) + parser.error(f"name the PR and -R owner/repo; they could not be worked out here ({exc})") + + gh = GitHub() + + def assess() -> Assessment: + if args.commit: + return assess_commit(gh, repo, args.commit, verbose=args.verbose) + return assess_pr(gh, repo, number, review=not args.no_review, verbose=args.verbose) + + def refused(exc: GhError) -> int: + print(f"GitHub refused the question, so check the PR, -R and --commit: {exc}", file=sys.stderr) + return 2 + + deadline = time.time() + args.timeout * 60 + seen: dict[str, float] = {} + errors = 0 + while True: + try: + result = assess() + errors = 0 + except GhError as exc: + if not exc.transient: + return refused(exc) + failed: Exception = exc + except Exception as exc: # noqa: BLE001 - any crash here must not read as a red build + failed = exc + else: + if args.watch: + escalate(result, seen, time.time()) + if not args.watch or result.state != PENDING or time.time() >= deadline: + break + print(f"{time.strftime('%H:%M:%S')} {result.verdict()}", file=sys.stderr, flush=True) + time.sleep(args.interval) + continue + # A 502 or a rate limit says nothing about CI, so a watch asks again. + errors += 1 + if not args.watch or errors >= WATCH_ERRORS or time.time() >= deadline: + return say_unreadable(failed) + print(f"{time.strftime('%H:%M:%S')} could not read GitHub, trying again: {failed}", file=sys.stderr) + time.sleep(args.interval) + print("\n".join(result.lines)) + print(result.verdict()) + return EXIT[result.state] + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/ci-status/test_ci_status.py b/tests/ci-status/test_ci_status.py new file mode 100644 index 0000000..530a7aa --- /dev/null +++ b/tests/ci-status/test_ci_status.py @@ -0,0 +1,1341 @@ +"""ci_status.py's verdicts against a fake GitHub. + +Each case is a way CI has been misread with `gh pr checks` or `gh run watch`. +The verdict has to come out right in both directions: the traps must not read as +green, and the harmless look-alikes (a cancelled duplicate, an edit's skipped +run, a failure that a rerun fixed) must not read as red. +""" + +from __future__ import annotations + +import base64 +import importlib.util +import subprocess +import sys +from pathlib import Path + +import pytest + +_SCRIPT = Path(__file__).resolve().parents[2] / "plugins/core/skills/ci-status/scripts/ci_status.py" +_spec = importlib.util.spec_from_file_location("ci_status", _SCRIPT) +cs = importlib.util.module_from_spec(_spec) +sys.modules["ci_status"] = cs +_spec.loader.exec_module(cs) + +REPO = "org/app" +HEAD = "a" * 40 +OLD = "b" * 40 +REVIEW_PATH = ".github/workflows/claude-code-review.yml" +TRACKED = "steps:\n - uses: anthropics/claude-code-action@v1\n with:\n track_progress: true\n" +UNTRACKED = "steps:\n - uses: anthropics/claude-code-action@v1\n" + + +T1, T2 = "2026-09-23T10:00:00Z", "2026-09-23T10:01:00Z" +T0 = "2026-09-23T09:00:00Z" +MAIN = "9" * 40 # the default branch's head +# The review workflow's recent history on the default branch. +HISTORY = f"repos/{REPO}/commits?sha={MAIN}&path={REVIEW_PATH}&per_page=10" + + +def _change(sha, dated, merged_at=None): + """A commit to the review workflow, dated `dated`, landed by a PR merged at `merged_at` if any.""" + pulls = [{"merged_at": merged_at, "base": {"ref": "main"}}] if merged_at else [] + return {"sha": sha, "commit": {"committer": {"date": dated}}}, {f"repos/{REPO}/commits/{sha}/pulls": pulls} + + +def _run(run_id, name="CI", *, event="pull_request", status="completed", conclusion="success", attempt=1, at=T1): + path = REVIEW_PATH if name == "Claude Code Review" else ".github/workflows/ci.yml" + return { + "id": run_id, + "name": name, + "path": path, + "event": event, + "status": status, + "conclusion": conclusion, + "run_attempt": attempt, + "created_at": at, + "run_started_at": at, + } + + +def _jobs(*pairs): + return [{"name": n, "status": "completed" if c else "in_progress", "conclusion": c} for n, c in pairs] + + +def _started(jobs): + """The same jobs, each taken by a runner that got as far as its first step.""" + return [{**j, "steps": [{"name": "Set up job"}]} for j in jobs] + + +def _content(text, sha): + return {"sha": sha, "content": base64.b64encode(text.encode()).decode()} + + +def _comment(run_id, *, ticked=True, at=T2): + box = "x" if ticked else " " + body = ( + f"**Claude finished** [View job](https://github.com/{REPO}/actions/runs/{run_id})\n\n" + f"- [x] Gather context\n- [{box}] Post findings\n" + ) + return { + "id": 11, + "created_at": at, + "user": {"login": "claude[bot]", "type": "Bot"}, + "body": body, + "html_url": "https://example.test/c1", + } + + +class FakeGitHub: + def __init__(self, routes): + self.routes = routes + self.calls = [] + self.fixed_versions = {} + + def _answer(self, path): + self.calls.append(path) + if path not in self.routes: + raise cs.GhError(["api", path], "gh: Not Found (HTTP 404)") + return self.routes[path] + + def get(self, path, *, fixed=False): + return self._answer(path) + + def items(self, path, key=None, *, fixed=False, version=""): + if fixed is True: + self.fixed_versions[path] = version + return self._answer(path) + + +def _pr(**over): + pr = { + "number": 7, + "state": "open", + "merged": False, + "draft": False, + "mergeable": True, + "mergeable_state": "clean", + "head": {"sha": HEAD, "ref": "feat/x", "repo": {"full_name": REPO}}, + "base": {"ref": "main", "sha": "c" * 40, "repo": {"default_branch": "main"}}, + } + pr.update(over) + return pr + + +def _routes(**over): + """A PR whose CI passed and whose review finished and commented.""" + routes = { + f"repos/{REPO}/pulls/7": _pr(), + f"repos/{REPO}/git/ref/heads/feat/x": {"object": {"sha": HEAD}}, + f"repos/{REPO}/git/ref/heads/main": {"object": {"sha": MAIN}}, + f"repos/{REPO}/actions/runs?head_sha={HEAD}&per_page=100": [_run(1), _run(2, "Claude Code Review")], + f"repos/{REPO}/commits/{HEAD}/check-runs?per_page=100": [], + f"repos/{REPO}/commits/{HEAD}/status?per_page=100": [], + f"repos/{REPO}/actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "success"), ("deploy", "skipped")), + f"repos/{REPO}/actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", "success")), + f"repos/{REPO}/pulls/7/files?per_page=100": [{"filename": "src/app.py"}], + f"repos/{REPO}/contents/{REVIEW_PATH}?ref={MAIN}": _content(TRACKED, "w1"), + f"repos/{REPO}/contents/{REVIEW_PATH}?ref={HEAD}": _content(TRACKED, "w1"), + f"repos/{REPO}/issues/7/comments?per_page=100": [_comment(2)], + f"repos/{REPO}/pulls/7/comments?per_page=100": [], + HISTORY: [], + } + routes.update({(k if k.startswith("repos/") else f"repos/{REPO}/{k}"): v for k, v in over.items()}) + return routes + + +def _assess(routes, *, local=None, standing="behind", review=True): + gh = FakeGitHub(routes) + out = cs.assess_pr( + gh, REPO, 7, review=review, local=lambda repo, ref, head: (local, standing), sleep=lambda s: None + ) + return out, gh + + +def _text(out): + return "\n".join(out.lines) + "\n" + out.verdict() + + +def test_a_passing_pr_with_a_finished_review_is_green(): + out, _ = _assess(_routes(), local=HEAD) + assert out.state == cs.GREEN, _text(out) + assert "https://example.test/c1" in _text(out) + assert f"gh api repos/{REPO}/issues/comments/11 --jq .body" in _text(out) + + +def test_an_edited_run_whose_jobs_all_skipped_does_not_hide_the_real_one(): + runs = [_run(3, at=T2), _run(1, at=T1, conclusion="failure"), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped"), ("deploy", "skipped")), + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "failure")), + } + ) + ) + assert out.state == cs.FAILED + assert "tests failure" in out.verdict() + assert "every job skipped" in _text(out) + + +def test_a_cancelled_duplicate_from_a_stack_push_is_set_aside(): + runs = [_run(5, conclusion="cancelled", at=T2), _run(1, at=T1), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/5/attempts/1/jobs?per_page=100": _jobs(("tests", "cancelled")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "superseded by run 1" in _text(out) + + +def test_a_job_a_cancelled_run_cut_short_is_not_hidden_by_an_edit_s_run(): + # A timeout cancels the job and the run; an edit's run then skips the tests. + runs = [_run(3, at=T2), _run(1, conclusion="cancelled", at=T1), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped"), ("contract", "success")), + "actions/runs/1/attempts/1/jobs?per_page=100": _started( + _jobs(("tests", "cancelled"), ("contract", "success")) + ), + } + ) + ) + assert out.state == cs.FAILED, _text(out) + assert "tests cancelled" in out.verdict() + assert "(from run 1)" in _text(out) + + +@pytest.mark.parametrize("cancelled_at", [T1, "2026-09-23T10:02:00Z"], ids=["older", "newer"]) +def test_a_cancelled_duplicate_lends_no_job_it_never_queued(cancelled_at): + # The survivor skips the deploy chain; the duplicate concurrency cancelled lists it, never queued, as cancelled. + runs = [_run(1, at=T2), _run(5, conclusion="cancelled", at=cancelled_at), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/5/attempts/1/jobs?per_page=100": [ + {**j, "steps": [], "runner_id": None} + for j in _jobs(("tests", "cancelled"), ("deploy", "cancelled")) + ], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "(from run 5)" not in _text(out) + + +def test_a_job_is_queued_once_it_has_a_runner_id_or_a_step(): + answer = [ + {"name": "a", "status": "completed", "conclusion": "cancelled", "steps": [], "runner_id": 0}, + {"name": "b", "status": "completed", "conclusion": "cancelled", "steps": [{"name": "x"}], "runner_id": None}, + {"name": "c", "status": "completed", "conclusion": "cancelled", "steps": [], "runner_id": 7}, + {"name": "d", "status": "completed", "conclusion": "cancelled", "steps": [], "runner_id": None}, + {"name": "e", "status": "completed", "conclusion": "cancelled"}, + ] + gh = FakeGitHub({f"repos/{REPO}/actions/runs/9/attempts/1/jobs?per_page=100": answer}) + jobs = cs.attempt_jobs(gh, REPO, 9, 1, finished=True) + assert [j.queued for j in jobs] == [True, True, True, False, False] + + +def test_a_run_cancelled_by_hand_while_its_tests_queued_is_not_hidden_by_an_edit_s_run(): + runs = [_run(3, at=T2), _run(1, conclusion="cancelled", at=T1), _run(2, "Claude Code Review")] + queued = [{**j, "steps": [], "runner_id": 0} for j in _jobs(("tests", "cancelled"), ("contract", "cancelled"))] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped"), ("contract", "success")), + "actions/runs/1/attempts/1/jobs?per_page=100": queued, + } + ) + ) + assert out.state == cs.FAILED, _text(out) + assert "tests cancelled" in out.verdict() + + +def test_a_live_run_lends_ahead_of_a_cancelled_one(): + runs = [ + _run(3, at=T2), + _run(4, conclusion="cancelled", at="2026-09-23T10:00:30Z"), + _run(1, at=T1), + _run(2, "Claude Code Review"), + ] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped"), ("contract", "success")), + "actions/runs/4/attempts/1/jobs?per_page=100": _started(_jobs(("tests", "cancelled"))), + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "success")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "success tests (from run 1)" in _text(out) + + +def test_a_run_that_reads_completed_while_a_job_runs_is_pending(): + out, _ = _assess(_routes(**{"actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", None))})) + assert out.state == cs.PENDING + assert "CI still running (0/1 jobs done)" in out.verdict() + + +def test_a_run_that_concluded_failure_with_no_failing_job_still_fails(): + runs = [_run(1, conclusion="startup_failure"), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/1/attempts/1/jobs?per_page=100": [], + } + ) + ) + assert out.state == cs.FAILED + assert "startup_failure" in out.verdict() + + +def test_no_runs_yet_is_pending_not_green(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []})) + assert out.state == cs.PENDING + assert "no runs yet" in out.verdict() + + +def test_a_conflicting_pr_fails_because_no_ci_will_run(): + out, _ = _assess(_routes(**{"pulls/7": _pr(mergeable=False, mergeable_state="dirty")})) + assert out.state == cs.FAILED + assert "no merge ref" in out.verdict() + + +def test_unknown_mergeability_is_asked_again_then_pending(): + out, gh = _assess(_routes(**{"pulls/7": _pr(mergeable=None, mergeable_state="unknown")})) + assert out.state == cs.PENDING + assert gh.calls.count(f"repos/{REPO}/pulls/7") == 4 + + +def test_a_branch_the_pr_head_has_not_followed_is_reported(): + out, _ = _assess(_routes(**{"git/ref/heads/feat/x": {"object": {"sha": OLD}}})) + assert f"stuck-head@{HEAD}" in out.waits + assert out.state == cs.PENDING + assert "force-with-lease" in out.verdict() + + +def test_a_local_branch_elsewhere_is_named_but_does_not_change_the_verdict(): + out, _ = _assess(_routes(), local=OLD) + assert out.state == cs.GREEN + assert f"local feat/x is at {OLD[:10]}" in _text(out) + + +def test_a_failure_a_rerun_fixed_is_green_and_still_shown(): + runs = [_run(1, attempt=2), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/1/attempts/2/jobs?per_page=100": _jobs(("tests", "success")), + "actions/runs/1/attempts/1": {"conclusion": "failure"}, + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "failure"), ("lint", "success")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "attempt 1 was failure: tests" in _text(out) + + +def test_a_pr_that_edits_the_review_workflow_is_not_reviewed(): + out, _ = _assess(_routes(**{"pulls/7/files?per_page=100": [{"filename": REVIEW_PATH}]})) + assert out.state == cs.FAILED + assert "review it by hand" in out.verdict() + + +def test_a_branch_with_an_older_review_workflow_is_told_to_rebase(): + runs = [_run(1), _run(2, "Claude Code Review", conclusion="failure")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", "failure")), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(TRACKED, "w0"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.FAILED + assert "rebase" in out.verdict() + + +def test_a_stale_review_workflow_fails_before_the_review_even_starts(): + runs = [_run(1), _run(2, "Claude Code Review", status="in_progress", conclusion=None)] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", None)), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(TRACKED, "w0"), + } + ) + ) + assert out.state == cs.FAILED + assert "rebase" in out.verdict() + + +def test_a_review_that_stopped_partway_fails(): + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [_comment(2, ticked=False)]})) + assert out.state == cs.FAILED + assert "Post findings" in _text(out) + + +def test_a_comment_from_an_earlier_review_run_does_not_count(): + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [_comment(99)]})) + assert out.state == cs.FAILED + assert "did not review" in out.verdict() + + +def test_a_run_link_quoted_in_a_code_block_does_not_count(): + quoted = _comment(99) + quoted["body"] += "\n```\nsee /actions/runs/2\n```\n" + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [quoted]})) + assert out.state == cs.FAILED + + +def test_an_untracked_review_that_posted_nothing_is_noted_not_failed(): + out, _ = _assess( + _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(UNTRACKED, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(UNTRACKED, "w1"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "does not track progress" in _text(out) + + +def test_a_repo_without_a_review_workflow_expects_no_review(): + routes = _routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": [_run(1)]}) + del routes[f"repos/{REPO}/contents/{REVIEW_PATH}?ref={MAIN}"] + out, _ = _assess(routes) + assert out.state == cs.GREEN, _text(out) + + +def test_a_review_not_yet_registered_is_pending(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": [_run(1)]})) + assert out.state == cs.PENDING + assert "no Claude review run" in out.verdict() + + +def test_no_review_skips_the_review_gate(): + out, gh = _assess(_routes(**{"issues/7/comments?per_page=100": []}), review=False) + assert out.state == cs.GREEN + assert not any("comments" in call for call in gh.calls) + + +def test_non_gate_runs_are_counted_not_judged(): + runs = [_run(1, event="push"), _run(9, "Claude Code", event="issue_comment", conclusion="skipped")] + gh = FakeGitHub( + { + f"repos/{REPO}/commits/main": {"sha": HEAD}, + f"repos/{REPO}/commits/{HEAD}/check-runs?per_page=100": [], + f"repos/{REPO}/commits/{HEAD}/status?per_page=100": [], + f"repos/{REPO}/actions/runs?head_sha={HEAD}&per_page=100": runs, + f"repos/{REPO}/actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("deploy", "success")), + } + ) + out = cs.assess_commit(gh, REPO, "main") + assert out.state == cs.GREEN, _text(out) + assert "1 issue_comment run(s), which are not gates" in _text(out) + assert f"repos/{REPO}/actions/runs/9/attempts/1/jobs?per_page=100" not in gh.calls + + +def test_a_skipped_deploy_behind_a_failed_smoke_test_fails_the_commit(): + gh = FakeGitHub( + { + f"repos/{REPO}/commits/main": {"sha": HEAD}, + f"repos/{REPO}/commits/{HEAD}/check-runs?per_page=100": [], + f"repos/{REPO}/commits/{HEAD}/status?per_page=100": [], + f"repos/{REPO}/actions/runs?head_sha={HEAD}&per_page=100": [_run(1, event="push", conclusion="failure")], + f"repos/{REPO}/actions/runs/1/attempts/1/jobs?per_page=100": _jobs( + ("Staging smoke tests", "failure"), ("Deploy to production", "skipped") + ), + } + ) + out = cs.assess_commit(gh, REPO, "main", verbose=True) + assert out.state == cs.FAILED + assert "skipped Deploy to production" in _text(out) + + +def test_a_comment_from_an_earlier_attempt_of_the_same_run_does_not_count(): + runs = [_run(1), _run(2, "Claude Code Review", attempt=2, at=T2)] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/2/attempts/2/jobs?per_page=100": _jobs(("claude-review", "success")), + "actions/runs/2/attempts/1": {"conclusion": "success"}, + "issues/7/comments?per_page=100": [_comment(2, at=T1)], + } + ) + ) + assert out.state == cs.FAILED + assert "did not review" in out.verdict() + + +def test_a_link_to_a_longer_run_id_does_not_count(): + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [_comment(21)]})) + assert out.state == cs.FAILED + + +def test_inline_comments_are_counted_by_the_commit_they_were_made_on(): + inline = [ + {"user": {"login": "claude[bot]", "type": "Bot"}, "commit_id": HEAD, "original_commit_id": OLD}, + {"user": {"login": "claude[bot]", "type": "Bot"}, "commit_id": HEAD, "original_commit_id": HEAD}, + ] + out, _ = _assess(_routes(**{"pulls/7/comments?per_page=100": inline})) + assert "with 1 inline comment(s) made on this head" in _text(out) + + +def test_an_unreadable_github_exits_4_not_as_a_ci_failure(monkeypatch): + def refuse(args): + raise cs.GhError(args, "HTTP 502: Bad Gateway") + + monkeypatch.setattr(cs, "_gh", refuse) + assert cs.main(["7", "-R", REPO]) == 4 + + +def test_a_newer_run_that_skipped_the_tests_does_not_hide_their_failure(): + runs = [_run(3, at=T2), _run(1, at=T1, conclusion="failure"), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("ci-ok", "success"), ("tests", "skipped")), + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("ci-ok", "failure"), ("tests", "failure")), + } + ) + ) + assert out.state == cs.FAILED + assert "tests failure" in out.verdict() + assert "ci-ok failure" not in out.verdict() + assert "tests (from run 1)" in _text(out) + + +def test_a_job_list_is_kept_across_polls_only_once_every_job_has_finished(monkeypatch): + answers = iter( + [ + '{"name": "tests", "status": "in_progress", "conclusion": null}', + '{"name": "tests", "status": "completed", "conclusion": "success"}', + ] + ) + calls = [] + + def gh(args): + calls.append(args) + return next(answers) + + monkeypatch.setattr(cs, "_gh", gh) + github = cs.GitHub() + assert cs.attempt_jobs(github, REPO, 1, 1, finished=True)[0].state == "in_progress" + assert cs.attempt_jobs(github, REPO, 1, 1, finished=True)[0].state == "success" + assert cs.attempt_jobs(github, REPO, 1, 1, finished=True)[0].state == "success" + assert len(calls) == 2 + + +def test_findings_written_as_a_task_list_are_not_unfinished_steps(): + finished = _comment(2) + finished["body"] += "\nFindings:\n\n- [ ] Add a test for the empty case\n" + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [finished]})) + assert out.state == cs.GREEN, _text(out) + + +def test_a_sticky_comment_edited_during_the_latest_attempt_counts(): + sticky = _comment(2, at="2026-09-23T09:00:00Z") + sticky["updated_at"] = T2 + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [sticky]})) + assert out.state == cs.GREEN, _text(out) + + +def test_a_branch_name_is_quoted_into_the_ref_path(): + routes = _routes(**{"pulls/7": _pr(head={"sha": HEAD, "ref": "fix#12", "repo": {"full_name": REPO}})}) + routes[f"repos/{REPO}/git/ref/heads/fix%2312"] = {"object": {"sha": OLD}} + out, _ = _assess(routes) + assert f"stuck-head@{HEAD}" in out.waits + + +@pytest.mark.parametrize("crash", [KeyError("missing"), ValueError("bad timestamp"), TypeError("None")]) +def test_a_crash_while_assessing_exits_4_not_as_a_ci_failure(monkeypatch, crash): + def assess_pr(*args, **kwargs): + raise crash + + monkeypatch.setattr(cs, "assess_pr", assess_pr) + assert cs.main(["7", "-R", REPO]) == 4 + + +def test_an_outage_while_finding_the_pr_is_not_a_usage_error(monkeypatch): + def refuse(args): + raise cs.GhError(args, "HTTP 503: Service Unavailable") + + monkeypatch.setattr(cs, "_gh", refuse) + assert cs.main([]) == 4 + + +def test_no_pr_for_the_branch_is_a_usage_error(monkeypatch): + def refuse(args): + raise cs.GhError(args, 'no pull requests found for branch "main"') + + monkeypatch.setattr(cs, "_gh", refuse) + with pytest.raises(SystemExit) as exited: + cs.main([]) + assert exited.value.code == 2 + + +def test_a_pr_github_does_not_know_is_a_usage_error(monkeypatch): + def refuse(args): + raise cs.GhError(args, "gh: Not Found (HTTP 404)") + + monkeypatch.setattr(cs, "_gh", refuse) + assert cs.main(["9999", "-R", REPO]) == 2 + + +def test_a_quoted_track_progress_still_counts_as_tracked(): + quoted = TRACKED.replace("track_progress: true", 'track_progress: "true"') + out, _ = _assess( + _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(quoted, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(quoted, "w1"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.FAILED + assert "did not review" in out.verdict() + + +def test_renaming_the_review_workflow_away_is_an_edit_to_it(): + renamed = [{"filename": ".github/workflows/review.yml", "previous_filename": REVIEW_PATH}] + out, _ = _assess(_routes(**{"pulls/7/files?per_page=100": renamed})) + assert "review it by hand" in out.verdict() + + +def test_inline_comments_from_a_reviewer_under_its_own_app_name_are_counted(): + inline = [{"user": {"login": "offworld-claude[bot]", "type": "Bot"}, "original_commit_id": HEAD}] + out, _ = _assess(_routes(**{"pulls/7/comments?per_page=100": inline})) + assert "with 1 inline comment(s) made on this head" in _text(out) + + +def test_no_review_also_drops_the_review_run_from_the_gate(): + out, _ = _assess( + _routes(**{"actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", "failure"))}), + review=False, + ) + assert out.state == cs.GREEN, _text(out) + + +def test_the_pr_file_list_is_read_once_per_head_and_base(monkeypatch): + calls = [] + + def gh(args): + calls.append(args) + return '{"filename": "a.py"}' + + monkeypatch.setattr(cs, "_gh", gh) + github = cs.GitHub() + for version in ("h1..b1", "h1..b1", "h2..b1"): + github.items("repos/o/r/pulls/7/files?per_page=100", fixed=True, version=version) + assert len(calls) == 2 + + +def test_the_pr_file_list_is_kept_only_for_the_head_and_base_it_was_read_at(): + _, gh = _assess(_routes()) + assert gh.fixed_versions[f"repos/{REPO}/pulls/7/files?per_page=100"] == f"{HEAD}..{'c' * 40}" + + +def _watched(out, seconds): + """What --watch makes of `out` after seeing it unchanged for `seconds`.""" + seen = {} + cs.escalate(out, seen, 0.0) + cs.escalate(out, seen, float(seconds)) + return out + + +def test_a_review_workflow_that_never_triggered_is_pending_then_final_under_watch(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": [_run(1)]})) + assert out.state == cs.PENDING + assert _watched(out, cs.REVIEW_GRACE_S).state == cs.FAILED + assert "did not trigger" in out.verdict() + + +def test_runs_that_all_skipped_every_job_are_pending_then_final_under_watch(): + runs = [_run(1), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped")), + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", "skipped")), + } + ) + ) + assert out.state == cs.PENDING + assert _watched(out, cs.ALL_SKIPPED_GRACE_S).state == cs.FAILED + assert "nothing was tested" in out.verdict() + + +def test_a_commit_no_workflow_picked_up_is_pending_then_final_under_watch(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []})) + assert out.state == cs.PENDING + assert _watched(out, cs.NO_RUNS_GRACE_S).state == cs.FAILED + assert "no workflow ran for this commit" in out.verdict() + + +def test_a_wait_shorter_than_its_grace_stays_pending(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []}), review=False) + assert _watched(out, cs.NO_RUNS_GRACE_S - 1).state == cs.PENDING + + +def test_a_wait_that_clears_starts_again_from_nothing(): + seen = {} + first, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []})) + cs.escalate(first, seen, 0.0) + cleared, _ = _assess(_routes()) + cs.escalate(cleared, seen, 100.0) + again, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []})) + cs.escalate(again, seen, float(cs.NO_RUNS_GRACE_S)) + assert again.state == cs.PENDING + + +def test_a_merged_pr_is_not_told_to_rebase_when_main_s_review_workflow_moved_on(): + out, _ = _assess( + _routes( + **{ + "pulls/7": _pr(state="closed", merged=True), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(TRACKED, "w0"), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "differs" not in _text(out) + + +def _stale_untracked(moved): + routes = _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(UNTRACKED, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(UNTRACKED, "w0"), + "issues/7/comments?per_page=100": [], + } + ) + if moved: + commit, pulls = _change("d" * 40, T2) + routes[HISTORY] = [commit] + routes.update(pulls) + return _assess(routes)[0] + + +def test_an_untracked_review_stays_neutral_when_main_s_copy_moved_after_it_ran(): + out = _stale_untracked(moved=True) + assert out.state == cs.GREEN, _text(out) + assert "does not track progress" in _text(out) + + +def test_a_stale_head_whose_review_run_finished_is_still_unreviewed(): + out = _stale_untracked(moved=False) + assert out.state == cs.FAILED + assert "rebase" in out.verdict() + + +def test_a_review_run_that_reads_completed_with_its_job_going_is_not_judged_yet(): + out, _ = _assess( + _routes( + **{ + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", None)), + "issues/7/comments?per_page=100": [_comment(2, ticked=False)], + } + ) + ) + assert out.state == cs.PENDING, _text(out) + + +@pytest.mark.parametrize(("state", "code"), [(cs.GREEN, 0), (cs.FAILED, 1), (cs.PENDING, 3)]) +def test_exit_codes(state, code): + assert cs.EXIT[state] == code + + +def test_a_pr_and_a_commit_together_are_refused(): + with pytest.raises(SystemExit) as exited: + cs.main(["7", "--commit", "main", "-R", REPO]) + assert exited.value.code == 2 + + +def test_a_failing_check_from_another_app_fails_and_actions_checks_are_not_counted_twice(): + checks = [ + {"name": "codecov/patch", "status": "completed", "conclusion": "failure", "app": {"slug": "codecov"}}, + {"name": "tests", "status": "completed", "conclusion": "failure", "app": {"slug": "github-actions"}}, + ] + out, _ = _assess(_routes(**{f"commits/{HEAD}/check-runs?per_page=100": checks})) + assert out.state == cs.FAILED + assert "codecov/patch (codecov) failure" in out.verdict() + assert "tests (github-actions)" not in out.verdict() + + +def test_a_pending_commit_status_is_pending(): + status = [{"context": "ci/external", "state": "pending"}] + out, _ = _assess(_routes(**{f"commits/{HEAD}/status?per_page=100": status})) + assert out.state == cs.PENDING + assert "ci/external pending" in out.verdict() + + +@pytest.mark.parametrize( + ("answer", "finished"), + [("", True), ('{"name": "setup", "status": "completed", "conclusion": "success"}', False)], + ids=["no jobs listed yet", "run still going"], +) +def test_a_job_list_that_may_still_grow_is_read_again(monkeypatch, answer, finished): + calls = [] + + def gh(args): + calls.append(args) + return answer + + monkeypatch.setattr(cs, "_gh", gh) + github = cs.GitHub() + cs.attempt_jobs(github, REPO, 1, 1, finished=finished) + cs.attempt_jobs(github, REPO, 1, 1, finished=finished) + assert len(calls) == 2 + + +@pytest.mark.parametrize( + ("stderr", "transient"), + [ + ('no pull requests found for branch "feat/geofence-alerts"', False), + ('no pull requests found for branch "fix/reconnect"', False), + ("gh: Not Found (HTTP 404)", False), + ("HTTP 502: Bad Gateway", True), + ("API rate limit exceeded (HTTP 403)", True), + ("error connecting to api.github.com", True), + ('Get "https://api.github.com/x": dial tcp 140.82.1.1:443: i/o timeout', True), + ("net/http: TLS handshake timeout", True), + ("connect: network is unreachable", True), + ], +) +def test_only_an_outage_is_transient(stderr, transient): + assert cs.GhError(["api"], stderr).transient is transient + + +@pytest.mark.parametrize("spelling", ["false", "False", "no", "off", "'false'"]) +def test_every_yaml_spelling_of_false_turns_tracking_off(spelling): + off = TRACKED.replace("track_progress: true", f"track_progress: {spelling}") + out, _ = _assess( + _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(off, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(off, "w1"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + + +def test_a_missing_gh_is_no_verdict_rather_than_a_bad_question(monkeypatch): + def missing(*args, **kwargs): + raise FileNotFoundError("gh") + + monkeypatch.setattr(cs.subprocess, "run", missing) + with pytest.raises(SystemExit) as exited: + cs._gh(["api", "user"]) + assert exited.value.code == 4 + + +def test_runs_another_pr_on_the_same_head_started_do_not_count(): + other = {**_run(9, at=T2), "pull_requests": [{"number": 8}]} + mine = {**_run(1, at=T1, conclusion="failure"), "pull_requests": [{"number": 7}]} + out, gh = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": [other, mine, _run(2, "Claude Code Review")], + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "failure")), + } + ) + ) + assert out.state == cs.FAILED + assert f"repos/{REPO}/actions/runs/9/attempts/1/jobs?per_page=100" not in gh.calls + + +def test_a_wait_restarts_when_the_head_moves(): + seen = {} + first, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": []}), review=False) + cs.escalate(first, seen, 0.0) + moved = _pr(head={"sha": OLD, "ref": "feat/x", "repo": {"full_name": REPO}}) + routes = _routes(**{"pulls/7": moved, f"actions/runs?head_sha={OLD}&per_page=100": []}) + routes[f"repos/{REPO}/git/ref/heads/feat/x"] = {"object": {"sha": OLD}} + routes[f"repos/{REPO}/commits/{OLD}/check-runs?per_page=100"] = [] + routes[f"repos/{REPO}/commits/{OLD}/status?per_page=100"] = [] + second, _ = _assess(routes, review=False) + cs.escalate(second, seen, float(cs.NO_RUNS_GRACE_S)) + assert second.state == cs.PENDING + + +def test_no_review_does_not_read_the_review_run_at_all(): + _, gh = _assess(_routes(), review=False) + assert f"repos/{REPO}/actions/runs/2/attempts/1/jobs?per_page=100" not in gh.calls + + +def test_a_watch_rides_out_a_blip_on_its_first_reading(monkeypatch): + readings = iter([cs.GhError(["api"], "HTTP 502: Bad Gateway"), cs.Assessment(head=HEAD)]) + + def assess_pr(*args, **kwargs): + answer = next(readings) + if isinstance(answer, Exception): + raise answer + return answer + + monkeypatch.setattr(cs, "assess_pr", assess_pr) + monkeypatch.setattr(cs.time, "sleep", lambda s: None) + assert cs.main(["7", "-R", REPO, "--watch"]) == 0 + + +def test_a_single_reading_does_not_retry_a_blip(monkeypatch): + calls = [] + + def assess_pr(*args, **kwargs): + calls.append(args) + raise cs.GhError(["api"], "HTTP 502: Bad Gateway") + + monkeypatch.setattr(cs, "assess_pr", assess_pr) + monkeypatch.setattr(cs.time, "sleep", lambda s: None) + assert cs.main(["7", "-R", REPO]) == 4 + assert len(calls) == 1 + + +def test_a_change_merged_after_the_run_began_counts_though_its_commit_is_older(): + routes = _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(UNTRACKED, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(UNTRACKED, "w0"), + "issues/7/comments?per_page=100": [], + } + ) + commit, pulls = _change("e" * 40, T0, merged_at=T2) + routes[HISTORY] = [commit] + routes.update(pulls) + out, _ = _assess(routes) + assert out.state == cs.GREEN, _text(out) + + +def test_a_review_that_ran_does_not_stand_in_for_ci_that_skipped_everything(): + out, _ = _assess( + _routes(**{"actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "skipped"), ("deploy", "skipped"))}) + ) + assert out.state == cs.PENDING + assert _watched(out, cs.ALL_SKIPPED_GRACE_S).state == cs.FAILED + assert "nothing was tested" in out.verdict() + + +def test_a_review_that_ran_does_not_stand_in_for_ci_that_never_registered(): + out, _ = _assess(_routes(**{f"actions/runs?head_sha={HEAD}&per_page=100": [_run(2, "Claude Code Review")]})) + assert out.state == cs.PENDING + assert "no runs yet" in out.verdict() + + +def test_a_review_still_going_is_not_failed_when_main_s_copy_moved_after_it_began(): + commit, pulls = _change("f" * 40, T2) + routes = _routes( + **{ + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("claude-review", None)), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(TRACKED, "w0"), + HISTORY: [commit], + } + ) + routes.update(pulls) + out, _ = _assess(routes) + assert out.state == cs.PENDING, _text(out) + + +def test_the_comment_of_the_run_that_lent_the_review_its_job_counts(): + runs = [_run(1), _run(12, "Claude Code Review", at=T2), _run(2, "Claude Code Review", at=T1)] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/12/attempts/1/jobs?per_page=100": _jobs( + ("gate", "success"), ("claude-review", "skipped") + ), + "actions/runs/2/attempts/1/jobs?per_page=100": _jobs(("gate", "success"), ("claude-review", "success")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + + +def test_track_progress_set_by_an_expression_is_read_as_untracked(): + expr = TRACKED.replace("track_progress: true", "track_progress: ${{ github.event_name == 'issue_comment' }}") + out, _ = _assess( + _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(expr, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(expr, "w1"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "by an expression" in _text(out) + + +def test_a_partial_review_names_the_rerun_that_works_when_no_job_failed(): + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [_comment(2, ticked=False)]})) + assert "gh run rerun 2" in out.verdict() + assert "--failed" not in out.verdict() + + +def test_a_missing_git_leaves_the_local_branch_unread(monkeypatch): + def missing(*args, **kwargs): + raise FileNotFoundError("git") + + monkeypatch.setattr(cs.subprocess, "run", missing) + assert cs.local_tip(REPO, "feat/x", HEAD) == (None, "unknown") + + +def test_a_failure_its_run_allows_does_not_gate(): + out, _ = _assess( + _routes(**{"actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "success"), ("nightly", "failure"))}) + ) + assert out.state == cs.GREEN, _text(out) + assert "nightly failed, which its run allows" in _text(out) + + +def test_a_failure_in_a_run_that_failed_still_gates(): + runs = [_run(1, conclusion="failure"), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "success"), ("nightly", "failure")), + } + ) + ) + assert out.state == cs.FAILED + + +CIRCLE = [{"name": "ci/circle", "status": "completed", "conclusion": "success", "app": {"slug": "circleci"}}] +WORKFLOWS = "actions/workflows?per_page=100" + + +def _workflow(path, state="active"): + return {"path": path, "state": state} + + +def test_checks_from_another_app_count_as_ci_where_actions_runs_none(): + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": [_run(2, "Claude Code Review")], + f"commits/{HEAD}/check-runs?per_page=100": CIRCLE, + WORKFLOWS: [_workflow(REVIEW_PATH), _workflow("dynamic/github-code-scanning/codeql")], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + + +def test_another_app_s_check_does_not_stand_in_for_actions_ci_not_yet_registered(): + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": [_run(2, "Claude Code Review")], + f"commits/{HEAD}/check-runs?per_page=100": CIRCLE, + WORKFLOWS: [_workflow(REVIEW_PATH), _workflow(".github/workflows/ci.yml")], + } + ) + ) + assert out.state == cs.PENDING + assert "no runs yet" in out.verdict() + + +def test_push_runs_from_another_branch_at_the_same_commit_do_not_count(): + elsewhere = {**_run(9, event="push", conclusion="failure", at=T2), "head_branch": "feat/y"} + mine = {**_run(8, event="push", at=T1), "head_branch": "feat/x"} + out, gh = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": [elsewhere, mine, _run(1), _run(2, "Claude Code Review")], + "actions/runs/8/attempts/1/jobs?per_page=100": _jobs(("tests", "success")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert f"repos/{REPO}/actions/runs/9/attempts/1/jobs?per_page=100" not in gh.calls + + +def test_an_inline_comment_after_track_progress_false_is_not_part_of_the_value(): + off = TRACKED.replace("track_progress: true", "track_progress: false # too noisy") + out, _ = _assess( + _routes( + **{ + f"contents/{REVIEW_PATH}?ref={MAIN}": _content(off, "w1"), + f"contents/{REVIEW_PATH}?ref={HEAD}": _content(off, "w1"), + "issues/7/comments?per_page=100": [], + } + ) + ) + assert out.state == cs.GREEN, _text(out) + + +def test_the_default_branch_s_copy_is_read_by_commit_so_a_watch_reads_it_once(): + _, gh = _assess(_routes()) + assert f"repos/{REPO}/contents/{REVIEW_PATH}?ref=main" not in gh.calls + assert f"repos/{REPO}/contents/{REVIEW_PATH}?ref={MAIN}" in gh.calls + + +def test_a_lent_failure_its_own_run_allowed_does_not_gate(): + runs = [_run(3, at=T2), _run(1, at=T1), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("tests", "success"), ("nightly", "skipped")), + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "success"), ("nightly", "failure")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "nightly failed, which its run allows" in _text(out) + + +def test_a_failure_in_a_run_still_going_gates(): + runs = [_run(1, status="in_progress", conclusion=None), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "failure"), ("lint", None)), + } + ) + ) + assert out.state == cs.FAILED + + +def test_a_repo_named_without_a_pr_is_a_usage_error(monkeypatch): + # The current branch's PR belongs to this directory's repo, not the one named. + asked = [] + monkeypatch.setattr(cs, "_gh", lambda args: asked.append(args) or "602") + monkeypatch.setattr(cs, "assess_pr", lambda *a, **k: cs.Assessment(head=HEAD)) + with pytest.raises(SystemExit) as exited: + cs.main(["-R", REPO]) + assert exited.value.code == 2 + assert asked == [] + + +def test_a_local_branch_with_unpushed_commits_fails_the_reading(): + out, _ = _assess(_routes(), local=OLD, standing="ahead") + assert out.state == cs.FAILED + assert "push" in out.verdict() + + +def test_a_pushed_commit_the_pr_head_has_not_followed_is_a_wait_not_unpushed_work(): + out, _ = _assess(_routes(**{"git/ref/heads/feat/x": {"object": {"sha": OLD}}}), local=OLD, standing="ahead") + assert out.state == cs.PENDING, _text(out) + assert "force-with-lease" in out.verdict() + + +def test_a_local_branch_a_remote_rebase_superseded_is_noted_not_failed(): + out, _ = _assess(_routes(), local=OLD, standing="superseded") + assert out.state == cs.GREEN, _text(out) + assert "superseded" in _text(out) + + +def test_a_local_branch_diverged_with_work_the_head_lacks_fails(): + out, _ = _assess(_routes(), local=OLD, standing="diverged") + assert out.state == cs.FAILED + assert "reconcile" in out.verdict() + + +def test_a_head_that_lags_its_branch_is_not_judged_by_its_old_ci(): + out, _ = _assess( + _routes( + **{ + "git/ref/heads/feat/x": {"object": {"sha": OLD}}, + f"actions/runs?head_sha={HEAD}&per_page=100": [ + _run(1, conclusion="failure"), + _run(2, "Claude Code Review"), + ], + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("tests", "failure")), + } + ) + ) + assert out.state == cs.PENDING, _text(out) + assert "tests failure" not in out.verdict() + + +def test_a_review_comment_with_no_checklist_is_not_a_review(): + bare = _comment(2) + bare["body"] = "**Claude finished** [View job](https://github.com/org/app/actions/runs/2)\n\nWorking..." + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [bare]})) + assert out.state == cs.FAILED + assert "no checklist" in out.verdict() + + +@pytest.mark.parametrize("gap", ["\n", " - a note on the first step\n"]) +def test_an_unticked_step_after_a_blank_line_or_sub_item_is_seen(gap): + body = ( + "**Claude finished** [View job](https://github.com/org/app/actions/runs/2)\n\n" + f"- [x] Gather context\n{gap}- [ ] Post findings\n\nFindings:\n\n- [ ] Add a test\n" + ) + comment = {**_comment(2), "body": body} + out, _ = _assess(_routes(**{"issues/7/comments?per_page=100": [comment]})) + assert out.state == cs.FAILED + assert "Post findings" in _text(out) + assert "Add a test" not in _text(out) + + +def test_the_local_branch_is_placed_against_the_head_by_real_git(tmp_path, monkeypatch): + def git(*args): + done = subprocess.run(["git", *args], cwd=tmp_path, capture_output=True, text=True, check=True) + return done.stdout.strip() + + git("init", "-q", "-b", "main") + git("config", "user.email", "t@example.test") + git("config", "user.name", "t") + git("remote", "add", "origin", f"git@github.com:{REPO}.git") + git("commit", "-q", "--allow-empty", "-m", "a") + first = git("rev-parse", "HEAD") + git("commit", "-q", "--allow-empty", "-m", "b") + second = git("rev-parse", "HEAD") + git("checkout", "-q", "-b", "other", first) + git("branch", "-q", "feat/x", second) + monkeypatch.chdir(tmp_path) + + assert cs.local_tip(REPO, "feat/x", first) == (second, "ahead") + git("branch", "-q", "-f", "feat/x", first) + assert cs.local_tip(REPO, "feat/x", second) == (first, "behind") + assert cs.local_tip(REPO, "feat/x", "f" * 40) == (first, "unknown") + assert cs.local_tip("org/elsewhere", "feat/x", second) == (None, "unknown") + # The branch was at `second` here and was then amended: unpushed work, not a rewrite elsewhere. + git("checkout", "-q", "feat/x") + git("reset", "-q", "--hard", second) + git("commit", "-q", "--amend", "--allow-empty", "-m", "b, amended") + amended = git("rev-parse", "HEAD") + assert cs.local_tip(REPO, "feat/x", second) == (amended, "rewritten") + + +def test_diverged_work_is_told_from_a_branch_a_remote_rebase_superseded(tmp_path, monkeypatch): + def git(*args): + done = subprocess.run(["git", *args], cwd=tmp_path, capture_output=True, text=True, check=True) + return done.stdout.strip() + + def commit(name, text): + (tmp_path / name).write_text(text) + git("add", name) + git("commit", "-q", "-m", name) + return git("rev-parse", "HEAD") + + git("init", "-q", "-b", "main") + git("config", "user.email", "t@example.test") + git("config", "user.name", "t") + git("remote", "add", "origin", f"git@github.com:{REPO}.git") + base = commit("a", "a") + git("checkout", "-q", "-b", "feat/x") + mine = commit("b", "b") + # Elsewhere, the head was rebased onto a newer main: b again, on top of c. + git("checkout", "-q", "-b", "remote", base) + commit("c", "c") + git("cherry-pick", mine) + rebased = git("rev-parse", "HEAD") + # And elsewhere again, a head built on a's successor that lacks b entirely. + git("checkout", "-q", "-b", "other", base) + other = commit("d", "d") + monkeypatch.chdir(tmp_path) + + assert cs.local_tip(REPO, "feat/x", rebased) == (mine, "superseded") + assert cs.local_tip(REPO, "feat/x", other) == (mine, "diverged") + + +def test_a_branch_amended_here_and_not_pushed_fails_the_reading(): + out, _ = _assess(_routes(), local=OLD, standing="rewritten") + assert out.state == cs.FAILED + assert "push" in out.verdict() + + +def test_a_skipped_matrix_job_is_lent_every_expansion_that_ran(): + # As GitHub lists them: an edit's run names a skipped matrix job once, unexpanded. + web = "Web (${{ (matrix.label || matrix.workspace) }})" + runs = [_run(3, at=T2), _run(1, at=T1, conclusion="failure"), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs( + ("contract", "success"), ("backend-tests", "skipped"), (web, "skipped") + ), + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs( + ("contract", "success"), + ("backend-tests (1)", "success"), + ("backend-tests (2)", "failure"), + ("Web (e2e)", "success"), + ("Web (dashboard tests 1/2)", "failure"), + ), + } + ) + ) + assert out.state == cs.FAILED + assert "backend-tests (2) failure" in out.verdict() + assert "Web (dashboard tests 1/2) failure" in out.verdict() + assert "${{" not in out.verdict() + + +@pytest.mark.parametrize( + ("skipped", "ran", "same"), + [ + ("lint", "lint", True), + ("lint", "lint-extra", False), + ("backend-tests", "backend-tests (3)", True), + ("Web (${{ matrix.label }})", "Web (e2e)", True), + ("Web (${{ matrix.label }})", "Webhooks (e2e)", False), + ("Web (e2e)", "Web (dashboard)", False), + ], +) +def test_matrix_names_match_their_expansions_only(skipped, ran, same): + assert cs._same_job(skipped, ran) is same + + +def test_a_rerun_in_a_run_that_lends_jobs_is_still_listed(): + runs = [_run(3, at=T2), _run(1, at=T1, attempt=2), _run(2, "Claude Code Review")] + out, _ = _assess( + _routes( + **{ + f"actions/runs?head_sha={HEAD}&per_page=100": runs, + "actions/runs/3/attempts/1/jobs?per_page=100": _jobs(("contract", "success"), ("tests", "skipped")), + "actions/runs/1/attempts/2/jobs?per_page=100": _jobs(("contract", "success"), ("tests", "success")), + "actions/runs/1/attempts/1": {"conclusion": "failure"}, + "actions/runs/1/attempts/1/jobs?per_page=100": _jobs(("contract", "success"), ("tests", "failure")), + } + ) + ) + assert out.state == cs.GREEN, _text(out) + assert "run 1's attempt 1 was failure: tests" in _text(out) + + +def test_a_pushed_diverged_tip_the_pr_head_has_not_followed_is_a_wait(): + out, _ = _assess(_routes(**{"git/ref/heads/feat/x": {"object": {"sha": OLD}}}), local=OLD, standing="diverged") + assert out.state == cs.PENDING, _text(out) + + +def test_a_fork_s_branch_is_not_compared_with_a_local_branch_of_the_same_name(): + seen = [] + fork = _pr(head={"sha": HEAD, "ref": "main", "repo": {"full_name": "someone/app"}}) + routes = _routes(**{"pulls/7": fork}) + routes["repos/someone/app/git/ref/heads/main"] = {"object": {"sha": HEAD}} + gh = FakeGitHub(routes) + out = cs.assess_pr( + gh, REPO, 7, local=lambda repo, ref, head: seen.append(ref) or (OLD, "diverged"), sleep=lambda s: None + ) + assert out.state == cs.GREEN, _text(out) + assert seen == []