From 75cdac47919a460ce1841629a0d11920d366d6ea Mon Sep 17 00:00:00 2001 From: Toby Hede Date: Thu, 8 Oct 2026 10:39:54 +1100 Subject: [PATCH 1/5] ci: add advisory ImpactGate reporting (#1137) --- .github/impact-gate/all.yml | 16 ++ .github/impact-gate/gate.yml | 2 + .github/impact-gate/requirements.txt | 2 + .github/impact-gate/source.yml | 45 +++++ .github/workflows/impact.yml | 87 ++++++++++ .gitignore | 4 + docs/plans/2026-10-08-impact-gate-rollout.md | 88 ++++++++++ scripts/__tests__/impact-workflow.test.mjs | 81 +++++++++ scripts/impact-gate.py | 165 ++++++++++++++++++ scripts/tests/test_impact_gate.py | 171 +++++++++++++++++++ 10 files changed, 661 insertions(+) create mode 100644 .github/impact-gate/all.yml create mode 100644 .github/impact-gate/gate.yml create mode 100644 .github/impact-gate/requirements.txt create mode 100644 .github/impact-gate/source.yml create mode 100644 .github/workflows/impact.yml create mode 100644 docs/plans/2026-10-08-impact-gate-rollout.md create mode 100644 scripts/__tests__/impact-workflow.test.mjs create mode 100644 scripts/impact-gate.py create mode 100644 scripts/tests/test_impact_gate.py diff --git a/.github/impact-gate/all.yml b/.github/impact-gate/all.yml new file mode 100644 index 000000000..d30ef0e20 --- /dev/null +++ b/.github/impact-gate/all.yml @@ -0,0 +1,16 @@ +# Appended to ImpactGate defaults. Production generators remain eligible. +ignore: + - '**/dist/**' + - 'dist/**' + - '**/target/**' + - 'target/**' + - '**/generated/**' + - 'generated/**' + - '**/*.generated.*' + - '**/*.gen.*' + - '**/*.d.ts' + - '**/node_modules/**' + - '**/vendor/**' + # Committed Go bindings carry DO NOT EDIT generated headers. + - '**/*_stash.go' + - '**/*_gen.go' diff --git a/.github/impact-gate/gate.yml b/.github/impact-gate/gate.yml new file mode 100644 index 000000000..89bad334d --- /dev/null +++ b/.github/impact-gate/gate.yml @@ -0,0 +1,2 @@ +# Explicit policy prevents a checkout-local root config changing CI analysis. +cognitive_max: null diff --git a/.github/impact-gate/requirements.txt b/.github/impact-gate/requirements.txt new file mode 100644 index 000000000..c63a063e7 --- /dev/null +++ b/.github/impact-gate/requirements.txt @@ -0,0 +1,2 @@ +impact-gate==0.4.1 +lizard==1.23.0 diff --git a/.github/impact-gate/source.yml b/.github/impact-gate/source.yml new file mode 100644 index 000000000..44223bb1d --- /dev/null +++ b/.github/impact-gate/source.yml @@ -0,0 +1,45 @@ +# Appended to ImpactGate defaults. Production generators remain eligible. +ignore: + - '**/dist/**' + - 'dist/**' + - '**/target/**' + - 'target/**' + - '**/generated/**' + - 'generated/**' + - '**/*.generated.*' + - '**/*.gen.*' + - '**/*.d.ts' + - '**/node_modules/**' + - '**/vendor/**' + # Test FILE exclusions do not remove inline Rust test modules. + - 'test/**' + - '**/test/**' + - 'tests/**' + - '**/tests/**' + - '**/__tests__/**' + - '__tests__/**' + - '**/__mocks__/**' + # Other fixture-only trees are under tests/__tests__/integration-tests above. + # eql-domains/src/fixtures is the production catalog DSL: keep it eligible. + - '**/cli/scripts/fixtures/**' + - 'fixtures/**' + - '**/test-fixtures/**' + - '**/integration-tests/**' + - 'e2e/**' + - '**/e2e/**' + - '**/*.test.*' + - '**/*.spec.*' + - '**/*_test.go' + - '*_test.go' + - '**/eql-tests-macros/**' + # Shared test harness and injectable fake backend, including historical homes. + - '**/test-kit/**' + - '**/internal/guest/testing.go' + - '**/*_stash.go' + - '**/*_gen.go' + - '**/__fixtures__/**' + - '**/test_*.py' + - '**/tests.rs' + - '**/proptest_invariants.rs' + - '**/internal/testpolicy/**' + - '**/internal/testusers/**' diff --git a/.github/workflows/impact.yml b/.github/workflows/impact.yml new file mode 100644 index 000000000..456ba322c --- /dev/null +++ b/.github/workflows/impact.yml @@ -0,0 +1,87 @@ +name: Change impact + +on: + pull_request: + push: + branches: [main] + +permissions: + contents: read + +concurrency: + group: impact-${{ github.ref }} + cancel-in-progress: true + +jobs: + impact: + name: Change impact (advisory) + runs-on: ubuntu-latest + timeout-minutes: 20 + defaults: + run: + shell: bash + env: + BASE_REF: refs/remotes/origin/${{ github.base_ref || github.ref_name }} + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 + with: + fetch-depth: 0 + persist-credentials: false + + - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6 + with: + python-version: '3.12' + + - name: Install analysis tools + run: python -m pip install -r .github/impact-gate/requirements.txt + + - name: Check reporting behavior with the real CLI + run: python -m unittest discover -s scripts/tests -p test_impact_gate.py -v + + - name: Pin and identify target branch + id: target + env: + TARGET_BRANCH: ${{ github.base_ref || github.ref_name }} + BASE_SHA: ${{ github.event.pull_request.base.sha || github.sha }} + run: | + # Keep one logical target ref across push/PR events, pinned to the + # event's commit even if the remote branch advances while queued. + git update-ref "$BASE_REF" "$BASE_SHA" + echo "BASELINE_DIR=$RUNNER_TEMP/impact-baseline" >> "$GITHUB_ENV" + python - <<'PY' + import hashlib + import os + with open(os.environ['GITHUB_OUTPUT'], 'a') as output: + key = hashlib.sha256(os.environ['TARGET_BRANCH'].encode()).hexdigest() + output.write(f'key={key}\n') + PY + + # The fallback stays within one tool/policy AND target branch. Never use a + # repository-wide fallback: another release branch has different history. + - name: Restore baselines + id: restore + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4 + with: + path: ${{ env.BASELINE_DIR }} + key: impact-${{ hashFiles('.github/impact-gate/**', 'scripts/impact-gate.py', '.github/workflows/impact.yml') }}-${{ steps.target.outputs.key }}-${{ github.event.pull_request.base.sha || github.sha }} + restore-keys: impact-${{ hashFiles('.github/impact-gate/**', 'scripts/impact-gate.py', '.github/workflows/impact.yml') }}-${{ steps.target.outputs.key }}- + + - name: Validate or build baselines + id: prepare + env: + REFRESH: ${{ github.event_name == 'push' && steps.restore.outputs.cache-hit != 'true' }} + run: | + args=() + if [[ "$REFRESH" == 'true' ]]; then args+=(--refresh); fi + python scripts/impact-gate.py prepare --base "$BASE_REF" --cache-dir "$BASELINE_DIR" "${args[@]}" + + - name: Save main baselines + if: github.event_name == 'push' && steps.prepare.outputs.rebuilt == 'true' + uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4 + with: + path: ${{ env.BASELINE_DIR }} + key: ${{ steps.restore.outputs.cache-primary-key }} + + - name: Report pull request impact + if: github.event_name == 'pull_request' + run: python scripts/impact-gate.py report --base "$BASE_REF" --cache-dir "$BASELINE_DIR" diff --git a/.gitignore b/.gitignore index 320a815ee..268d67b85 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,10 @@ dist mise.local.toml .env +# Local ImpactGate output (CI uses runner.temp) and Python test bytecode. +/.impact-baseline/ +__pycache__/ + # Three generated WASM declaration files are tracked deliberately. They have to # be re-included from HERE: git cannot re-include a file whose parent directory # is excluded, and the bare `dist` above excludes the directory itself, so a diff --git a/docs/plans/2026-10-08-impact-gate-rollout.md b/docs/plans/2026-10-08-impact-gate-rollout.md new file mode 100644 index 000000000..20896b3e0 --- /dev/null +++ b/docs/plans/2026-10-08-impact-gate-rollout.md @@ -0,0 +1,88 @@ +# Advisory ImpactGate rollout evidence + +Calibration date: 2026-10-08. This implements [Stack issue #1137](https://github.com/cipherstash/stack/issues/1137). No branch-protection or existing FTA/CRAP policy changes are intended. + +## Tool and source provenance + +- Stack base: `14bcc638769504ebc4dbb062d0c48fa7e935fe43`. +- Hyper reproduction source: `cd294f1c02e9718c7627a8d7fe7dd35ecd3ee407`, `.scratch/impact-gate/issues/05-reduce-the-lizard-typescript-span-defect.md`. +- Installed and exercised: `impact-gate==0.4.1`, `lizard==1.23.0`. Lizard 1.24.0 is excluded by upstream ImpactGate package metadata; it is not an upgrade candidate for this rollout. +- Local measurement runtime: Python 3.13 on macOS arm64. Dependencies resolved during calibration: PyYAML 6.0.3, pygments 2.21.0, pathspec 1.1.1. Runtime timings are local evidence, not hosted-runner guarantees. +- Official release source: [CLI](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/cli.py), [baseline](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/baseline.py), [measurement scope](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/core/config.py). Flags and baseline serialization were also inspected in the installed wheel. + +## Parser checks + +Both of Hyper's minimal synthetic reproductions were rerun through the selected Lizard parser. They remain defective: + +| Input | Observed parser output | Expected source span | +| --- | --- | --- | +| TSX `f<{ a: 1 }>()` preceding two functions | No functions | `small` 3–5 and `big` 7–10 | +| Same source with `.ts` suffix | `small` 3–5, CC 1; `big` 7–10, CC 2 | Matches | +| Callable parameter `first(cb: (t: string) => string)` in TS and TSX | `first` 1–1, CC 1 | 1–4, including its conditional | +| Following `probe` function in that reproduction | 6–10, CC 3 | Matches | + +Manual comparison of representative Stack source at the base SHA also found a real TypeScript span problem. In `languages/typescript/packages/stack/src/eql/v3/selector-path.ts`, `parseSelectorSegments` occupies lines 32–69 but is reported as 32–72 (CC 9); `jsonPathOf` occupies 72–74 but is reported as 72–83 (CC 1). The parser absorbs the start of the next function. Therefore ordinary TypeScript can be mismeasured even without either synthetic trigger. + +The inspected Rust `is_fresh_at` in `packages/stack-encrypt/src/keyset.rs` is correctly reported at 103–105 (CC 1). Go `StatusError` and `PackedResult` in `languages/golang/internal/guest/status.go` are correctly reported at 45–106 (CC 29) and 111–116 (CC 2). These are sample checks, not proof that either parser is universally correct. + +Keep the report advisory. Do not modify production source to satisfy this parser. These silent span defects are different from process failures: successful CLI exit cannot establish accurate function boundaries. + +## Coverage and interpretation + +Both reports use the same generated-output exclusions. The second excludes test files; it cannot remove inline Rust `#[cfg(test)]` modules from production files. SQL and TypeScript `.mts`/`.cts` are outside upstream's default language map. A no-eligible-source result is no assessment of those changes. + +The upstream baseline serializes `_meta` (`tool`, `n`, `head`, `base_ref`) plus an ascending integer `distribution`. Empty source observations are omitted. Its history walk follows the mainline and recursively extracts landed branch changes; a mainline limit does not strictly cap observations. The project contribution to grading is `n / (n + 200)` with the selected prior weight. + +## Actions verification still required + +Local calibration and tests cannot verify GitHub cache scoping, main-run cache saves, later PR restores, or the rendered job summary. After the workflow is available in Actions, record links to a cold run, a main run saving the pair, and a PR run restoring it. Verify both report sections, policy/version cache invalidation, malformed-pair rebuilding, and explicit seed-only disclosure. Do not mark these checks complete based on local cache simulation. + +## Imported history and rename checks + +The candidate 200-entry mainline window runs from `f435ce5764ea24e5c3af0904ea7efe22eed188a0` (2026-07-09) through the base SHA (2026-10-08). It includes the protect-ffi and EQL imports, the stack-crate import, and the TypeScript directory move. No subject exclusion is proposed: legitimate large changes stay in calibration. + +A diagnostic pass with upstream's default scope scored historical first-parent net diffs through the installed engine (not a replacement score formula): + +| Revision and change | Eligible files | Impact | +| --- | ---: | ---: | +| `7109994002e277074765662c8ff35d41ef6c10f9`, exact TypeScript directory move | 856 | 0 | +| `d2772b0c520e32546873a73d70cc6dec89507257`, Prisma package-name rename | 132 | 531,564 | +| `f1895d8bbf6c9753ed48787df6abc32f253bbedb`, stack-crate import merge net diff | 261 | 104,199,813 | +| `4a203453c1973c849f540079c69e7fffcf16fc62`, Go generator merge net diff | 158 | 169,721,862 | + +Git reports 1,017 exact, zero-line-change renames for the directory move. ImpactGate detects these as zero impact while still counting eligible files. A mechanical rename that edits source text can score materially, as the package-name rename shows. Merge net scores above are diagnostic comparisons; the baseline walker recursively extracts branch observations rather than necessarily inserting each displayed merge net score. + +There were no entirely test-only, SQL-only, or generated-only net diffs in the selected mainline sample. For coverage diagnostics, path-filtered subsets were therefore used and must not be mistaken for complete PR scores: 51 generated paths from `c851db593af30767c5f0dc028aec53aa845253e6` produced no eligible files; three SQL paths from `f61542b7eb529e673dcf505db3a61ee426512c76` produced no eligible files. The Go-test subset of `4a203453c1973c849f540079c69e7fffcf16fc62` contains 45 changed paths and scores 35 eligible files at 14,164,500 under default scope. The isolated real-CLI tests separately exercise complete small Git changes. + +The same historical Go-test subset was checked with the rollout policies: `all.yml` retains those 35 eligible files (impact 14,164,500), while `source.yml` has no eligible files (impact 0). This confirms the test-file distinction against actual repository content, in addition to isolated test histories. + +## Reproducing baseline and reporting measurements + +Install the pinned tools in a disposable virtual environment. From the repository root, repeat the following for `all` and `source`; keep generated JSON outside the checkout: + +```sh +impact-gate baseline --base-ref 14bcc638769504ebc4dbb062d0c48fa7e935fe43 \ + --max-commits 200 --baseline-file /tmp/stack-impact-all.json \ + --measure-config .github/impact-gate/all.yml +impact-gate score --mode range --base 415b62cd7bdd957f6b9df277115ce1e955a10510 \ + --curve --enforcement warn --warn-percentile 90 --block-percentile 98 \ + --curve-prior-weight 200 --format markdown \ + --baseline-file /tmp/stack-impact-all.json \ + --measure-config .github/impact-gate/all.yml +``` + +The historical reporting range intentionally covers the latest three mainline entries at the recorded base (through the EQL plan targets and Go generator changes), providing a nonempty mixed-language report. It is a timing sample, not this infrastructure branch's PR score. + +A preliminary all-source baseline using the earlier scope finished in 217.45 seconds with 418 observations. That run preceded the final generated-Go exclusions and is not the adopted baseline. A default-scope exploratory run was interrupted after 177.30 seconds to avoid contention with the policy run; interruption was not a tool failure. Final-policy measurements below supersede those exploratory runs. + +At the recorded base, applying the final scope to tracked filenames yields the following eligibility inventory (eligibility does not guarantee accurate function parsing): + +| Language | Including test files | Excluding test files | +| --- | ---: | ---: | +| TypeScript | 985 | 541 | +| Rust | 337 | 168 | +| Go | 151 | 106 | +| JavaScript | 109 | 29 | +| Python | 5 | 3 | + +There are also 281 tracked SQL files and seven `.mts`/`.cts` files outside the language map. The EQL `eql-domains/src/fixtures` tree is a production catalog DSL and remains eligible; fixture exclusions deliberately do not blanket-match every directory named `fixtures`. Shared `test-kit` code and Go's fake allocator backend in `internal/guest/testing.go` are excluded only by the test-file policy. Generated Go `_stash.go`/`_gen.go` files and declaration outputs are excluded from both. diff --git a/scripts/__tests__/impact-workflow.test.mjs b/scripts/__tests__/impact-workflow.test.mjs new file mode 100644 index 000000000..2c6348750 --- /dev/null +++ b/scripts/__tests__/impact-workflow.test.mjs @@ -0,0 +1,81 @@ +import { describe, expect, it } from 'vitest' +import { readWorkflow } from './lib/workflows.mjs' + +const workflowPath = '.github/workflows/impact.yml' + +describe('advisory change impact', () => { + it('reports every pull request and refreshes main with read-only permissions', () => { + const workflow = readWorkflow(workflowPath) + expect(workflow.on).toEqual({ + pull_request: null, + push: { branches: ['main'] }, + }) + expect(workflow.permissions).toEqual({ contents: 'read' }) + const job = workflow.jobs.impact + expect(job.permissions).toBeUndefined() + expect(job.needs).toBeUndefined() + const checkout = job.steps.find((step) => + step.uses?.startsWith('actions/checkout@'), + ) + expect(checkout.with).toMatchObject({ + 'fetch-depth': 0, + 'persist-credentials': false, + }) + }) + + it('keeps baseline caches compatible and saves only main-push results', () => { + const job = readWorkflow(workflowPath).jobs.impact + const restore = job.steps.find((step) => + step.uses?.startsWith('actions/cache/restore@'), + ) + expect(restore).toBeDefined() + expect(restore.with.key).toContain('hashFiles(') + expect(restore.with.key).toContain("'.github/impact-gate/**'") + expect(restore.with.key).toContain("'scripts/impact-gate.py'") + expect(restore.with.key).toContain("'.github/workflows/impact.yml'") + expect(restore.with.key).toContain('steps.target.outputs.key') + expect(restore.with.key).toContain( + 'github.event.pull_request.base.sha || github.sha', + ) + expect(restore.with['restore-keys']).toBe( + restore.with.key.replace( + // biome-ignore lint/suspicious/noTemplateCurlyInString: literal Actions expression + '-${{ github.event.pull_request.base.sha || github.sha }}', + '-', + ), + ) + const save = job.steps.find((step) => + step.uses?.startsWith('actions/cache/save@'), + ) + expect(save.if).toBe( + "github.event_name == 'push' && steps.prepare.outputs.rebuilt == 'true'", + ) + expect(save.with.path).toBe(restore.with.path) + }) + + it('runs the real-CLI checks and reports PRs without masking tool errors', () => { + const job = readWorkflow(workflowPath).jobs.impact + const checks = job.steps.find((step) => + step.run?.includes('unittest discover'), + ) + expect(checks).toBeDefined() + const report = job.steps.find((step) => + step.run?.includes('scripts/impact-gate.py report'), + ) + expect(report).toBeDefined() + expect(report.if).toBe("github.event_name == 'pull_request'") + expect(report.run).toContain('--base "$BASE_REF"') + expect(report.run).toContain('--cache-dir "$BASELINE_DIR"') + expect(report.run).not.toMatch(/\|\||\|\s*tee/) + expect(job.env.BASE_REF).toContain('github.base_ref') + expect(job['continue-on-error']).toBeUndefined() + expect(job.steps.every((step) => !step['continue-on-error'])).toBe(true) + expect( + job.steps.some((step) => step.uses?.startsWith('officefloor/')), + ).toBe(false) + expect(job.steps.some((step) => step.run?.includes('pnpm install'))).toBe( + false, + ) + expect(job.defaults.run.shell).toBe('bash') + }) +}) diff --git a/scripts/impact-gate.py b/scripts/impact-gate.py new file mode 100644 index 000000000..d2139b81a --- /dev/null +++ b/scripts/impact-gate.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +"""Prepare a compatible baseline pair and publish advisory ImpactGate reports. + +Run from the Git checkout being assessed. Policy belongs to this script's checkout, +so tests can exercise the same workflow against isolated Git histories. +""" +import argparse +import hashlib +import importlib.metadata +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile + +from impact_gate.baseline import build_baseline, save_baseline +from impact_gate.core.config import MeasureConfig + +ROOT = Path(__file__).resolve().parents[1] +POLICY = ROOT / '.github' / 'impact-gate' +HISTORY_COMMITS = 200 +SCOPES = {'all': 'Including test files', 'source': 'Excluding test files'} + + +def run(*args: str) -> str: + return subprocess.check_output(args, text=True).strip() + + +def cli(*args: str) -> str: + return run(sys.executable, '-m', 'impact_gate', *args) + + +def identity() -> str: + digest = hashlib.sha256(Path(__file__).read_bytes()) + for path in sorted(POLICY.glob('*')): + if path.is_file(): + digest.update(path.name.encode()) + digest.update(path.read_bytes()) + for package in ('impact-gate', 'lizard'): + digest.update(importlib.metadata.version(package).encode()) + return digest.hexdigest() + + +def read_baseline(path: Path, head: str) -> int: + data = json.loads(path.read_text()) + meta, values = data['_meta'], data['distribution'] + if (meta['tool'] != 'impact-gate' or meta['head'] != head + or type(meta['n']) is not int or not isinstance(values, list) + or meta['n'] != len(values) + or any(type(v) is not int or v < 0 for v in values) + or values != sorted(values)): + raise ValueError(f'Invalid baseline: {path}') + return meta['n'] + + +def valid_pair(cache: Path, policy: str, base: str) -> dict | None: + try: + meta = json.loads((cache / 'metadata.json').read_text()) + if meta['policy'] != policy or meta['base_ref'] != base: + return None + # A fallback cache must come from this target's landed history. + result = subprocess.run(['git', 'merge-base', '--is-ancestor', meta['head'], base], + capture_output=True) + if result.returncode != 0: + return None + for scope in SCOPES: + path = cache / f'{scope}.json' + read_baseline(path, meta['head']) + if hashlib.sha256(path.read_bytes()).hexdigest() != meta['hashes'][scope]: + return None + return meta + except (OSError, ValueError, KeyError, TypeError): + return None + + +def output(name: str, value: str) -> None: + if os.environ.get('GITHUB_OUTPUT'): + with open(os.environ['GITHUB_OUTPUT'], 'a') as out: + out.write(f'{name}={value}\n') + print(f'{name}={value}') + + +def prepare(args: argparse.Namespace, policy: str, head: str) -> None: + if not args.refresh and valid_pair(args.cache_dir, policy, args.base): + output('rebuilt', 'false') + return + args.cache_dir.mkdir(parents=True, exist_ok=True) + # Stage both files: failed generation never advertises a usable half-pair. + with tempfile.TemporaryDirectory(dir=args.cache_dir) as temporary: + staged = Path(temporary) + hashes = {} + for scope in SCOPES: + path = staged / f'{scope}.json' + # Upstream's CLI rejects n=0. Its library distinguishes a valid empty + # history from failures without matching stderr or masking exceptions. + baseline = build_baseline('.', MeasureConfig.load(str(POLICY / f'{scope}.yml')), + base_ref=head, max_commits=HISTORY_COMMITS) + save_baseline(baseline, str(path)) + read_baseline(path, head) + hashes[scope] = hashlib.sha256(path.read_bytes()).hexdigest() + meta = {'policy': policy, 'base_ref': args.base, 'head': head, + 'max_commits': HISTORY_COMMITS, 'hashes': hashes} + for scope in SCOPES: + (staged / f'{scope}.json').replace(args.cache_dir / f'{scope}.json') + (staged / 'metadata.json').write_text(json.dumps(meta, indent=2) + '\n') + (staged / 'metadata.json').replace(args.cache_dir / 'metadata.json') + output('rebuilt', 'true') + + +def publish(text: str) -> None: + print(text) + if os.environ.get('GITHUB_STEP_SUMMARY'): + with open(os.environ['GITHUB_STEP_SUMMARY'], 'a') as summary: + summary.write(text + '\n') + + +def report(args: argparse.Namespace, policy: str) -> None: + meta = valid_pair(args.cache_dir, policy, args.base) + if meta is None: + raise ValueError('Baseline pair is missing or incompatible; run prepare first.') + publish('# Advisory change impact\n\n' + 'SQL and TypeScript .mts/.cts files are unsupported. No eligible source means ' + 'no quality assessment. Excluding test files retains inline Rust tests. ' + 'TypeScript function-span defects and mechanical renames can distort scores.\n\n' + f'Baseline target: `{args.base}`; commit: `{meta["head"]}`. ' + f'History: latest {HISTORY_COMMITS} first-parent commits; recursively expanded ' + 'merge observations may exceed this limit. p90/p98 are advisory; prior weight: 200.') + for scope, label in SCOPES.items(): + path = args.cache_dir / f'{scope}.json' + n = read_baseline(path, meta['head']) + publish(f'## {label}\n\nProject observations: {n}; project weight: {n / (n + 200):.4f}.' + + (' **Seed-only grading: no eligible history observations.**' if n == 0 else '')) + result = cli('score', '--config', str(POLICY / 'gate.yml'), + '--mode', 'range', '--base', args.base, '--curve', + '--enforcement', 'warn', '--format', 'markdown', '--warn-percentile', '90', + '--block-percentile', '98', '--curve-prior-weight', '200', + '--baseline-file', str(path), '--measure-config', str(POLICY / f'{scope}.yml')) + publish(result) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('command', choices=('prepare', 'report')) + parser.add_argument('--base', required=True) + parser.add_argument('--cache-dir', type=Path, required=True) + parser.add_argument('--refresh', action='store_true') + args = parser.parse_args() + args.cache_dir = args.cache_dir.resolve() + try: + head = run('git', 'rev-parse', '--verify', args.base + '^{commit}') + policy = identity() + if args.command == 'prepare': + prepare(args, policy, head) + else: + report(args, policy) + except (subprocess.CalledProcessError, OSError, ValueError, KeyError, TypeError, + importlib.metadata.PackageNotFoundError) as error: + publish(f'**ImpactGate analysis failed:** {error}') + return error.returncode if isinstance(error, subprocess.CalledProcessError) else 1 + return 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/scripts/tests/test_impact_gate.py b/scripts/tests/test_impact_gate.py new file mode 100644 index 000000000..f5dc7b8b0 --- /dev/null +++ b/scripts/tests/test_impact_gate.py @@ -0,0 +1,171 @@ +"""Exercise the workflow driver against the pinned real CLI and Git histories.""" +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +DRIVER = Path(__file__).resolve().parents[1] / 'impact-gate.py' + + +class ImpactWorkflowTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.repo = Path(self.temp.name) + self.cache = self.repo / 'cache' + self.summary = self.repo / 'summary.md' + self.output = self.repo / 'output' + self.git('init', '-b', 'main') + self.git('config', 'user.email', 'test@example.com') + self.git('config', 'user.name', 'Test') + self.git('config', 'commit.gpgsign', 'false') + self.commit('README.md', 'fixture\n') + self.commit('src/main.ts', 'export function count(x: number) {\n return x + 1;\n}\n') + + def git(self, *args): + return subprocess.check_output(['git', *args], cwd=self.repo, text=True).strip() + + def commit(self, name, content): + path = self.repo / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content) + self.git('add', name) + self.git('commit', '-qm', name) + return self.git('rev-parse', 'HEAD') + + def run_driver(self, command, *args, expect=0): + result = subprocess.run([sys.executable, str(DRIVER), command, + '--base', 'main', '--cache-dir', str(self.cache), *args], + cwd=self.repo, text=True, capture_output=True, + env={**os.environ, 'GITHUB_OUTPUT': str(self.output), + 'GITHUB_STEP_SUMMARY': str(self.summary)}) + self.assertEqual(result.returncode, expect, result.stdout + result.stderr) + return result + + def test_cold_baseline_then_reuse_and_refresh(self): + self.run_driver('prepare') + self.assertIn('rebuilt=true', self.output.read_text()) + original = (self.cache / 'metadata.json').read_bytes() + self.output.write_text('') + self.run_driver('prepare') + self.assertEqual(self.output.read_text(), 'rebuilt=false\n') + self.assertEqual((self.cache / 'metadata.json').read_bytes(), original) + self.run_driver('prepare', '--refresh') + self.assertTrue(self.output.read_text().endswith('rebuilt=true\n')) + self.git('checkout', '-qb', 'feature') + self.commit('src/main.ts', 'export function count(x: number) {\n if (x > 0) return x + 2;\n return 0;\n}\n') + self.run_driver('report') + summary = self.summary.read_text() + self.assertIn('Including test files', summary) + self.assertIn('Excluding test files', summary) + self.assertIn('Project observations: 1', summary) + self.assertIn('src/main.ts', summary) + + def test_corrupt_pair_and_policy_mismatch_rebuild(self): + self.run_driver('prepare') + for damage in ('truncated', 'shape', 'policy'): + with self.subTest(damage=damage): + if damage == 'truncated': + (self.cache / 'all.json').write_text('{') + elif damage == 'shape': + (self.cache / 'source.json').write_text('[]') + else: + path = self.cache / 'metadata.json' + metadata = json.loads(path.read_text()) + metadata['policy'] = 'old tool or measure policy' + path.write_text(json.dumps(metadata)) + self.output.write_text('') + self.run_driver('prepare') + self.assertEqual(self.output.read_text(), 'rebuilt=true\n') + self.run_driver('report') + + def test_target_provenance_rejects_unrelated_history(self): + self.run_driver('prepare') + self.git('checkout', '--orphan', 'other') + self.git('rm', '-rf', '--cached', '.') + self.commit('other.ts', 'export function other() { return 2; }\n') + self.git('branch', '-f', 'main', 'HEAD') + self.output.write_text('') + self.run_driver('prepare') + self.assertEqual(self.output.read_text(), 'rebuilt=true\n') + + def test_ancestor_cache_reused_but_other_target_ref_rebuilds(self): + self.run_driver('prepare') + self.commit('src/next.go', 'package next\nfunc Next() int { return 1 }\n') + self.output.write_text('') + self.run_driver('prepare') + self.assertEqual(self.output.read_text(), 'rebuilt=false\n') + self.git('branch', 'release') + self.output.write_text('') + self.run_driver('prepare', '--base', 'release') + self.assertEqual(self.output.read_text(), 'rebuilt=true\n') + + def test_production_fixture_catalog_survives_test_filter(self): + self.run_driver('prepare') + self.git('checkout', '-qb', 'feature') + self.commit('packages/eql/crates/eql-domains/src/fixtures/catalog.rs', + 'pub fn catalog() -> i32 { 1 }\n') + self.commit('packages/test-kit/src/helper.ts', 'export function helper() { return 1; }\n') + self.commit('pkg/generated_stash.go', 'package pkg\nfunc Generated() int { return 1 }\n') + self.run_driver('report') + including, excluding = self.summary.read_text().split('## Excluding test files') + self.assertIn('catalog.rs', excluding) + self.assertIn('helper.ts', including) + self.assertNotIn('helper.ts', excluding) + self.assertNotIn('generated_stash.go', including) + + def test_analysis_failure_is_not_hidden_by_summary(self): + result = self.run_driver('prepare', '--base', 'missing-target', expect=128) + self.assertIn('analysis failed', self.summary.read_text()) + self.assertNotIn('rebuilt=true', result.stdout) + + def test_scopes_and_empty_history(self): + self.git('checkout', '-qb', 'feature') + self.commit('src/main.test.ts', 'export function test() { return 3; }\n') + self.run_driver('prepare') + self.run_driver('report') + including, excluding = self.summary.read_text().split('## Excluding test files') + self.assertIn('main.test.ts', including) + self.assertNotIn('main.test.ts', excluding) + self.git('branch', '-f', 'main', 'HEAD~2') + self.run_driver('prepare') + self.summary.write_text('') + self.run_driver('report') + self.assertIn('Seed-only grading', self.summary.read_text()) + + def test_no_eligible_source_and_mechanical_rename(self): + self.run_driver('prepare') + self.git('checkout', '-qb', 'feature') + self.commit('generated/client.ts', 'export function generated() { return 3; }\n') + self.commit('schema.sql', 'select 1;\n') + self.commit('src/types.mts', 'export const value = 1;\n') + self.run_driver('report') + summary = self.summary.read_text() + self.assertIn('No eligible source', summary) + self.assertNotIn('generated/client.ts', summary) + self.assertNotIn('schema.sql', summary) + self.assertNotIn('src/types.mts', summary) + self.git('mv', 'src/main.ts', 'src/renamed.ts') + self.git('commit', '-qm', 'rename') + self.summary.write_text('') + self.run_driver('report') + self.assertIn('mechanical renames', self.summary.read_text()) + + def test_high_impact_is_advisory_and_local_policy_is_ignored(self): + self.run_driver('prepare') + self.git('checkout', '-qb', 'feature') + self.commit('.impact-gate.yml', 'cognitive_max: 0\nenforcement: block\n') + for index in range(12): + self.commit(f'src/large{index}.ts', 'export function large(x: number) {\n' + + ''.join(f' if (x === {i}) return {i};\n' for i in range(60)) + + ' return x;\n}\n') + self.run_driver('report') + self.assertIn('large', self.summary.read_text()) + self.assertIn('WARN', self.summary.read_text()) + + +if __name__ == '__main__': + unittest.main() From 0a8476146487b6d753604cc6b394bc7955025b6b Mon Sep 17 00:00:00 2001 From: Toby Hede Date: Thu, 8 Oct 2026 10:58:07 +1100 Subject: [PATCH 2/5] test: record ImpactGate calibration and review checks --- docs/plans/2026-10-08-impact-gate-rollout.md | 30 ++++++++++++++++++++ scripts/__tests__/impact-workflow.test.mjs | 6 ++++ 2 files changed, 36 insertions(+) diff --git a/docs/plans/2026-10-08-impact-gate-rollout.md b/docs/plans/2026-10-08-impact-gate-rollout.md index 20896b3e0..212773fba 100644 --- a/docs/plans/2026-10-08-impact-gate-rollout.md +++ b/docs/plans/2026-10-08-impact-gate-rollout.md @@ -86,3 +86,33 @@ At the recorded base, applying the final scope to tracked filenames yields the f | Python | 5 | 3 | There are also 281 tracked SQL files and seven `.mts`/`.cts` files outside the language map. The EQL `eql-domains/src/fixtures` tree is a production catalog DSL and remains eligible; fixture exclusions deliberately do not blanket-match every directory named `fixtures`. Shared `test-kit` code and Go's fake allocator backend in `internal/guest/testing.go` are excluded only by the test-file policy. Generated Go `_stash.go`/`_gen.go` files and declaration outputs are excluded from both. + +## Final calibration and operational budget + +The final matching baseline pair was generated sequentially from the pinned base SHA using the exact policies below. Each process exited 0; generation was fresh, with no restored baseline. Git objects and the OS filesystem cache were already present, so “cold” here means cold derived baseline, not a cold machine or repository clone. Installation and checkout time are excluded. + +| Policy | Generation time | Observations (`n`) | Distribution minimum | Maximum | +| --- | ---: | ---: | ---: | ---: | +| Including test files | 194.15 s | 418 | 0 | 55,662,006,240 | +| Excluding test files | 85.40 s | 344 | 0 | 210,644,488 | +| Sequential pair | **279.55 s** | | | | + +Policy SHA-256 digests used for these measurements: + +- `all.yml`: `2e45fcb24fe9535a26c99d1685c0b3fbce1c1285e65f5c4474cbba179110b655` +- `source.yml`: `810eaafd26b3c5df42b5992395855ac40a5c2f1c516be469fc9c8626f489bb6e` + +Using those existing baseline files, the sample range described above took 1.02 seconds including test files and 0.60 seconds excluding them, **1.63 seconds combined**. This measures local baseline reuse, not an Actions cache hit. The all-source report counted 172 files, impact 545,037,040, p99.51; the report excluding test files counted 119 files, impact 310,445,058, p99.82. Both exceeded p98, showed WARN, and exited **0**, directly confirming advisory behavior on a large real Stack change. The underlying CLI's canned wording about future blocking does not change the configured warn enforcement. + +Adopt **200 recent first-parent entries** and a **20-minute workflow timeout**. The observed cold pair fits with more than four times its local measured runtime available for a slower runner, tool installation, checkout, and reporting. This is an operational allowance, not a hosted-runner SLA. There is no reason from these timings to narrow the window to omit the imports; 50- or 30-entry alternatives were therefore not needed. The recursive history yielded more observations than mainline entries, so report `n` rather than describing the baseline as “200 changes.” The project weights are approximately 0.6764 and 0.6324; the remaining grading weight comes from the shipped seed distribution. + +The large historical maximum and parser defects argue for retaining advisory enforcement. Do not trim large observations to make scores look better. Revisit the time budget if future imports or unusually deep merge histories approach the timeout; failed analysis must remain visible. + +## Implementation validation + +- The nine real-tool tests passed against temporary Git histories: cold generation, reuse/refresh, invalid and incompatible cache rebuilding, target ancestry, production-fixture retention, test-file filtering, unsupported/generated changes, renames, explicit seed-only grading, high-impact advisory success, and visible operational failure. Generation calls the pinned upstream baseline API because its CLI rejects a genuinely empty distribution; reporting still invokes the real CLI. +- The repository script suite passed: 72 files, 1,269 tests, 28 skips. A subsequent review added one generated-scope parity guard; the focused workflow suite then passed all four tests. That guard keeps the duplicated exclusion policies aligned. +- Repository-wide typechecking passed all 16 Turbo tasks. The workflow test also passed direct TypeScript check-JS validation; the Python driver/tests passed mypy 1.18.2 with `--ignore-missing-imports --follow-imports skip --check-untyped-defs`. Keep mypy's cache outside the checkout when running locally, since Biome otherwise scans its generated JSON. +- All 28 supply-chain tests passed. The workflow passed `actionlint`; the repository Biome check exited successfully with existing warnings/information diagnostics and no errors. +- The full package suite was attempted with `pnpm test --continue`: 12 of 19 Turbo tasks passed. Auth, profile, Stack, migrate, wizard, CLI, and Supabase tasks failed with missing local native bindings. Two auth package-packing tests additionally failed: sandboxed npm could not write its cache; rerunning just those tests outside the sandbox removed that error but exposed their existing assumption about the shape of npm's JSON output (`[0].files` was undefined). No package runtime or packaging source is changed by this work, and the full package suite is not claimed green. +- Separate standards/spec reviews found no documented-standard or implementation-correctness violations. The standards review's shared-exclusion drift concern is covered by the new guard. Final calibration evidence resolves the timing-evidence gap; the hosted Actions checks described above remain rollout work. diff --git a/scripts/__tests__/impact-workflow.test.mjs b/scripts/__tests__/impact-workflow.test.mjs index 2c6348750..c4cf52cd7 100644 --- a/scripts/__tests__/impact-workflow.test.mjs +++ b/scripts/__tests__/impact-workflow.test.mjs @@ -4,6 +4,12 @@ import { readWorkflow } from './lib/workflows.mjs' const workflowPath = '.github/workflows/impact.yml' describe('advisory change impact', () => { + it('applies every generated-output exclusion to both report scopes', () => { + const all = readWorkflow('.github/impact-gate/all.yml') + const source = readWorkflow('.github/impact-gate/source.yml') + expect(source.ignore).toEqual(expect.arrayContaining(all.ignore)) + }) + it('reports every pull request and refreshes main with read-only permissions', () => { const workflow = readWorkflow(workflowPath) expect(workflow.on).toEqual({ From 79e9324978702121553e3b7e97464d0467e7afa9 Mon Sep 17 00:00:00 2001 From: Toby Hede Date: Fri, 9 Oct 2026 11:57:23 +1100 Subject: [PATCH 3/5] fix(ci): ImpactGate failures reach the step summary with the tool's stderr --- scripts/impact-gate.py | 24 +++++++++++++++++----- scripts/tests/test_impact_gate.py | 34 +++++++++++++++++++++++++++++-- 2 files changed, 51 insertions(+), 7 deletions(-) diff --git a/scripts/impact-gate.py b/scripts/impact-gate.py index d2139b81a..c834b59c8 100644 --- a/scripts/impact-gate.py +++ b/scripts/impact-gate.py @@ -24,7 +24,19 @@ def run(*args: str) -> str: - return subprocess.check_output(args, text=True).strip() + # Capture stderr: CalledProcessError's message omits it, and the summary needs it. + return subprocess.run(args, text=True, check=True, capture_output=True).stdout.strip() + + +def describe(error: Exception) -> str: + if not isinstance(error, subprocess.CalledProcessError): + return f'{type(error).__name__}: {error}' + args = [str(arg) for arg in error.cmd] + # Name the tool and subcommand, not the absolute-path argv. + command = ' '.join(args[2:4] if args[1:2] == ['-m'] else args[:2]) + text = f'`{command}` exited with status {error.returncode}.' + stderr = (error.stderr or '').strip() + return text + (f'\n\n```\n{stderr}\n```' if stderr else '') def cli(*args: str) -> str: @@ -154,10 +166,12 @@ def main() -> int: prepare(args, policy, head) else: report(args, policy) - except (subprocess.CalledProcessError, OSError, ValueError, KeyError, TypeError, - importlib.metadata.PackageNotFoundError) as error: - publish(f'**ImpactGate analysis failed:** {error}') - return error.returncode if isinstance(error, subprocess.CalledProcessError) else 1 + except subprocess.CalledProcessError as error: + publish(f'**ImpactGate analysis failed:** {describe(error)}') + return error.returncode + except Exception as error: # A broken policy file must still reach the summary. + publish(f'**ImpactGate analysis failed:** {describe(error)}') + return 1 return 0 diff --git a/scripts/tests/test_impact_gate.py b/scripts/tests/test_impact_gate.py index f5dc7b8b0..0bbeecca2 100644 --- a/scripts/tests/test_impact_gate.py +++ b/scripts/tests/test_impact_gate.py @@ -2,6 +2,7 @@ import json import os from pathlib import Path +import shutil import subprocess import sys import tempfile @@ -36,8 +37,19 @@ def commit(self, name, content): self.git('commit', '-qm', name) return self.git('rev-parse', 'HEAD') - def run_driver(self, command, *args, expect=0): - result = subprocess.run([sys.executable, str(DRIVER), command, + def policy_copy(self, **files): + # Policy is bound to the driver's checkout, so damage a copied checkout. + root = Path(self.temp.name) / 'driver-checkout' + shutil.copytree(DRIVER.parents[1] / '.github' / 'impact-gate', + root / '.github' / 'impact-gate') + (root / 'scripts').mkdir() + shutil.copy(DRIVER, root / 'scripts' / DRIVER.name) + for name, content in files.items(): + (root / '.github' / 'impact-gate' / name).write_text(content) + return root / 'scripts' / DRIVER.name + + def run_driver(self, command, *args, expect=0, driver=DRIVER): + result = subprocess.run([sys.executable, str(driver), command, '--base', 'main', '--cache-dir', str(self.cache), *args], cwd=self.repo, text=True, capture_output=True, env={**os.environ, 'GITHUB_OUTPUT': str(self.output), @@ -122,6 +134,24 @@ def test_analysis_failure_is_not_hidden_by_summary(self): self.assertIn('analysis failed', self.summary.read_text()) self.assertNotIn('rebuilt=true', result.stdout) + def test_malformed_policy_failure_reaches_summary(self): + for content in ('ignore: [unclosed\n', '- a\n'): + with self.subTest(content=content): + shutil.rmtree(Path(self.temp.name) / 'driver-checkout', ignore_errors=True) + self.summary.write_text('') + driver = self.policy_copy(**{'all.yml': content}) + self.run_driver('prepare', '--refresh', expect=1, driver=driver) + self.assertIn('ImpactGate analysis failed', self.summary.read_text()) + + def test_score_failure_summary_includes_tool_stderr(self): + self.run_driver('prepare') + driver = self.policy_copy(**{'gate.yml': 'warn_percentile: 99\nblock_percentile: 50\n'}) + # The copied checkout has a different policy identity; rebuild its pair. + self.run_driver('prepare', '--refresh', driver=driver) + self.summary.write_text('') + self.run_driver('report', expect=1, driver=driver) + self.assertIn('warn_percentile must be <= block_percentile', self.summary.read_text()) + def test_scopes_and_empty_history(self): self.git('checkout', '-qb', 'feature') self.commit('src/main.test.ts', 'export function test() { return 3; }\n') From 523fed7b29c5cd331beb88844427b27278e9e2a7 Mon Sep 17 00:00:00 2001 From: Toby Hede Date: Fri, 9 Oct 2026 11:57:23 +1100 Subject: [PATCH 4/5] fix(ci): give main pushes their own ImpactGate group and save PR baselines --- .github/workflows/impact.yml | 14 +- docs/plans/2026-10-08-impact-gate-rollout.md | 3 +- scripts/__tests__/impact-workflow.test.mjs | 127 ++++++++++++++++++- 3 files changed, 135 insertions(+), 9 deletions(-) diff --git a/.github/workflows/impact.yml b/.github/workflows/impact.yml index 456ba322c..e2ee9c6a2 100644 --- a/.github/workflows/impact.yml +++ b/.github/workflows/impact.yml @@ -8,8 +8,11 @@ on: permissions: contents: read +# A newer push to the same pull request supersedes the older run. Main pushes +# are grouped per SHA instead: one shared main group cancelled a merge's run +# before it saved that SHA's baselines, leaving PRs on it to rebuild. concurrency: - group: impact-${{ github.ref }} + group: impact-${{ github.event_name == 'push' && github.sha || github.ref }} cancel-in-progress: true jobs: @@ -33,7 +36,7 @@ jobs: python-version: '3.12' - name: Install analysis tools - run: python -m pip install -r .github/impact-gate/requirements.txt + run: python -m pip install --require-hashes --no-deps -r .github/impact-gate/requirements.txt - name: Check reporting behavior with the real CLI run: python -m unittest discover -s scripts/tests -p test_impact_gate.py -v @@ -75,8 +78,11 @@ jobs: if [[ "$REFRESH" == 'true' ]]; then args+=(--refresh); fi python scripts/impact-gate.py prepare --base "$BASE_REF" --cache-dir "$BASELINE_DIR" "${args[@]}" - - name: Save main baselines - if: github.event_name == 'push' && steps.prepare.outputs.rebuilt == 'true' + # Any rebuilt pair is saved, including on pull requests: a PR's cache is + # scoped to its own ref, so it cannot replace main's, and later pushes to + # the PR reuse it instead of rebuilding both scopes again. + - name: Save baselines + if: steps.prepare.outputs.rebuilt == 'true' uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4 with: path: ${{ env.BASELINE_DIR }} diff --git a/docs/plans/2026-10-08-impact-gate-rollout.md b/docs/plans/2026-10-08-impact-gate-rollout.md index 212773fba..3ddd9a6f0 100644 --- a/docs/plans/2026-10-08-impact-gate-rollout.md +++ b/docs/plans/2026-10-08-impact-gate-rollout.md @@ -8,6 +8,7 @@ Calibration date: 2026-10-08. This implements [Stack issue #1137](https://github - Hyper reproduction source: `cd294f1c02e9718c7627a8d7fe7dd35ecd3ee407`, `.scratch/impact-gate/issues/05-reduce-the-lizard-typescript-span-defect.md`. - Installed and exercised: `impact-gate==0.4.1`, `lizard==1.23.0`. Lizard 1.24.0 is excluded by upstream ImpactGate package metadata; it is not an upgrade candidate for this rollout. - Local measurement runtime: Python 3.13 on macOS arm64. Dependencies resolved during calibration: PyYAML 6.0.3, pygments 2.21.0, pathspec 1.1.1. Runtime timings are local evidence, not hosted-runner guarantees. +- CI installs `.github/impact-gate/requirements.txt`, which pins those transitive dependencies at the same versions with hashes for every distribution, using `pip install --require-hashes --no-deps`. Its header records the regeneration command; Dependabot proposes pip updates for that directory. - Official release source: [CLI](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/cli.py), [baseline](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/baseline.py), [measurement scope](https://github.com/officefloor/ImpactGate/blob/v0.4.1/impact_gate/core/config.py). Flags and baseline serialization were also inspected in the installed wheel. ## Parser checks @@ -35,7 +36,7 @@ The upstream baseline serializes `_meta` (`tool`, `n`, `head`, `base_ref`) plus ## Actions verification still required -Local calibration and tests cannot verify GitHub cache scoping, main-run cache saves, later PR restores, or the rendered job summary. After the workflow is available in Actions, record links to a cold run, a main run saving the pair, and a PR run restoring it. Verify both report sections, policy/version cache invalidation, malformed-pair rebuilding, and explicit seed-only disclosure. Do not mark these checks complete based on local cache simulation. +Local calibration and tests cannot verify GitHub cache scoping, cache saves, later PR restores, or the rendered job summary. Any run that rebuilds the pair saves it, pull requests included; a PR's cache is scoped to its own ref, so it cannot replace main's, and it spares later pushes to that PR a full rebuild. Main pushes run in per-SHA concurrency groups so a quick second merge cannot cancel the first before it saves. After the workflow is available in Actions, record links to a cold run, a main run saving the pair, a PR run restoring it, and a PR run reusing its own saved pair. Verify both report sections, policy/version cache invalidation, malformed-pair rebuilding, and explicit seed-only disclosure. Do not mark these checks complete based on local cache simulation. ## Imported history and rename checks diff --git a/scripts/__tests__/impact-workflow.test.mjs b/scripts/__tests__/impact-workflow.test.mjs index c4cf52cd7..0ccd47b47 100644 --- a/scripts/__tests__/impact-workflow.test.mjs +++ b/scripts/__tests__/impact-workflow.test.mjs @@ -1,8 +1,45 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' import { describe, expect, it } from 'vitest' +import { REPO_ROOT } from './lib/repo-root.mjs' import { readWorkflow } from './lib/workflows.mjs' const workflowPath = '.github/workflows/impact.yml' +// Evaluates the small subset of the Actions expression language these guards +// need (context paths, string literals, ==, !=, &&, ||, !), so a test asserts +// what an expression DOES for an event rather than how it is spelled. +function evaluate(expression, context) { + const body = String(expression) + .replace(/^\s*\$\{\{\s*/, '') + .replace(/\s*\}\}\s*$/, '') + const js = body + .replace(/\b(github|steps)((?:\.[\w-]+)+)/g, (_, root, path) => + JSON.stringify( + path + .slice(1) + .split('.') + .reduce((value, key) => value?.[key], context[root]) ?? '', + ), + ) + .replace(/==/g, '===') + .replace(/!===/g, '!==') + return new Function(`return (${js})`)() +} + +function interpolate(template, context) { + return String(template).replace(/\$\{\{(.*?)\}\}/g, (_, expression) => + String(evaluate(expression, context)), + ) +} + +const pushEvent = (sha) => ({ + github: { event_name: 'push', ref: 'refs/heads/main', sha }, +}) +const pullRequestEvent = (sha) => ({ + github: { event_name: 'pull_request', ref: 'refs/pull/7/merge', sha }, +}) + describe('advisory change impact', () => { it('applies every generated-output exclusion to both report scopes', () => { const all = readWorkflow('.github/impact-gate/all.yml') @@ -29,7 +66,7 @@ describe('advisory change impact', () => { }) }) - it('keeps baseline caches compatible and saves only main-push results', () => { + it('keeps baseline caches compatible and saves every rebuilt pair', () => { const job = readWorkflow(workflowPath).jobs.impact const restore = job.steps.find((step) => step.uses?.startsWith('actions/cache/restore@'), @@ -53,12 +90,94 @@ describe('advisory change impact', () => { const save = job.steps.find((step) => step.uses?.startsWith('actions/cache/save@'), ) - expect(save.if).toBe( - "github.event_name == 'push' && steps.prepare.outputs.rebuilt == 'true'", - ) + // F4: this used to pin `github.event_name == 'push' && …`, which is the + // defect — a PR with no valid base cache rebuilt both scopes (~280 s) on + // every synchronize. A PR cache is scoped to the PR ref and cannot clobber + // main's, so a rebuilt pair is saved on both events, and never otherwise. + for (const event of [pushEvent('a'), pullRequestEvent('b')]) { + for (const rebuilt of ['true', 'false']) { + const context = { + ...event, + steps: { prepare: { outputs: { rebuilt } } }, + } + expect( + Boolean(evaluate(save.if, context)), + `${event.github.event_name}, rebuilt=${rebuilt}`, + ).toBe(rebuilt === 'true') + } + } expect(save.with.path).toBe(restore.with.path) }) + it('never lets one main push cancel another before its baselines are saved', () => { + // F3: a shared `impact-refs/heads/main` group with cancel-in-progress + // cancelled a merge's run before "Save main baselines", so that SHA never + // got a cache and PRs on it fell back to an ancestor or a full rebuild. + const { concurrency } = readWorkflow(workflowPath) + const first = interpolate(concurrency.group, pushEvent('1111')) + const second = interpolate(concurrency.group, pushEvent('2222')) + expect(first).not.toBe(second) + // A superseded pull request run is still cancelled. + const prA = pullRequestEvent('3333') + const prB = pullRequestEvent('4444') + expect(interpolate(concurrency.group, prA)).toBe( + interpolate(concurrency.group, prB), + ) + expect(String(interpolate(concurrency['cancel-in-progress'], prA))).toBe( + 'true', + ) + }) + + it('installs a fully pinned, hash-checked analysis toolchain', () => { + // F5: two top-level pins left PyYAML, Pygments and pathspec floating, with + // no hashes and no Dependabot coverage. + const requirements = readFileSync( + join(REPO_ROOT, '.github/impact-gate/requirements.txt'), + 'utf8', + ) + const entries = requirements + .replace(/\\\n/g, ' ') + .split('\n') + .map((line) => line.replace(/#.*/, '').trim()) + .filter(Boolean) + const names = entries.map((entry) => + entry + .split(/[=<>!~ ;[]/)[0] + .toLowerCase() + .replace(/_/g, '-'), + ) + expect(names).toEqual( + expect.arrayContaining([ + 'impact-gate', + 'lizard', + 'pathspec', + 'pygments', + 'pyyaml', + ]), + ) + for (const entry of entries) { + expect(entry, entry).toMatch(/^[\w.-]+==[\w.]+(\s|;)/) + expect(entry, entry).toMatch(/--hash=sha256:[0-9a-f]{64}/) + } + const install = readWorkflow(workflowPath).jobs.impact.steps.find((step) => + step.run?.includes('.github/impact-gate/requirements.txt'), + ) + expect(install.run).toContain('--require-hashes') + expect(install.run).toContain('--no-deps') + + const pip = readWorkflow('.github/dependabot.yml').updates.find( + (update) => + update['package-ecosystem'] === 'pip' && + update.directory === '/.github/impact-gate', + ) + expect(pip).toBeDefined() + expect(pip.cooldown).toEqual({ 'default-days': 7 }) + expect(pip.ignore).toContainEqual({ + 'dependency-name': '*', + 'update-types': ['version-update:semver-major'], + }) + }) + it('runs the real-CLI checks and reports PRs without masking tool errors', () => { const job = readWorkflow(workflowPath).jobs.impact const checks = job.steps.find((step) => From 587d06eb95060fa4c251f554b3559e461f6fc12d Mon Sep 17 00:00:00 2001 From: Toby Hede Date: Fri, 9 Oct 2026 11:57:23 +1100 Subject: [PATCH 5/5] build(ci): pin and hash the ImpactGate toolchain, track it with Dependabot --- .changeset/impact-gate-pip-supply-chain.md | 5 + .github/dependabot.yml | 36 +++++++ .github/impact-gate/requirements.txt | 105 +++++++++++++++++++- e2e/tests/supply-chain.e2e.test.ts | 3 + skills/stash-supply-chain-security/SKILL.md | 8 +- 5 files changed, 154 insertions(+), 3 deletions(-) create mode 100644 .changeset/impact-gate-pip-supply-chain.md diff --git a/.changeset/impact-gate-pip-supply-chain.md b/.changeset/impact-gate-pip-supply-chain.md new file mode 100644 index 000000000..d526908c0 --- /dev/null +++ b/.changeset/impact-gate-pip-supply-chain.md @@ -0,0 +1,5 @@ +--- +'stash': patch +--- + +The supply-chain skill names `pip` among the Dependabot ecosystems the repository monitors, and says that a pip requirements file installed in CI should pin every package with a hash and be installed with `pip install --require-hashes --no-deps`. diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 6b4c31cb3..c64213fce 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -269,6 +269,42 @@ updates: update-types: - version-update:semver-major + # ── pip (.github/impact-gate — the advisory ImpactGate toolchain) ── + # impact.yml installs this hashed requirements.txt with `--require-hashes + # --no-deps`, so every transitive package is a pin here and nothing resolves + # in CI. Dependabot regenerates the hashes when it bumps one. Any change here + # also changes the baseline cache key, so a bump costs one cold rebuild. + - package-ecosystem: pip + directory: /.github/impact-gate + # Monthly, matching the other non-npm toolchains: the tools only feed an + # advisory report, and security fixes are driven by alerts, not `schedule`. + # No `day:`, for the reason recorded on the cargo entries. + schedule: + interval: monthly + cooldown: + default-days: 7 + open-pull-requests-limit: 3 + labels: + - dependencies + - supply-chain + commit-message: + prefix: "chore" + include: scope + groups: + # impact-gate constrains lizard (it excludes 1.24.0), so the set moves + # together in one PR rather than as conflicting per-package bumps. + impact-gate-minor-patch: + patterns: + - "*" + update-types: + - minor + - patch + ignore: + # Major bumps are reviewed and applied manually, not by Dependabot. + - dependency-name: "*" + update-types: + - version-update:semver-major + # ── GitHub Actions ───────────────────────────────────────────── - package-ecosystem: github-actions directory: / diff --git a/.github/impact-gate/requirements.txt b/.github/impact-gate/requirements.txt index c63a063e7..ccf657c97 100644 --- a/.github/impact-gate/requirements.txt +++ b/.github/impact-gate/requirements.txt @@ -1,2 +1,103 @@ -impact-gate==0.4.1 -lizard==1.23.0 +# Advisory ImpactGate toolchain: every package pinned, every distribution hashed. +# CI installs it with `pip install --require-hashes --no-deps`, so nothing here +# resolves at install time and a transitive dependency cannot float. +# +# The top-level pins are impact-gate==0.4.1 and lizard==1.23.0 (Lizard 1.24.0 is +# excluded by ImpactGate's own metadata). To regenerate after changing them, from +# the repository root: +# +# printf 'impact-gate==0.4.1\nlizard==1.23.0\n' \ +# | uv pip compile - --generate-hashes --universal --python-version 3.12 \ +# --no-header -o .github/impact-gate/requirements.txt +# +# then restore this header. Any change here changes the baseline cache key. +impact-gate==0.4.1 \ + --hash=sha256:9d9c8fb4c6f2fa56559d0530ff82569bc6d42b28ff5a77103e504e2470f198c6 \ + --hash=sha256:e6dab6abec925151bd4dbffff92d4a40f742d561a6f8527b3fbe15dd68b25518 +lizard==1.23.0 \ + --hash=sha256:e9111e35c8a5f2e00d55cab318fca3504622991417411ea16cf46874fb752f42 \ + --hash=sha256:ed75cd45f086a2f51d6be64b0149b71bda820f92f95e30898254528bb949f795 + # via impact-gate +pathspec==1.1.1 \ + --hash=sha256:17db5ecd524104a120e173814c90367a96a98d07c45b2e10c2f3919fff91bf5a \ + --hash=sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189 + # via lizard +pygments==2.21.0 \ + --hash=sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9 \ + --hash=sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c + # via lizard +pyyaml==6.0.3 \ + --hash=sha256:00c4bdeba853cc34e7dd471f16b4114f4162dc03e6b7afcc2128711f0eca823c \ + --hash=sha256:0150219816b6a1fa26fb4699fb7daa9caf09eb1999f3b70fb6e786805e80375a \ + --hash=sha256:02893d100e99e03eda1c8fd5c441d8c60103fd175728e23e431db1b589cf5ab3 \ + --hash=sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956 \ + --hash=sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6 \ + --hash=sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c \ + --hash=sha256:16249ee61e95f858e83976573de0f5b2893b3677ba71c9dd36b9cf8be9ac6d65 \ + --hash=sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a \ + --hash=sha256:1ebe39cb5fc479422b83de611d14e2c0d3bb2a18bbcb01f229ab3cfbd8fee7a0 \ + --hash=sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b \ + --hash=sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1 \ + --hash=sha256:22ba7cfcad58ef3ecddc7ed1db3409af68d023b7f940da23c6c2a1890976eda6 \ + --hash=sha256:27c0abcb4a5dac13684a37f76e701e054692a9b2d3064b70f5e4eb54810553d7 \ + --hash=sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e \ + --hash=sha256:2e71d11abed7344e42a8849600193d15b6def118602c4c176f748e4583246007 \ + --hash=sha256:34d5fcd24b8445fadc33f9cf348c1047101756fd760b4dacb5c3e99755703310 \ + --hash=sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4 \ + --hash=sha256:3c5677e12444c15717b902a5798264fa7909e41153cdf9ef7ad571b704a63dd9 \ + --hash=sha256:3ff07ec89bae51176c0549bc4c63aa6202991da2d9a6129d7aef7f1407d3f295 \ + --hash=sha256:41715c910c881bc081f1e8872880d3c650acf13dfa8214bad49ed4cede7c34ea \ + --hash=sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0 \ + --hash=sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e \ + --hash=sha256:4a2e8cebe2ff6ab7d1050ecd59c25d4c8bd7e6f400f5f82b96557ac0abafd0ac \ + --hash=sha256:4ad1906908f2f5ae4e5a8ddfce73c320c2a1429ec52eafd27138b7f1cbe341c9 \ + --hash=sha256:501a031947e3a9025ed4405a168e6ef5ae3126c59f90ce0cd6f2bfc477be31b7 \ + --hash=sha256:5190d403f121660ce8d1d2c1bb2ef1bd05b5f68533fc5c2ea899bd15f4399b35 \ + --hash=sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb \ + --hash=sha256:5cf4e27da7e3fbed4d6c3d8e797387aaad68102272f8f9752883bc32d61cb87b \ + --hash=sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69 \ + --hash=sha256:5ed875a24292240029e4483f9d4a4b8a1ae08843b9c54f43fcc11e404532a8a5 \ + --hash=sha256:5fcd34e47f6e0b794d17de1b4ff496c00986e1c83f7ab2fb8fcfe9616ff7477b \ + --hash=sha256:5fdec68f91a0c6739b380c83b951e2c72ac0197ace422360e6d5a959d8d97b2c \ + --hash=sha256:6344df0d5755a2c9a276d4473ae6b90647e216ab4757f8426893b5dd2ac3f369 \ + --hash=sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd \ + --hash=sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824 \ + --hash=sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198 \ + --hash=sha256:66e1674c3ef6f541c35191caae2d429b967b99e02040f5ba928632d9a7f0f065 \ + --hash=sha256:6adc77889b628398debc7b65c073bcb99c4a0237b248cacaf3fe8a557563ef6c \ + --hash=sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c \ + --hash=sha256:7c6610def4f163542a622a73fb39f534f8c101d690126992300bf3207eab9764 \ + --hash=sha256:7f047e29dcae44602496db43be01ad42fc6f1cc0d8cd6c83d342306c32270196 \ + --hash=sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b \ + --hash=sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00 \ + --hash=sha256:8d1fab6bb153a416f9aeb4b8763bc0f22a5586065f86f7664fc23339fc1c1fac \ + --hash=sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8 \ + --hash=sha256:8dc52c23056b9ddd46818a57b78404882310fb473d63f17b07d5c40421e47f8e \ + --hash=sha256:9149cad251584d5fb4981be1ecde53a1ca46c891a79788c0df828d2f166bda28 \ + --hash=sha256:93dda82c9c22deb0a405ea4dc5f2d0cda384168e466364dec6255b293923b2f3 \ + --hash=sha256:96b533f0e99f6579b3d4d4995707cf36df9100d67e0c8303a0c55b27b5f99bc5 \ + --hash=sha256:9c57bb8c96f6d1808c030b1687b9b5fb476abaa47f0db9c0101f5e9f394e97f4 \ + --hash=sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b \ + --hash=sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf \ + --hash=sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5 \ + --hash=sha256:a80cb027f6b349846a3bf6d73b5e95e782175e52f22108cfa17876aaeff93702 \ + --hash=sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8 \ + --hash=sha256:b3bc83488de33889877a0f2543ade9f70c67d66d9ebb4ac959502e12de895788 \ + --hash=sha256:b865addae83924361678b652338317d1bd7e79b1f4596f96b96c77a5a34b34da \ + --hash=sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d \ + --hash=sha256:ba1cc08a7ccde2d2ec775841541641e4548226580ab850948cbfda66a1befcdc \ + --hash=sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c \ + --hash=sha256:c1ff362665ae507275af2853520967820d9124984e0f7466736aea23d8611fba \ + --hash=sha256:c2514fceb77bc5e7a2f7adfaa1feb2fb311607c9cb518dbc378688ec73d8292f \ + --hash=sha256:c3355370a2c156cffb25e876646f149d5d68f5e0a3ce86a5084dd0b64a994917 \ + --hash=sha256:c458b6d084f9b935061bc36216e8a69a7e293a2f1e68bf956dcd9e6cbcd143f5 \ + --hash=sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26 \ + --hash=sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f \ + --hash=sha256:ebc55a14a21cb14062aa4162f906cd962b28e2e9ea38f9b4391244cd8de4ae0b \ + --hash=sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be \ + --hash=sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c \ + --hash=sha256:efd7b85f94a6f21e4932043973a7ba2613b059c4a000551892ac9f1d11f5baf3 \ + --hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \ + --hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \ + --hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0 + # via impact-gate diff --git a/e2e/tests/supply-chain.e2e.test.ts b/e2e/tests/supply-chain.e2e.test.ts index ad8e4292e..e4c7cc607 100644 --- a/e2e/tests/supply-chain.e2e.test.ts +++ b/e2e/tests/supply-chain.e2e.test.ts @@ -622,6 +622,9 @@ const MANIFEST_BY_ECOSYSTEM: Record = { gomod: 'go.mod', bundler: 'Gemfile', composer: 'composer.json', + // A pip `directory` holds a requirements file rather than a lockfile, so the + // lockfile scan above never demands this entry; it only checks it is aimed. + pip: 'requirements.txt', uv: 'pyproject.toml', mix: 'mix.exs', pub: 'pubspec.yaml', diff --git a/skills/stash-supply-chain-security/SKILL.md b/skills/stash-supply-chain-security/SKILL.md index 56ebcf476..a5f63d371 100644 --- a/skills/stash-supply-chain-security/SKILL.md +++ b/skills/stash-supply-chain-security/SKILL.md @@ -57,9 +57,15 @@ that publishes to npm — ran a bare `pnpm install` the whole time. The single install allowed to resolve outside the lockfile was the one whose output goes to the registry. +A pip requirements file installed in CI is held to the same bar: pin every +package, transitive ones included, with `==` and a `--hash`, and install it with +`pip install --require-hashes --no-deps -r `, so nothing resolves at +install time. `uv pip compile --generate-hashes` produces such a file. (This one +is checked by the workflow's own test, not by `supply-chain.e2e.test.ts`.) + ### 5. Cooldown'd auto-updates — practice #6 -Dependabot opens grouped, cooldown'd PRs (7 days minor/patch) for `npm`, `cargo`, `gomod`, `github-actions` and `docker` (the digest of the Alpine image the musl binaries are built in). Major bumps are not proposed at all — every entry ignores `version-update:semver-major`, so majors are reviewed and applied by hand. +Dependabot opens grouped, cooldown'd PRs (7 days minor/patch) for `npm`, `cargo`, `gomod`, `pip` (the hashed requirements file for a Python tool CI runs), `github-actions` and `docker` (the digest of the Alpine image the musl binaries are built in). Major bumps are not proposed at all — every entry ignores `version-update:semver-major`, so majors are reviewed and applied by hand. There is deliberately **no `semver-major-days` cooldown** on any entry. It would delay major *version update* PRs, which the `ignore` above means Dependabot never opens, and cooldown does not reach the security path either ("the cooldown option is only available for version updates, not security updates"). Don't add one back as a safety net for the day the `ignore` is dropped — dead config reads as policy, and the test below fails on the pair.