Skip to content

Commit 0fcb28d

Browse files
committed
Show where a capture differs from its golden
The diff a failing run wrote was a raw absolute difference: near-black wherever the change was subtle, which is most of the ones worth looking at. Mark the counted pixels magenta over a dimmed copy of the capture instead, and dim the ignored status strip further so telemetry is never mistaken for a change. Add `just goldens-diff <case>`, which reports every shot rather than only the failures. A difference inside its tolerance is invisible during a run, and that is exactly what you want to see before deciding whether a golden has gone stale. It reads each shot's tolerance and ignored strip from the run's events, which now record both, and says so plainly when asked about an older run that does not.
1 parent 9890f9a commit 0fcb28d

4 files changed

Lines changed: 145 additions & 4 deletions

File tree

‎justfile‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -165,6 +165,12 @@ e2e-report *args:
165165
goldens-status:
166166
@PYTHONPATH="{{tool_pythonpath}}" uv run --locked sbc-e2e goldens-status
167167

168+
# Where a case's last captures differ from its goldens, marked in magenta.
169+
# Reports the within-tolerance differences a passing run never shows.
170+
[group('test')]
171+
goldens-diff case *args:
172+
@PYTHONPATH="{{tool_pythonpath}}" uv run --locked sbc-e2e goldens-diff "{{case}}" {{args}}
173+
168174
# Approve a case's reference images (the human OK): `just goldens-approve rotation`.
169175
# Optionally name individual shots. Only a human runs this.
170176
[group('test')]

‎tools/e2e/cli_e2e.py‎

Lines changed: 92 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
1+
import json
12
from collections.abc import Iterator
23
from dataclasses import dataclass
34
from pathlib import Path
@@ -7,7 +8,15 @@
78

89
from .driver.cases import TARGETS, Case, select_cases, target_cases
910
from .driver.utils.paths import ARTIFACT_ROOT, create_suite_artifact_dir
10-
from .fixtures.golden import GOLDEN_ROOT, STATUS_APPROVED, approve_review, load_review
11+
from .fixtures.golden import (
12+
GOLDEN_ROOT,
13+
STATUS_APPROVED,
14+
approve_review,
15+
differing_pixels,
16+
golden_path,
17+
load_review,
18+
write_diff,
19+
)
1120
from .runner import E2ERun
1221
from .suite_report import artifact_paths, default_report_path, write_report
1322

@@ -121,6 +130,63 @@ def goldens_status() -> None:
121130
typer.echo(f"\n{pending} image(s) awaiting approval.")
122131

123132

133+
@app.command("goldens-diff")
134+
def goldens_diff(
135+
case: Annotated[str, typer.Argument(help="Golden case directory.")],
136+
run: Annotated[Path | None, typer.Option(help="Run directory; default is the latest for the case.")] = None,
137+
out: Annotated[Path | None, typer.Option(help="Where to write the overlays.")] = None,
138+
) -> None:
139+
"""Show where a case's captures differ from its goldens, as magenta overlays.
140+
141+
Reports every shot, including the ones whose difference stays inside their
142+
tolerance -- those are invisible in a run, and they are exactly what you want
143+
to see before deciding whether a golden is stale.
144+
"""
145+
run_dir = run or _latest_run_dir(case)
146+
if run_dir is None:
147+
raise typer.BadParameter(f"no run artifacts for {case}; run `just test-e2e {case}` first")
148+
checks = _golden_checks(run_dir)
149+
if not checks:
150+
raise typer.BadParameter(f"{run_dir} recorded no goldens")
151+
destination = out or run_dir / "golden-diffs"
152+
destination.mkdir(parents=True, exist_ok=True)
153+
if any("ignored_bottom" not in check for check in checks.values()):
154+
typer.echo(
155+
f"warning: {run_dir.name} predates recorded golden metadata, so the "
156+
"ignored status strip and each shot's tolerance are unknown here; "
157+
"counts include regions the run itself does not read. Re-run the case."
158+
)
159+
for name, check in sorted(checks.items()):
160+
actual = Path(check["path"])
161+
golden = golden_path(case, name)
162+
if not actual.is_file() or not golden.is_file():
163+
typer.echo(f"{name:32} missing capture or golden")
164+
continue
165+
ignored_bottom = int(check.get("ignored_bottom", 0))
166+
channel_tolerance = int(check.get("channel_tolerance", 1))
167+
tolerance = int(check.get("tolerance", 0))
168+
differing = differing_pixels(
169+
golden,
170+
actual,
171+
channel_tolerance=channel_tolerance,
172+
ignored_bottom=ignored_bottom,
173+
)
174+
if not differing:
175+
typer.echo(f"{name:32} identical")
176+
continue
177+
target = destination / f"{name}.diff.png"
178+
write_diff(
179+
golden,
180+
actual,
181+
target,
182+
channel_tolerance=channel_tolerance,
183+
ignored_bottom=ignored_bottom,
184+
)
185+
verdict = "over tolerance" if differing > tolerance else f"within tolerance {tolerance}"
186+
typer.echo(f"{name:32} {differing:>8} px {verdict:<22} {target}")
187+
typer.echo(f"\ndiffs: {destination}")
188+
189+
124190
@app.command("approve-goldens")
125191
def approve_goldens(
126192
case: Annotated[str, typer.Argument(help="Golden case directory.")],
@@ -256,3 +322,28 @@ def _case_batches(cases: list[Case], reuse_sessions: bool) -> Iterator[list[Case
256322

257323
def _batch_key(case: Case) -> tuple[tuple[tuple[str, str], ...], tuple[tuple[str, str], ...]]:
258324
return tuple(sorted(case.flags.items())), tuple(sorted(case.env.items()))
325+
326+
327+
def _latest_run_dir(case: str) -> Path | None:
328+
"""The newest artifact directory holding this case, standalone or in a suite."""
329+
if not ARTIFACT_ROOT.is_dir():
330+
return None
331+
candidates = [path for path in ARTIFACT_ROOT.glob(f"*-{case}") if path.is_dir()]
332+
candidates += [path / case for path in ARTIFACT_ROOT.glob("*-suite") if (path / case).is_dir()]
333+
return max(candidates, key=lambda path: path.stat().st_mtime, default=None)
334+
335+
336+
def _golden_checks(run_dir: Path) -> dict[str, dict[str, object]]:
337+
"""The `golden` events of a run, newest entry per shot."""
338+
events = run_dir / "events.jsonl"
339+
if not events.is_file():
340+
return {}
341+
checks: dict[str, dict[str, object]] = {}
342+
for line in events.read_text().splitlines():
343+
try:
344+
event = cast("dict[str, object]", json.loads(line))
345+
except json.JSONDecodeError:
346+
continue
347+
if event.get("kind") == "golden" and isinstance(event.get("name"), str):
348+
checks[cast("str", event["name"])] = event
349+
return checks

‎tools/e2e/driver/_capture.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -195,7 +195,9 @@ def _finish_screenshots(self) -> list[str]:
195195
status=status,
196196
path=str(check.path),
197197
capture_ms=check.capture_ms,
198+
tolerance=check.tolerance,
198199
channel_tolerance=check.channel_tolerance,
200+
ignored_bottom=check.ignored_bottom,
199201
conversion_ms=conversion.elapsed_ms,
200202
compare_ms=int((monotonic() - started) * 1000),
201203
)

‎tools/e2e/fixtures/golden.py‎

Lines changed: 45 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,10 @@
2828
STATUS_AI = "ai-reviewed"
2929
STATUS_APPROVED = "approved"
3030

31+
# Magenta: the UI theme has no such colour, so a marked pixel cannot be mistaken
32+
# for one of its own.
33+
DIFF_COLOR = (255, 0, 255)
34+
3135

3236
class ReviewEntry(TypedDict):
3337
status: str
@@ -115,9 +119,38 @@ def differing_pixels(
115119
return _count_pixels_over(difference, channel_tolerance)
116120

117121

118-
def write_diff(golden: Path, actual: Path, out: Path) -> None:
122+
def write_diff(
123+
golden: Path,
124+
actual: Path,
125+
out: Path,
126+
*,
127+
channel_tolerance: int = 1,
128+
ignored_bottom: int = 0,
129+
) -> None:
130+
"""Paint the changed pixels magenta over a dimmed copy of the capture.
131+
132+
A raw absolute difference is nearly black wherever the change is subtle,
133+
which is most of the interesting ones. Locating the change matters more than
134+
measuring it here, so the base image stays recognisable and every pixel the
135+
comparison counts is made the one colour nothing in the UI uses.
136+
137+
Rows inside `ignored_bottom` are dimmed further and never marked: they carry
138+
live telemetry that no check reads.
139+
"""
119140
with Image.open(golden) as expected, Image.open(actual) as observed:
120-
ImageChops.difference(expected.convert("RGBA"), observed.convert("RGBA")).save(out)
141+
base = observed.convert("RGB")
142+
difference = ImageChops.difference(expected.convert("RGBA"), base.convert("RGBA"))
143+
changed = _over_tolerance_mask(difference, channel_tolerance)
144+
if ignored_bottom:
145+
keep = max(0, base.height - ignored_bottom)
146+
changed.paste(0, (0, keep, base.width, base.height))
147+
canvas = Image.blend(base, Image.new("RGB", base.size, (0, 0, 0)), 0.65)
148+
if ignored_bottom:
149+
keep = max(0, base.height - ignored_bottom)
150+
strip = canvas.crop((0, keep, base.width, base.height))
151+
canvas.paste(Image.blend(strip, Image.new("RGB", strip.size, (0, 0, 0)), 0.6), (0, keep))
152+
canvas.paste(Image.new("RGB", base.size, DIFF_COLOR), mask=changed)
153+
canvas.save(out)
121154

122155

123156
def compare(
@@ -182,6 +215,11 @@ def review_path(case_name: str) -> Path:
182215
return GOLDEN_ROOT / case_name / "review.json"
183216

184217

218+
def _over_tolerance_mask(image: Image.Image, threshold: int) -> Image.Image:
219+
"""A 1-bit mask of the pixels `_count_pixels_over` counts."""
220+
return _channel_maximum(image).point(lambda value: 255 if value > threshold else 0).convert("1")
221+
222+
185223
def _count_pixels_over(image: Image.Image, threshold: int) -> int:
186224
"""Count pixels whose largest channel exceeds ``threshold``.
187225
@@ -190,8 +228,12 @@ def _count_pixels_over(image: Image.Image, threshold: int) -> int:
190228
`max(channel) > threshold` predicate while making large golden comparisons
191229
cheap enough to keep the suite's visual coverage.
192230
"""
231+
return sum(_channel_maximum(image).histogram()[threshold + 1 :])
232+
233+
234+
def _channel_maximum(image: Image.Image) -> Image.Image:
193235
channels = image.split()
194236
maximum = channels[0]
195237
for channel in channels[1:]:
196238
maximum = ImageChops.lighter(maximum, channel)
197-
return sum(maximum.histogram()[threshold + 1 :])
239+
return maximum

0 commit comments

Comments
 (0)