From 86f1c960ef0f38de361056b7abedddc3c80894dc Mon Sep 17 00:00:00 2001 From: Pengfei Hu Date: Thu, 24 Sep 2026 17:02:19 -0700 Subject: [PATCH 1/3] feat: compare application agent wiring without prior setup --- README.md | 14 + docs/application-comparison.md | 68 +++ src/agents_shipgate/cli/application_diff.py | 401 ++++++++++++++++++ src/agents_shipgate/cli/diff.py | 30 +- src/agents_shipgate/core/agent_bindings.py | 18 +- src/agents_shipgate/inputs/google_adk.py | 11 +- .../inputs/openai_sdk_static.py | 4 +- tests/test_application_diff.py | 310 ++++++++++++++ 8 files changed, 838 insertions(+), 18 deletions(-) create mode 100644 docs/application-comparison.md create mode 100644 src/agents_shipgate/cli/application_diff.py create mode 100644 tests/test_application_diff.py diff --git a/README.md b/README.md index b4a71bb4d..258536b3d 100644 --- a/README.md +++ b/README.md @@ -28,6 +28,20 @@ config, Codex plugin, n8n, and Conductor OSS workflow artifacts, then writes a deterministic **Tool-Use Readiness Report** before your agent gets production-like permissions. +## Application agent PRs without setup (unreleased) + +The source build can compare OpenAI Agents SDK and Google ADK tool bindings +without a manifest or baseline: + +```bash +agents-shipgate diff --application --workspace /path/to/repo --base BASE_SHA --head HEAD_SHA +``` + +It shows source-observed changes per agent, including before/after signatures +and source locations. See [application comparison](docs/application-comparison.md) +for scoped applications, moves, exact refs and coverage limits. This option is +not yet in the published 1.1.0 package. + ## What did this PR change? Start with a pull request you already have that changes what a coding agent diff --git a/docs/application-comparison.md b/docs/application-comparison.md new file mode 100644 index 000000000..b910965a3 --- /dev/null +++ b/docs/application-comparison.md @@ -0,0 +1,68 @@ +# Application comparison without prior setup + +**Unreleased source feature.** The published 1.1.0 package does not yet have +`diff --application`. Run the source checkout's `./shipgate`, or a build +containing this feature. + +For an OpenAI Agents SDK or Google ADK application, compare committed PR refs: + +```bash +agents-shipgate diff --application --workspace /path/to/repository \ + --base BASE_SHA --head HEAD_SHA --scope backend +``` + +No `init`, manifest, saved baseline or authored declaration is required. The +command writes no files in the subject repository and runs no application code. +Fetch the refs first; the command never fetches. `--scope` defaults to the +repository root; `--head` defaults to committed `HEAD`, excluding dirty edits. +The base is the requested base ref's merge base with the selected head. + +The result identifies each observed agent object and its added, removed or +changed tool bindings. It shows before/after signatures, binding and function +locations, and implementation digests. Existing reader observations supply the +binding edges; unrelated tool definitions are not agent capabilities. A missing +deployment-root selection does not hide known per-agent edges. + +```text +ADDED synthesizer_agent → python_exec + before: no observed binding + after: python_exec(code, dataset_json) -> str at backend/app/agents/synthesizer.py:38 +``` + +A reviewer can inspect the new callable and decide whether that agent should +receive it. A body change is a request to review the implementation; it does not +establish widening, narrowing, business impact or runtime behavior. + +## Scopes, moves and incomplete inputs + +For an application directory moved by the PR, select its old location separately: + +```bash +agents-shipgate diff --application --base BASE_SHA --head HEAD_SHA \ + --base-scope old/application --scope new/application --json +``` + +The two scopes are discovered independently. Exact Git blob renames supply file +correspondence, preserving original evidence locations; file identity is not a +claim about deployed agent identity. An absent directory is absence at that +Git path, not proof that no agent exists anywhere else. Missing manifests do not +supply an empty base. + +`--json` emits `application_comparison_schema_version: "0.1"`, engine identity, +requested and compared refs/tree IDs, per-side scope/coverage, rows, source +correspondence, and a deterministic `comparison_id`. This is a separate advisory +artifact from the existing host diff JSON and verifier receipt. + +- `compared`: the selected supported source observations were compared. +- `partial`: a parse/discovery/binding gap remains. Proven positive observations + may be shown; unread input cannot establish absence or removal. +- `not_established`: neither side established a supported application agent. +- Exit 2: refs/materialization/input could not be read. This is not no change. + +Discovery is bounded by `--max-python-files` (default 1000) and a 2 MB per-Python +file limit. Partial discovery remains visible. The first increment uses existing +SDK/ADK readers; unresolved imports, dynamic factories and built-ins remain +explicit reader limitations. It does not support other application frameworks +yet. Indirect helper effects, runtime loading, deployed reachability and business +authority are outside this comparison. It grants no release or merge permission +and cannot stand in for a reviewed verifier base or qualification evidence. diff --git a/src/agents_shipgate/cli/application_diff.py b/src/agents_shipgate/cli/application_diff.py new file mode 100644 index 000000000..7bab3c6e2 --- /dev/null +++ b/src/agents_shipgate/cli/application_diff.py @@ -0,0 +1,401 @@ +"""Manifest-free comparison of source-observed application agent wiring. + +This entry reads immutable trees and the existing SDK/ADK adapters. It never +constructs a release manifest, supplies declarations, or publishes a verdict. +""" +from __future__ import annotations + +import ast +import hashlib +import json +import tempfile +from dataclasses import dataclass, field +from pathlib import Path, PurePosixPath +from typing import Any + +import typer + +from agents_shipgate.cli.discovery import detect_workspace +from agents_shipgate.cli.discovery.artifacts import _candidate_files +from agents_shipgate.cli.scan.source_loading import _build_canonical_tools +from agents_shipgate.cli.verify.git import archive_tree, commit_sha, ensure_git_workspace, tree_sha +from agents_shipgate.core.agent_bindings import resolve_agent_binding_graph +from agents_shipgate.core.artifacts import ArtifactBag +from agents_shipgate.core.errors import ConfigError +from agents_shipgate.core.privacy import sanitize_report_payload +from agents_shipgate.core.verification_identity import build_engine_requirement +from agents_shipgate.inputs.google_adk import load_google_adk_artifacts +from agents_shipgate.inputs.openai_sdk_static import load_openai_sdk_static_tools +from agents_shipgate.schemas.manifest import ToolSourceConfig + +SUPPORTED = frozenset({"openai_agents_sdk", "google_adk"}) +SCHEMA_VERSION = "0.1" +MAX_PYTHON_BYTES = 2_000_000 + + +def _digest(value: Any) -> str: + return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + + +def _scope(value: str) -> str: + path = PurePosixPath(value) + if not value or path.is_absolute() or ".." in path.parts or "\\" in value or ":" in value: + raise typer.BadParameter("Scope must be a repository-relative directory without '..'.") + return path.as_posix() + + +@dataclass +class Observations: + scope: str + status: str = "complete" + agents: dict[tuple[str, str], dict[str, Any]] = field(default_factory=dict) + bindings: dict[tuple[str, str, str], dict[str, Any]] = field(default_factory=dict) + limits: list[str] = field(default_factory=list) + sources: list[dict[str, str]] = field(default_factory=list) + + def summary(self) -> dict[str, Any]: + return {"scope": self.scope, "status": self.status, + "sources": self.sources, "agents": list(self.agents.values()), + "binding_count": len(self.bindings), "limits": sorted(set(self.limits))} + + + +def _source_path(root: Path, ref: str) -> str: + # ADK handoff observations can carry "path.py:line" as their source. + # The line locates evidence; it must never become callable identity. + if (root / ref).is_file(): + return ref + path, separator, line = ref.rpartition(":") + if separator and line.isdecimal() and (root / path).is_file(): + return path + return ref + +def _definition(root: Path, tool: Any) -> dict[str, Any]: + """Read a unique function's AST; line movement/comments are not changes.""" + path = _source_path(root, tool.source_ref or "") + symbol = (tool.native_locator or "").split("#")[-1] or tool.name + if "#" not in (tool.native_locator or ""): + symbol = (tool.function_signature or tool.name).split("(")[0] + file = root / path + if not file.is_file() or not file.resolve().is_relative_to(root.resolve()): + return {"source": path, "line": None, "implementation_sha256": None} + try: + nodes = [n for n in ast.walk(ast.parse(file.read_bytes())) + if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef)) and n.name == symbol] + except (SyntaxError, ValueError, RecursionError): + nodes = [] + if len(nodes) != 1: + return {"source": path, "line": None, "implementation_sha256": None} + node = nodes[0] + # Docstrings are description evidence, not an implementation change. + body = node.body + if body and isinstance(body[0], ast.Expr) and isinstance(body[0].value, ast.Constant) and isinstance(body[0].value.value, str): + body = body[1:] + node.body = body + # Include defaults and decorators: approval decorators and changed default + # bounds are review-relevant even when the executable body is unchanged. + return {"source": path, "line": node.lineno, + "implementation_sha256": _digest(ast.dump(node, include_attributes=False))} + + +def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: + result = Observations(scope) + root = tree / scope + if not root.exists(): + result.status = "absent" + return result + if not root.is_dir() or root.is_symlink(): + raise ConfigError(f"Application scope is not a regular directory: {scope}") + root = root.resolve() + if not root.is_relative_to(tree.resolve()): + raise ConfigError(f"Application scope escapes its tree: {scope}") + python_files = [p for p in _candidate_files(root) if p.suffix == ".py"] + for file in python_files: + if file.stat().st_size > MAX_PYTHON_BYTES: + result.limits.append(f"Python input exceeds {MAX_PYTHON_BYTES} bytes: {file.relative_to(root)}") + if result.limits: + result.status = "partial" + return result + detected = detect_workspace(root, max_python_files=max_python_files) + if detected.python_parse_truncated: + result.limits.append(f"Python discovery truncated at {max_python_files} files.") + result.limits.extend(f"Discovery could not read {p}." for p in detected.host_discovery_incomplete_paths) + result.limits.extend(f"Excluded candidate: {item}" for item in detected.excluded_sources) + # Discovery omits malformed Python; preserve that gap rather than an empty + # candidate list becoming negative evidence. Bound this second parse too. + if len(python_files) > max_python_files: + result.limits.append(f"Python input census exceeds {max_python_files} files.") + for file in python_files[:max_python_files]: + if file.is_symlink(): + result.limits.append(f"Linked Python input: {file.relative_to(root)}") + continue + try: + ast.parse(file.read_bytes()) + except (SyntaxError, ValueError, RecursionError, OSError): + result.limits.append(f"Python input could not be parsed: {file.relative_to(root)}") + for framework in detected.frameworks: + if framework.type not in SUPPORTED and framework.candidate_files: + result.limits.append(f"Application comparison does not yet support {framework.type}: {', '.join(framework.candidate_files)}") + entries = sorted({(f.type, p) for f in detected.frameworks if f.type in SUPPORTED + for p in f.candidate_files}) + sources = [ToolSourceConfig(id=f"{kind}:{path}", type=kind, path=path) for kind, path in entries] + result.sources = [{"type": s.type, "path": s.path} for s in sources] + bag = ArtifactBag() + loaded = [] + for source in sources: + if source.type == "openai_agents_sdk": + loaded.append(load_openai_sdk_static_tools(source, None, root)) + adk, artifacts = load_google_adk_artifacts(None, root, sources=sources) + loaded.extend(adk) + if artifacts is not None: + bag.set("google_adk", artifacts) + result.limits.extend(artifacts.warnings) + result.limits.extend(w for item in loaded for w in item.warnings) + result.limits.extend(f"Omitted source surface: {o}" for item in loaded for o in item.omissions) + tools, warnings = _build_canonical_tools(loaded) + result.limits.extend(warnings) + graph, _ = resolve_agent_binding_graph(None, tools, bag, loaded) + # Deployment-root selection is irrelevant to observed per-agent edges. + # A root with an explicit empty tool list can also leave unrelated catalog + # definitions unbound. That is a release-scope concern, not missing observed + # wiring. Dynamic/partial lists have their own retained reader/graph issues. + result.limits.extend(i.message for i in graph.issues if i.kind not in + {"ambiguous_root_agent", "missing_binding_evidence"}) + tool_by_id = {t.id: t for t in tools} + agent_keys = {} + ambiguous_agents = set() + for agent in graph.agents: + key = (_source_path(root, agent.source_ref or ""), agent.name) + agent_keys[agent.agent_id] = key + if key in result.agents: + result.limits.append(f"Ambiguous agent identity: {key}") + ambiguous_agents.add(key) + result.agents[key] = {"name": agent.name, "source": key[0], + "location": agent.source_pointer} + ambiguous_bindings = set() + for edge in graph.tool_edges: + key = agent_keys[edge.agent_id] + if key in ambiguous_agents: + continue + tool = tool_by_id[edge.tool_id] + definition = _definition(root, tool) + if definition["implementation_sha256"] is None: + result.limits.append(f"Implementation location unresolved: {tool.name} ({tool.source_ref}).") + binding_key = (*key, tool.name) + if binding_key in ambiguous_bindings: + continue + if binding_key in result.bindings: + result.limits.append(f"Ambiguous tool identity: {binding_key}") + result.bindings.pop(binding_key) + ambiguous_bindings.add(binding_key) + continue + result.bindings[binding_key] = { + "agent": key[1], "agent_source": key[0], "tool": tool.name, + "binding_location": edge.source_pointer, "edge_type": edge.edge_type, + "definition": definition, "input_schema": tool.input_schema, + "output_schema": tool.output_schema, "signature": tool.function_signature, + "evidence_basis": edge.provenance_kind, + } + for edge in graph.handoff_edges: + source, target = agent_keys[edge.source_agent_id], agent_keys[edge.target_agent_id] + if source in ambiguous_agents or target in ambiguous_agents: + continue + key = (*source, f"handoff:{target[0]}:{target[1]}") + result.bindings[key] = {"agent": source[1], "agent_source": source[0], + "tool": target[1], "edge_type": edge.edge_type, + "target_source": target[0], "binding_location": edge.source_pointer, + "evidence_basis": edge.provenance_kind} + if result.limits: + result.status = "partial" + return result + + +def _meaning(binding: dict[str, Any]) -> dict[str, Any]: + return {k: v for k, v in binding.items() if k not in + {"binding_location", "definition", "evidence_basis", "agent_source"}} | { + "implementation_sha256": binding.get("definition", {}).get("implementation_sha256")} + + +def compare(base: Observations, head: Observations, *, + target_moves: dict[str, str] | None = None) -> list[dict[str, Any]]: + rows = [] + for key in sorted(base.bindings.keys() | head.bindings.keys()): + before, after = base.bindings.get(key), head.bindings.get(key) + if before is not None and after is not None: + before_meaning = _meaning(before) + if "target_source" in before_meaning: + path = before_meaning["target_source"] + before_meaning["target_source"] = (target_moves or {}).get(path, path) + if before_meaning == _meaning(after): + continue + if before is None and base.status == "partial": + head.limits.append(f"Cannot establish whether {key[1]}.{key[2]} was absent at base.") + continue + if after is None and head.status == "partial": + head.limits.append(f"Cannot establish removal of {key[1]}.{key[2]} from incomplete head inputs.") + continue + kind = "added" if before is None else "removed" if after is None else "changed" + rows.append({"agent": key[1], "agent_source": key[0], "tool": key[2], + "change": kind, "before": before, "after": after, + "why": {"added": "The source now binds this callable to this agent.", + "removed": "The source no longer binds this callable to this agent.", + "changed": "The bound callable's interface or implementation changed; authority direction is not established."}[kind], + "review_question": f"Should {key[1]} have this {kind} binding to {key[2]}? Review the before/after signature and implementation locations."}) + return rows + + + +def _align_exact_moves(workspace: Path, base: str, head: str, + old: Observations, new: Observations) -> list[dict[str, str]]: + from agents_shipgate.cli.verify.git import _run_git + + diff = _run_git(workspace, ["diff", "--no-ext-diff", "--no-textconv", "--name-status", + "-z", "-M100%", base, head, "--"]) + if diff.returncode: + raise ConfigError("Could not establish source relocation evidence.") + parts = iter(diff.stdout.split("\0")) + moves = {} + for status in parts: + if not status: + continue + source = next(parts) + if status.startswith(("R", "C")): + target = next(parts) + if status == "R100": + try: + left = PurePosixPath(source).relative_to(old.scope).as_posix() + right = PurePosixPath(target).relative_to(new.scope).as_posix() + except ValueError: + continue + moves[left] = right + # Exact Git blob identity supplies file correspondence only, not deployed + # agent identity. Preserve original locations in each before/after record. + aligned = {} + collisions = set() + for (path, agent, tool), value in old.bindings.items(): + if "target_source" in value: + target = value["target_source"] + tool = f"handoff:{moves.get(target, target)}:{value['tool']}" + key = (moves.get(path, path), agent, tool) + if key in aligned: + collisions.add(key) + aligned[key] = value + for key in collisions: + aligned.pop(key, None) + new.bindings.pop(key, None) + old.limits.append(f"Ambiguous relocated binding: {key}") + old.status = "partial" + old.bindings = aligned + old_names = {name for path, name in old.agents if moves.get(path, path) not in + {new_path for new_path, new_name in new.agents if new_name == name}} + new_names = {name for path, name in new.agents if path not in + {moves.get(old_path, old_path) for old_path, old_name in old.agents if old_name == name}} + for name in sorted(old_names & new_names): + message = f"Agent {name!r} occurs at different unpaired source paths; select --base-scope/--scope or review relocation." + old.limits.append(message) + new.limits.append(message) + old.status = new.status = "partial" + return [{"base_source": a, "head_source": b, "basis": "git_rename_identical_blob"} + for a, b in sorted(moves.items())] + + +def _location(scope: str, path: str | None) -> str | None: + if path is None: + return None + return str(PurePosixPath(scope) / path) + + +def _published_binding(binding: dict[str, Any] | None, scope: str) -> dict[str, Any] | None: + if binding is None: + return None + result = dict(binding) + result["agent_source"] = _location(scope, result["agent_source"]) + result["binding_location"] = _location(scope, result.get("binding_location")) + if "target_source" in result: + result["target_source"] = _location(scope, result["target_source"]) + if "definition" in result: + result["definition"] = dict(result["definition"]) + result["definition"]["source"] = _location(scope, result["definition"]["source"]) + return result + +def run_application_diff(*, workspace: Path, base: str | None, head: str, + scope: str, base_scope: str | None, + max_python_files: int, json_output: bool) -> int: + from agents_shipgate.cli.diff import _one_line, _resolve_base + + workspace = ensure_git_workspace(workspace) + scope, old_scope = _scope(scope), _scope(base_scope if base_scope is not None else scope) + if not head.strip() or head.startswith("-"): + raise typer.BadParameter("Head ref must be non-empty and cannot start with a dash.") + head_commit = commit_sha(workspace, head) + if head_commit is None: + raise typer.BadParameter(f"Head ref {head!r} is unavailable locally. Fetch it first.") + from agents_shipgate.cli.verify.git import detect_default_base + + base_ref = base if base is not None else detect_default_base( + workspace, head_commit, allow_local_when_no_remote=True, allow_equal_head=True) + if base_ref is None: + # Keep the established missing-base diagnostic and recovery wording. + _resolve_base(workspace, None, head_commit) + raise ConfigError("No comparison base is available.") + if not base_ref.strip() or base_ref.startswith("-"): + raise typer.BadParameter("Base ref must be non-empty and cannot start with a dash.") + requested_base_commit = commit_sha(workspace, base_ref) + _, base_commit = _resolve_base(workspace, requested_base_commit or base_ref, head_commit) + engine = build_engine_requirement(plugins_enabled=False).model_dump(mode="json") + with tempfile.TemporaryDirectory(prefix="shipgate-application-diff-") as raw: + scratch = Path(raw) + archive_tree(workspace, base_commit, scratch / "base") + archive_tree(workspace, head_commit, scratch / "head") + old = observe(scratch / "base", old_scope, max_python_files=max_python_files) + new = observe(scratch / "head", scope, max_python_files=max_python_files) + moves = _align_exact_moves(workspace, base_commit, head_commit, old, new) + rows = compare(old, new, target_moves={m["base_source"]: m["head_source"] for m in moves}) + for row in rows: + row["before"] = _published_binding(row["before"], old_scope) + row["after"] = _published_binding(row["after"], scope) + status = "partial" if "partial" in {old.status, new.status} else "compared" + if not old.agents and not new.agents and status == "compared": + status = "not_established" + payload = { + "application_comparison_schema_version": SCHEMA_VERSION, + "comparison_status": status, "static_analysis_only": True, + "comparison_basis": "source_observed_per_agent_wiring", + "input_origin": "independent_tree_discovery", "engine": engine, + "base": {"requested_ref": base_ref, "requested_commit": requested_base_commit, + "compared_commit": base_commit, "tree": tree_sha(workspace, base_commit), **old.summary()}, + "head": {"requested_ref": head, "compared_commit": head_commit, + "tree": tree_sha(workspace, head_commit), **new.summary()}, + "options": {"max_python_files": max_python_files, "max_python_bytes": MAX_PYTHON_BYTES}, + "source_correspondence": moves, "rows": rows, + "limits": ["Covers supported OpenAI Agents SDK and Google ADK source wiring only.", + "Deployment-root reachability, runtime behavior, indirect helper effects and business authority are not established.", + "This comparison is advisory evidence and supplies no release verdict or merge permission."], + } + payload["comparison_id"] = _digest(payload) + payload = sanitize_report_payload(payload) + if json_output: + typer.echo(json.dumps(payload, ensure_ascii=False, indent=2)) + else: + typer.echo(f"Application comparison: {status} ({base_commit[:12]} → {head_commit[:12]})") + for row in payload["rows"]: + typer.echo(f"{row['change'].upper()} {_one_line(row['agent'])} → {_one_line(row['tool'])}") + for side in ("before", "after"): + value = row[side] + if value is None: + typer.echo(f" {side}: no observed binding") + else: + definition = value.get("definition", {}) + typer.echo(f" {side}: {_one_line(value.get('signature') or value['tool'])} at {_one_line(value.get('binding_location'))}") + if definition: + typer.echo(f" implementation: {_one_line(definition['source'])}:{definition['line']} ({str(definition['implementation_sha256'])[:12]})") + typer.echo(f" {_one_line(row['why'])}") + typer.echo(f" Review: {_one_line(row['review_question'])}") + if not rows: + typer.echo("No supported application agents were established." if status == "not_established" else + "No established binding/interface/implementation changes in the observed surface.") + for side in ("base", "head"): + for limit in payload[side]["limits"]: + typer.echo(f" {side} limit: {_one_line(limit)}") + typer.echo(payload["limits"][-1]) + return 0 diff --git a/src/agents_shipgate/cli/diff.py b/src/agents_shipgate/cli/diff.py index b7c59daa6..f908a042a 100644 --- a/src/agents_shipgate/cli/diff.py +++ b/src/agents_shipgate/cli/diff.py @@ -47,7 +47,7 @@ DIFF_SCHEMA_VERSION = "0.4" -def _resolve_base(workspace: Path, base: str | None) -> tuple[str, str]: +def _resolve_base(workspace: Path, base: str | None, head: str = "HEAD") -> tuple[str, str]: """The base ref and the merge-base commit this diff compares against.""" from agents_shipgate.cli.verify.git import ( @@ -92,7 +92,7 @@ def refuse_shallow() -> None: # remote is the authority it might be stale against, and used only where # the repository has no remote at all. requested = base or detect_default_base( - workspace, "HEAD", allow_local_when_no_remote=True, allow_equal_head=True + workspace, head, allow_local_when_no_remote=True, allow_equal_head=True ) if requested is None: raise typer.BadParameter( @@ -110,12 +110,12 @@ def refuse_shallow() -> None: "this command never fetches.", param_hint="--base", ) - resolved = merge_base_sha(workspace, requested, "HEAD") + resolved = merge_base_sha(workspace, requested, head) if resolved is None: if truncated: refuse_shallow() raise typer.BadParameter( - f"No merge base between {requested!r} and HEAD, so there is no " + f"No merge base between {requested!r} and {head}, so there is no " "common point to compare from.", param_hint="--base", ) @@ -127,7 +127,7 @@ def refuse_shallow() -> None: # remaining root may be a graft hiding a better common ancestor. # HEAD/self and fully visible paths to the candidate remain usable. base_commit = commit_sha(workspace, requested) - head_commit = commit_sha(workspace, "HEAD") + head_commit = commit_sha(workspace, head) if base_commit is None or head_commit is None: refuse_shallow() if not shallow_merge_base_is_proven(workspace, base_commit, head_commit, resolved): @@ -484,10 +484,30 @@ def diff( "comparison runs from its merge base with HEAD." ), ), + application: bool = typer.Option(False, "--application", help="Compare source-observed application agent wiring without setup (SDK/ADK)."), + head: str | None = typer.Option(None, "--head", help="Application comparison head ref; defaults to committed HEAD."), + scope: str = typer.Option(".", "--scope", help="Application directory relative to the repository."), + base_scope: str | None = typer.Option(None, "--base-scope", help="Old application directory for an explicitly selected scope move."), + max_python_files: int = typer.Option(1000, "--max-python-files", min=1, help="Application discovery parse bound."), json_output: bool = typer.Option(False, "--json", help="Emit the rows as JSON."), ) -> None: """Show what this change does to the agent's authority.""" + if application: + from agents_shipgate.cli.application_diff import run_application_diff + from agents_shipgate.core.errors import ConfigError, InputParseError + try: + code = run_application_diff(workspace=workspace, base=base, head=head or "HEAD", + scope=scope, base_scope=base_scope, + max_python_files=max_python_files, json_output=json_output) + except (ConfigError, InputParseError) as exc: + from agents_shipgate.cli.agent_mode import emit_agent_mode_error + typer.echo(f"Application comparison could not read its inputs: {exc}", err=True) + emit_agent_mode_error("input_parse_error", message=str(exc), exit_code=2) + raise typer.Exit(2) from exc + raise typer.Exit(code) + if head is not None or scope != "." or base_scope is not None: + raise typer.BadParameter("--head, --scope and --base-scope require --application.") raise typer.Exit( run_capability_diff(workspace=workspace, base=base, json_output=json_output) ) diff --git a/src/agents_shipgate/core/agent_bindings.py b/src/agents_shipgate/core/agent_bindings.py index c3d6d6889..94fa57516 100644 --- a/src/agents_shipgate/core/agent_bindings.py +++ b/src/agents_shipgate/core/agent_bindings.py @@ -100,7 +100,7 @@ class _RawHandoffEdge: def resolve_agent_binding_graph( - manifest: AgentsShipgateManifest, + manifest: AgentsShipgateManifest | None, tools: list[Tool], artifacts: ArtifactBag, loaded_sources: list[LoadedToolSource] | None = None, @@ -109,6 +109,8 @@ def resolve_agent_binding_graph( Catalog membership is deliberately not evidence. Only reviewed manifest declarations and framework-specific structural observations create edges. + With ``manifest=None`` only reader observations contribute; callers may + compare per-agent wiring without supplying any reviewed declaration. """ ( @@ -118,7 +120,7 @@ def resolve_agent_binding_graph( partials, invalid_annotations, ) = _observations(tools, artifacts, loaded_sources or []) - declarations = manifest.agent_bindings.declarations + declarations = manifest.agent_bindings.declarations if manifest is not None else [] # A declaration may introduce an agent the extractors never saw — a # decorator-only repository has no agent object to observe. It must NOT # introduce a second node for one the extractors *did* see: the seeded @@ -932,7 +934,7 @@ def _selector_field(selector: Any, field: str) -> str | None: def _declared_source_surfaces( - manifest: AgentsShipgateManifest, + manifest: AgentsShipgateManifest | None, ) -> list[tuple[ToolSourceConfig, str, _AgentObservation]]: """Every ``tool_sources[]`` entry whose published surface is reviewed. @@ -957,13 +959,13 @@ def _declared_source_surfaces( tool_source_surface=True, ), ) - for index, source in enumerate(manifest.tool_sources) + for index, source in enumerate(manifest.tool_sources if manifest is not None else []) if source.binding is not None ] def _select_root( - manifest: AgentsShipgateManifest, + manifest: AgentsShipgateManifest | None, agents: list[AgentBindingNode], raw_handoffs: list[_RawHandoffEdge], surface_agent_ids: frozenset[str] = frozenset(), @@ -984,7 +986,7 @@ def _select_root( either way. """ - root_config = manifest.agent_bindings.root + root_config = manifest.agent_bindings.root if manifest is not None else None observed = [agent for agent in agents if agent.agent_id not in surface_agent_ids] surfaces = [agent for agent in agents if agent.agent_id in surface_agent_ids] candidates = observed @@ -1003,9 +1005,9 @@ def _select_root( agent for agent in matches if agent.agent_id not in surface_agent_ids ] candidates = observed_matches or matches - elif manifest.agent.sdk and manifest.agent.sdk.object: + elif manifest is not None and manifest.agent.sdk and manifest.agent.sdk.object: candidates = [agent for agent in observed if agent.name == manifest.agent.sdk.object] - elif len(manifest.agent_bindings.declarations) == 1: + elif manifest is not None and len(manifest.agent_bindings.declarations) == 1: declared = manifest.agent_bindings.declarations[0].agent if declared == "root": if len(observed) == 1: diff --git a/src/agents_shipgate/inputs/google_adk.py b/src/agents_shipgate/inputs/google_adk.py index d971610fd..1f8c3705e 100644 --- a/src/agents_shipgate/inputs/google_adk.py +++ b/src/agents_shipgate/inputs/google_adk.py @@ -321,13 +321,18 @@ def _attach_remote_bindings( def load_google_adk_artifacts( - manifest: AgentsShipgateManifest, + manifest: AgentsShipgateManifest | None, base_dir: Path, + *, + sources: list[ToolSourceConfig] | None = None, ) -> tuple[list[LoadedToolSource], GoogleAdkArtifacts | None]: + # Discovery supplies source locations only; it makes no manifest claims. source_refs = [ - source for source in manifest.tool_sources if source.type == "google_adk" + source for source in (sources if sources is not None else + manifest.tool_sources if manifest is not None else []) + if source.type == "google_adk" ] - config = manifest.google_adk + config = manifest.google_adk if manifest is not None else None if not source_refs and (config is None or not config.has_inputs()): return [], None diff --git a/src/agents_shipgate/inputs/openai_sdk_static.py b/src/agents_shipgate/inputs/openai_sdk_static.py index 79839793a..31f6f9dfb 100644 --- a/src/agents_shipgate/inputs/openai_sdk_static.py +++ b/src/agents_shipgate/inputs/openai_sdk_static.py @@ -46,9 +46,9 @@ def load_openai_sdk_static_tools( - source: ToolSourceConfig, manifest: AgentsShipgateManifest, base_dir: Path + source: ToolSourceConfig, manifest: AgentsShipgateManifest | None, base_dir: Path ) -> LoadedToolSource: - entrypoint = source.path or (manifest.agent.sdk.entrypoint if manifest.agent.sdk else None) + entrypoint = source.path or (manifest.agent.sdk.entrypoint if manifest is not None and manifest.agent.sdk else None) if not entrypoint: return LoadedToolSource( source_id=source.id, diff --git a/tests/test_application_diff.py b/tests/test_application_diff.py new file mode 100644 index 000000000..cd81be14d --- /dev/null +++ b/tests/test_application_diff.py @@ -0,0 +1,310 @@ +"""Application comparisons must work before adoption and retain uncertainty.""" +import json +import subprocess + +import pytest +from typer.testing import CliRunner + +from agents_shipgate.cli.main import app + +SDK = '''from agents import Agent, function_tool +@function_tool +def lookup(query: str) -> str: + return query +@function_tool +def execute(code: str) -> str: + return code +agent = Agent(name="assistant", tools=TOOLS) +''' + + +def git(root, *args): + result = subprocess.run(['git', '-c', 'core.excludesFile=/dev/null', *args], + cwd=root, capture_output=True, text=True, check=True) + return result.stdout.strip() + + +def commit(root, files): + for name, content in files.items(): + path = root / name + if content is None: + path.unlink() + else: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content) + git(root, 'add', '-A') + git(root, '-c', 'user.name=Test', '-c', 'user.email=test@example.com', + '-c', 'commit.gpgsign=false', 'commit', '--allow-empty', '-qm', 'fixture') + return git(root, 'rev-parse', 'HEAD') + + +@pytest.fixture +def repo(tmp_path): + git(tmp_path, 'init', '-q', '-b', 'main') + return tmp_path + + +def run(repo, base, head, *args): + result = CliRunner().invoke(app, ['diff', '--application', '--workspace', str(repo), + '--base', base, '--head', head, '--json', *args]) + assert result.exit_code == 0, result.output + repr(result.exception) + return json.loads(result.output) + + +def test_fresh_repository_reports_actual_wiring_not_catalog(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup, execute]')}) + result = run(repo, base, head) + assert result['comparison_status'] == 'compared' + assert [(r['agent'], r['tool'], r['change']) for r in result['rows']] == [('agent', 'execute', 'added')] + row = result['rows'][0] + assert row['before'] is None + assert row['after']['definition']['line'] == 6 + assert row['after']['signature'] == 'execute(code) -> str' + assert result['input_origin'] == 'independent_tree_discovery' + assert result['base']['compared_commit'] == base + assert result['head']['compared_commit'] == head + assert not (repo / 'shipgate.yaml').exists() + assert not (repo / '.agents-shipgate-local-review.yaml').exists() + assert not (repo / 'agents-shipgate-reports').exists() + assert git(repo, 'status', '--porcelain') == '' + assert not any(k in result for k in ['control', 'decision', 'release_decision', 'merge_verdict']) + assert result == run(repo, base, head) + + +def test_definition_without_wiring_does_not_become_capability(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]').replace('return code', 'return code.upper()')}) + assert run(repo, base, head)['rows'] == [] + + +def test_two_agents_with_no_deployment_root_keep_their_own_edges(repo): + initial = SDK.replace('TOOLS', '[lookup]') + '\nworker = Agent(name="worker", tools=[])\n' + base = commit(repo, {'agent.py': initial}) + head = commit(repo, {'agent.py': initial.replace('tools=[]', 'tools=[execute]')}) + rows = run(repo, base, head)['rows'] + assert [(r['agent'], r['tool']) for r in rows] == [('worker', 'execute')] + + +@pytest.mark.parametrize('change', ['signature', 'body', 'remove']) +def test_bound_tool_changes_and_removals(repo, change): + source = SDK.replace('TOOLS', '[lookup, execute]') + base = commit(repo, {'agent.py': source}) + after = {'signature': source.replace('code: str', 'code: int'), + 'body': source.replace('return code', 'return code.upper()'), + 'remove': source.replace('[lookup, execute]', '[lookup]')}[change] + head = commit(repo, {'agent.py': after}) + rows = run(repo, base, head)['rows'] + assert len(rows) == 1 + assert rows[0]['change'] == ('removed' if change == 'remove' else 'changed') + + +def test_comments_and_line_movement_are_not_capability_changes(repo): + source = SDK.replace('TOOLS', '[lookup]') + base = commit(repo, {'agent.py': source}) + head = commit(repo, {'agent.py': '# Comment\n\n' + source}) + assert run(repo, base, head)['rows'] == [] + + +def test_scope_added_absence_is_tree_bound(repo): + base = commit(repo, {'README.md': 'repository'}) + head = commit(repo, {'new/agent.py': SDK.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head, '--scope', 'new') + assert result['base']['status'] == 'absent' + assert len(result['rows']) == 1 + + +def test_explicit_scope_move_does_not_invent_added_tools(repo): + source = SDK.replace('TOOLS', '[lookup]') + base = commit(repo, {'old/agent.py': source}) + head = commit(repo, {'old/agent.py': None, 'new/agent.py': source}) + result = run(repo, base, head, '--scope', 'new', '--base-scope', 'old') + assert result['rows'] == [] + assert result['base']['scope'] == 'old' + assert result['head']['scope'] == 'new' + + +def test_bad_base_cannot_become_empty_base(repo): + base = commit(repo, {'agent.py': 'from agents import Agent\nAgent(\n'}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head) + assert result['comparison_status'] == 'partial' + assert result['rows'] == [] + assert any('parsed' in x for x in result['base']['limits']) + + +def test_bad_head_cannot_become_removed_tools(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + head = commit(repo, {'agent.py': 'from agents import Agent\nAgent(\n'}) + result = run(repo, base, head) + assert result['comparison_status'] == 'partial' + assert result['rows'] == [] + + +def test_truncation_cannot_become_no_change(repo): + base = commit(repo, {'a.py': '# ignored', 'agent.py': SDK.replace('TOOLS', '[lookup]')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup, execute]')}) + result = run(repo, base, head, '--max-python-files', '1') + assert result['comparison_status'] == 'partial' + assert any('truncat' in x or 'census' in x for x in result['base']['limits']) + + +def test_exact_head_does_not_read_dirty_worktree(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup, execute]')}) + (repo / 'agent.py').write_text('raise RuntimeError("never execute")') + assert len(run(repo, base, head)['rows']) == 1 + + +def test_adk_direct_binding_without_manifest(repo): + source = '''from google.adk.agents import Agent + +def lookup(query: str) -> str: + return query +root_agent = Agent(name="helper", tools=TOOLS) +''' + base = commit(repo, {'agent.py': source.replace('TOOLS', '[]')}) + head = commit(repo, {'agent.py': source.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head) + assert len(result['rows']) == 1 + assert result['rows'][0]['tool'] == 'lookup' + + +@pytest.mark.parametrize('scope', ['../outside', '/tmp', 'x/../../y']) +def test_scope_escape_refused(repo, scope): + ref = commit(repo, {'README.md': 'x'}) + result = CliRunner().invoke(app, ['diff', '--application', '--workspace', str(repo), + '--base', ref, '--scope', scope]) + assert result.exit_code == 2 + + +def test_identical_git_rename_preserves_binding_identity(repo): + source = SDK.replace('TOOLS', '[lookup]') + base = commit(repo, {'before/agent.py': source}) + head = commit(repo, {'before/agent.py': None, 'after/agent.py': source}) + result = run(repo, base, head) + assert result['rows'] == [] + assert result['source_correspondence'] == [{'base_source': 'before/agent.py', + 'head_source': 'after/agent.py', 'basis': 'git_rename_identical_blob'}] + + +def test_non_agent_repository_is_not_a_successful_no_change(repo): + base = commit(repo, {'README.md': 'Hello'}) + head = commit(repo, {'README.md': 'Hello again'}) + assert run(repo, base, head)['comparison_status'] == 'not_established' + + +def test_changed_function_default_is_not_silently_unchanged(repo): + source = SDK.replace('TOOLS', '[lookup]').replace('query: str', 'query: str = "first"') + base = commit(repo, {'agent.py': source}) + head = commit(repo, {'agent.py': source.replace('"first"', '"second"')}) + assert run(repo, base, head)['rows'][0]['change'] == 'changed' + + +def test_refs_record_requested_base_separately_from_merge_base(repo): + ancestor = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + git(repo, 'checkout', '-qb', 'feature') + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup, execute]')}) + git(repo, 'checkout', 'main') + base_tip = commit(repo, {'README.md': 'base moved'}) + result = run(repo, 'main', head) + assert result['base']['requested_commit'] == base_tip + assert result['base']['compared_commit'] == ancestor + assert result['head']['compared_commit'] == head + assert result['rows'][0]['tool'] == 'execute' + + +def test_application_code_is_never_executed(repo): + marker = repo / 'EXECUTED' + source = f'from pathlib import Path\nPath({str(marker)!r}).write_text("bad")\n' + SDK + base = commit(repo, {'agent.py': source.replace('TOOLS', '[]')}) + head = commit(repo, {'agent.py': source.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head) + assert len(result['rows']) == 1, result + assert not marker.exists() + + +def test_gitlink_refusal_is_not_a_successful_comparison(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[]')}) + git(repo, 'update-index', '--add', '--cacheinfo', f'160000,{base},external') + git(repo, '-c', 'user.name=Test', '-c', 'user.email=test@example.com', + '-c', 'commit.gpgsign=false', 'commit', '-qm', 'gitlink') + result = CliRunner().invoke(app, ['diff', '--application', '--workspace', str(repo), + '--base', base, '--head', 'HEAD', '--json']) + assert result.exit_code == 2 + assert '160000' in result.output + + +def test_python_size_bound_precedes_discovery(repo, monkeypatch): + import agents_shipgate.cli.application_diff as module + ref = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + monkeypatch.setattr(module, 'MAX_PYTHON_BYTES', 8) + monkeypatch.setattr(module, 'detect_workspace', lambda *a, **kw: pytest.fail('oversized input parsed')) + result = run(repo, ref, ref) + assert result['comparison_status'] == 'partial' + assert result['rows'] == [] + + +def test_adk_handoff_line_numbers_are_not_binding_identity(repo): + source = '''from google.adk.agents import Agent +worker = Agent(name="worker", tools=[]) +root_agent = Agent(name="root", tools=[], sub_agents=[worker]) +''' + base = commit(repo, {'agent.py': source}) + head = commit(repo, {'agent.py': '# shifted evidence locations\n\n' + source}) + assert run(repo, base, head)['rows'] == [] + + +def test_explicit_empty_tools_is_known_empty_wiring_not_missing_manifest(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', '[]')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head) + assert result['comparison_status'] == 'compared' + assert result['base']['binding_count'] == 0 + assert result['rows'][0]['tool'] == 'lookup' + + +def test_unresolved_tools_are_not_an_empty_surface(repo): + base = commit(repo, {'agent.py': SDK.replace('TOOLS', 'make_tools()')}) + head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) + result = run(repo, base, head) + assert result['comparison_status'] == 'partial' + assert result['rows'] == [] + + +def test_nonidentical_unpaired_move_does_not_invent_new_capabilities(repo): + source = SDK.replace('TOOLS', '[lookup]') + base = commit(repo, {'old/agent.py': source}) + head = commit(repo, {'old/agent.py': None, 'new/agent.py': '# changed comment\n' + source}) + result = run(repo, base, head) + assert result['comparison_status'] == 'partial' + assert result['rows'] == [] + assert any('unpaired' in x for x in result['head']['limits']) + + +def test_exact_file_move_preserves_handoff_identity(repo): + source = '''from google.adk.agents import Agent +worker = Agent(name="worker", tools=[]) +root_agent = Agent(name="root", tools=[], sub_agents=[worker]) +''' + base = commit(repo, {'old/agent.py': source}) + head = commit(repo, {'old/agent.py': None, 'new/agent.py': source}) + result = run(repo, base, head) + assert result['rows'] == [] + assert result['comparison_status'] == 'compared' + + +@pytest.mark.parametrize('json_output', [False, True]) +def test_diagnostic_limits_use_shared_redaction(repo, monkeypatch, json_output): + import agents_shipgate.cli.application_diff as module + ref = commit(repo, {'README.md': 'x'}) + secret = 'sk-privacyaaaaaaaaaaaaaaaa' + monkeypatch.setattr(module, 'observe', lambda *a, **kw: + module.Observations('.', status='partial', limits=[f'unresolved {secret}'])) + args = ['diff', '--application', '--workspace', str(repo), '--base', ref] + if json_output: + args.append('--json') + result = CliRunner().invoke(app, args) + assert result.exit_code == 0, result.output + assert secret not in result.output + assert '[REDACTED:' in result.output From c4530670385f34b6eb9e014d5077b6374e8b13a1 Mon Sep 17 00:00:00 2001 From: Pengfei Hu Date: Thu, 24 Sep 2026 18:09:22 -0700 Subject: [PATCH 2/3] fix: address application comparison review findings --- CHANGELOG.md | 4 + README.md | 7 +- docs/application-comparison.md | 40 +- docs/distribution-surfaces.md | 3 +- src/agents_shipgate/cli/application_diff.py | 523 +++++++++++++++----- src/agents_shipgate/cli/diff.py | 16 +- src/agents_shipgate/cli/verify/git.py | 12 +- tests/test_adapter_static_only.py | 2 +- tests/test_application_diff.py | 12 +- tests/test_application_diff_review.py | 328 ++++++++++++ tests/test_distribution_surface_parity.py | 8 + 11 files changed, 810 insertions(+), 145 deletions(-) create mode 100644 tests/test_application_diff_review.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 05d8b8474..35db9261b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased +### Application comparison without prior setup + +- Add `diff --application` for OpenAI Agents SDK and Google ADK source-observed per-agent wiring, with exact base/head refs and independently selected scopes. No manifest, saved baseline or authored declarations are needed. The advisory `application_comparison_schema_version: "0.1"` result records before/after evidence, scoped coverage gaps and explicit uncertain candidates; it supplies no release verdict or merge permission. Includes scoped Git materialization, partial-clone recovery, and definition lookup using reader-resolved Python symbols and locations. See [application comparison](docs/application-comparison.md). (#871) + ### Changes - Move the published-release pins, examples and adoption prompts to `v1.1.0` (contract 40) now that it is published, re-capture the README and quickstart `diff` answers from the published `1.1.0`, and re-measure the pilot ledger's Route H dry run on it. No schema or contract change. (#778) diff --git a/README.md b/README.md index 258536b3d..192d751dd 100644 --- a/README.md +++ b/README.md @@ -28,7 +28,7 @@ config, Codex plugin, n8n, and Conductor OSS workflow artifacts, then writes a deterministic **Tool-Use Readiness Report** before your agent gets production-like permissions. -## Application agent PRs without setup (unreleased) +## Application agent PRs without setup The source build can compare OpenAI Agents SDK and Google ADK tool bindings without a manifest or baseline: @@ -39,8 +39,9 @@ agents-shipgate diff --application --workspace /path/to/repo --base BASE_SHA --h It shows source-observed changes per agent, including before/after signatures and source locations. See [application comparison](docs/application-comparison.md) -for scoped applications, moves, exact refs and coverage limits. This option is -not yet in the published 1.1.0 package. +for scoped applications, moves, exact refs and coverage limits. Version availability +is recorded in the [CHANGELOG entry](CHANGELOG.md#application-comparison-without-prior-setup); +while it is under Unreleased, use a source build containing the feature. ## What did this PR change? diff --git a/docs/application-comparison.md b/docs/application-comparison.md index b910965a3..527cfa0dc 100644 --- a/docs/application-comparison.md +++ b/docs/application-comparison.md @@ -1,8 +1,8 @@ # Application comparison without prior setup -**Unreleased source feature.** The published 1.1.0 package does not yet have -`diff --application`. Run the source checkout's `./shipgate`, or a build -containing this feature. +**Availability:** see the [CHANGELOG entry](../CHANGELOG.md#application-comparison-without-prior-setup). +While that entry is under Unreleased, run the source checkout's `./shipgate` +or a build containing the feature. For an OpenAI Agents SDK or Google ADK application, compare committed PR refs: @@ -48,14 +48,21 @@ claim about deployed agent identity. An absent directory is absence at that Git path, not proof that no agent exists anywhere else. Missing manifests do not supply an empty base. +An absent side names the missing scope and suggests `--base-scope`/`--scope` +for relocation. If neither selected directory exists, the command refuses with +exit 2. A removal describes the selected source path, not the entire repository. + `--json` emits `application_comparison_schema_version: "0.1"`, engine identity, requested and compared refs/tree IDs, per-side scope/coverage, rows, source correspondence, and a deterministic `comparison_id`. This is a separate advisory artifact from the existing host diff JSON and verifier receipt. - `compared`: the selected supported source observations were compared. -- `partial`: a parse/discovery/binding gap remains. Proven positive observations - may be shown; unread input cannot establish absence or removal. +- `partial`: a parse/discovery/binding gap remains. Gaps identify their source + and agent where known. Only affected candidates become `change: not_established` + rows, carrying `candidate_change` and per-side `uncertainty`; independent known + additions/removals remain visible. Unknown implementation evidence cannot hide + an observed binding addition. Unattributed discovery bounds still cover the scope. - `not_established`: neither side established a supported application agent. - Exit 2: refs/materialization/input could not be read. This is not no change. @@ -66,3 +73,26 @@ explicit reader limitations. It does not support other application frameworks yet. Indirect helper effects, runtime loading, deployed reachability and business authority are outside this comparison. It grants no release or merge permission and cannot stand in for a reviewed verifier base or qualification evidence. + +## Evidence identity and recovery + +`coverage_gaps` records each gap's source/agent/tool and whether it affects +binding presence or only the implementation. Locations are relative to that +side's selected scope. A partial result with no rows is not a no-change answer. +Reader gaps are scoped to their input file unless typed agent evidence narrows +them further; this does not infer cross-module bindings by matching names. + +Implementation digests hash the resolved function's AST, including defaults +and decorators, excluding source positions and its leading docstring. Empty +AST fields are retained on Python 3.13+ to match 3.12. Digests are qualified by +the emitted engine/Python identity, not promised across arbitrary future AST +schema changes. `comparison_id` is SHA-256 over the **sanitized** published +object with that key removed, encoded as UTF-8 JSON with sorted keys, +`separators=(",", ":")` and `ensure_ascii=True`; readers can recompute it. + +Each side materializes only its selected scope through the existing verified +Git materializer, which retains symlinks and containment checks. Partial clones +with unfetched objects exit 2 with `objects_missing`, name the affected side, +and provide the existing `git fetch --refetch --no-filter ` recovery. +The comparison never runs that fetch. Other configuration/materialization errors +use `config_error`; malformed reader inputs use `input_parse_error` in agent mode. diff --git a/docs/distribution-surfaces.md b/docs/distribution-surfaces.md index d4459b271..73ee6e7eb 100644 --- a/docs/distribution-surfaces.md +++ b/docs/distribution-surfaces.md @@ -74,7 +74,8 @@ and this document are checked against each other by | `human_review_request` | `docs/human-review-request.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | One complete-evidence documentation-quality class only; no authority or decision ingestion. | | `human_review_decision` | `docs/human-review-decision.md` | `release_decision_vocabulary` | `test_surface_enumerations_match_the_engine_vocabulary` | Host-neutral read-only evaluator; no GitHub acquisition, persistence or operation authority. | | `github_action` | `action.yml`, `scripts/github_action_outputs.py` | `merge_verdict_vocabulary` | `test_action_input_enumerates_engine_merge_verdicts`, `test_action_output_script_shares_the_engine_merge_verdicts` | The paired `shipgate_wheel`/`shipgate_wheel_sha256` inputs install a caller-supplied local wheel instead of a published version, so that route names no channel and claims no `executable_pin`; it is refused unless both halves are given, and it installs `--no-deps`. `tests/test_action_engine_install.py` proves the refusals. Every `python` the Action starts in the workspace runs with `-P` or as a script path, so a pull request's `pip/` or `agents_shipgate/` package cannot stand in for pip or the engine; the same file executes the install and merge-verdict steps against such a checkout. The `v1.0.0` tag predates that fix; the published `v1.1.0` carries it. | -| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Host route only; workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, an unread argument input or an unresolved launch is named there and in the `why` of a row reporting its launch, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A surface the reader reaches through an in-tree link it reads through qualifies only when that link, a link with the same text at each link on the way, and the file it lands on, the same blob at the same path, are both unchanged, read from the base's Git tree entries against a commit's or, without following any link, the working tree's; any change to either is treated as before. The same proof decides which shared plugin-reference limits `check` leaves out, so behind such a link `check` compares, and publishes the rows it finds, exactly as for a limit at its own path, and a comparison `partial` only because of such a limit is `comparable` with it in `unchanged_limits` (#822, `tests/test_linked_unchanged_limits.py`). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL, its package and argument digest (#819) and the env and header key names its grant already publishes, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or another setting; a hook with each handler field that changed — its group's matcher, its command as its executable's name and digest, its timeout — before and after, a handler only one side declares, or the published handlers in a different order with a detail not shown that may also differ, and past the handler bound the same kind of sentence naming a handler past it, all read from the handlers its host-grants `0.7` grant publishes, which hold no command or argument text, and never re-derived here, and a declaration outside the documented hooks shape named as not shown rather than guessed (#819, `tests/test_hook_mcp_detail_fields.py`); those hook and MCP members display what `config_sha256` already binds, so grant equality and every inventory digest leave them out, a saved baseline holds none of them, and no row, row value, reason, digest or control answer moves; the PR comment gives the lines 1.1.0 printed their room first, the coverage block included, and prints an entry whole when the whole comment fits, otherwise cut to the widest length of at least 60 characters at which it does, or else in its shortest form (a difference cut after its name, an added or removed grant as its row), never longer than the entry 1.1.0 printed, with one line naming `verifier.json`, so no long entry hides a row, the coverage block, the change count, the review question, the reproduction or the advisory that 1.1.0 kept (#819 review, cycles 4 and 6); an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `capability_diff` | `src/agents_shipgate/cli/diff.py`, `src/agents_shipgate/core/capability_diff_rows.py`, `src/agents_shipgate/core/host_comparison.py`, `src/agents_shipgate/report/host_comparison.py`, `src/agents_shipgate/core/unread_inputs.py`, `src/agents_shipgate/cli/verify/changed_inputs.py` | — | — | Answers no question the engine answers: it emits no verdict, no release decision and no pin. Every field is read from the drift payload the engine already produces — `risk` is the engine's severity and `expansion_signals` is the engine's word on widening — so there is no second implementation to drift. A `permission_mode` or `sandbox` row names the setting and its value as the file spells it (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`), recovered from the grant's published value and digest, and a Claude Code setting's `why` is the basis the engine's one setting table (`core/host_settings.py`) records for the value; that table also rates the grant and `check`'s violation, so a row's severity and the violation's risk give one answer (#827, `tests/test_prompt_disabling_settings.py`). `verify`/PR and `check` reuse the host comparator (#684, `tests/test_manifest_free_pr_rows.py`), and the source name each named reusable-workflow secret refers to, also non-widening, with a redacting name or target refused rather than compared, and an unreadable value neither compared nor named on this surface — only the host inventory and `audit --host` name its `job/destination`, as on `1.0.0` (#693, `tests/test_reusable_workflow_secret_mappings.py`); check retains argument redaction and its existing local-policy control. Missing comparison evidence never supplies empty comparable rows. Default host mode only (`--application` is registered separately below); workflow rows compare effective writes and reusable secret recipients (#685, `tests/test_workflow_capability_diff.py`) and each job's remote step action references, as a non-widening change (#771, `tests/test_workflow_step_action_references.py`); and each job's agent launches — a documented agent action's permission inputs, the permission flags of a `run:` that is one plain `claude -p` / `codex exec` command — and checkout refs, compared as text, as a change unless the job gains a documented widening rule, which the engine names in `expansion_signals` (`workflow_agent_widened_*`) — a rule read only from text the engine reads exactly (no shell is parsed; an argument input that is not a plain list of words is compared by a digest and read for no rule), and a gain the engine does not claim (a rule moved in from a job the launch left, one an unread step of the job rewritten as a read launch may already have met, or one the job's launch held before in an expression or an unread argument input) named in the `why` from the same engine function, never counted — with a note on a workflow row naming the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, read off the grant the engine published and moving no direction; an unread `run:` agent step (never compared, so never a row), an unreadable value or a setting published redacted is named only by the host inventory and `audit --host`, as for an unread secret value, an unread argument input or an unresolved launch is named there and in the `why` of a row reporting its launch, and a checkout ref holding credential-shaped text is refused as a redacting step reference is (#823, `tests/test_workflow_agent_launches.py`); every job id, step label, trigger and scope name those rows print is the label the engine published once where it built the grant, redacted, never re-derived here; `check`'s workflow evidence is derived from the raw declarations, which it still compares, and redacts job and scope names by the same rule; two distinct job ids or triggers in one workflow, or scope names in one `permissions` mapping, that publish alike are refused rather than compared, so while such a workflow exists `check` refuses on every run even when it is unchanged (#802, `tests/test_workflow_label_redaction.py`); artifact-only edits remain separate evidence. Tool-source subjects are #655. Where a partial or experimental surface is byte-identical on both sides, `diff` and `verify` compare the rest and name it in `unchanged_limits`; `check` keeps refusing, because its boundary result cannot carry a limit yet (#721). A surface the reader reaches through an in-tree link it reads through qualifies only when that link, a link with the same text at each link on the way, and the file it lands on, the same blob at the same path, are both unchanged, read from the base's Git tree entries against a commit's or, without following any link, the working tree's; any change to either is treated as before. The same proof decides which shared plugin-reference limits `check` leaves out, so behind such a link `check` compares, and publishes the rows it finds, exactly as for a limit at its own path, and a comparison `partial` only because of such a limit is `comparable` with it in `unchanged_limits` (#822, `tests/test_linked_unchanged_limits.py`). A hook row's `why` states the grant's loading basis, read from its published `source`, `access` and `risk` by the engine's `hook_loading_basis`; only a hook the host loads for this project earns an expansion signal — one a settings layer declares, or one a plugin selects that the repository's project settings enable from an in-repository marketplace — so a declared-only hook, or one a plugin selects without that enablement, is a row and never an expansion, and a removal names no basis (#714). `check` compares without a plugin-reference limit both sides share on an untouched source, which it cannot name and, untouched, does not route; a limit only one side carries makes its comparison incomparable. Those rows are not what routes a change: `check`, and the boundary check a manifest-backed `verify` runs, route a changed hook declaration of a plugin the project settings enable through the existing protected-surface rule, from the plugin hook reader's selection on both compared sides, and count a changed hook file such a plugin selects that the reader does not open as incomplete input; the rows beside either are unchanged (#809, `tests/test_enabled_plugin_hook_routing.py`). A partial clone that never fetched the base's objects is refused as `objects_missing`, exit `2`, never compared and never fetched; the refusal ends with the remediation sentence `verify` reports for the same reason, produced by the same function (#817, `tests/test_capability_diff_partial_clone.py`). The text of `diff`, `verify`, the PR comment and `check` reads the rows through one function, `review_changes`, and adds no row and changes no row value in any JSON projection (#795, `tests/test_host_diff_review_changes.py`): a permission rule is named with its disposition; an MCP server with the command name (never its path) or redacted URL, its package and argument digest (#819) and the env and header key names its grant already publishes, a URL printing only in the engine's sanitized scheme-and-host form and otherwise as `url not shown`, or, when none of those differ, a sentence naming what was compared and that the change is in a detail not shown, such as the command's path or another setting; a hook with each handler field that changed — its group's matcher, its command as its executable's name and digest, its timeout — before and after, a handler only one side declares, or the published handlers in a different order with a detail not shown that may also differ, and past the handler bound the same kind of sentence naming a handler past it, all read from the handlers its host-grants `0.7` grant publishes, which hold no command or argument text, and never re-derived here, and a declaration outside the documented hooks shape named as not shown rather than guessed (#819, `tests/test_hook_mcp_detail_fields.py`); those hook and MCP members display what `config_sha256` already binds, so grant equality and every inventory digest leave them out, a saved baseline holds none of them, and no row, row value, reason, digest or control answer moves; the PR comment gives the lines 1.1.0 printed their room first, the coverage block included, and prints an entry whole when the whole comment fits, otherwise cut to the widest length of at least 60 characters at which it does, or else in its shortest form (a difference cut after its name, an added or removed grant as its row), never longer than the entry 1.1.0 printed, with one line naming `verifier.json`, so no long entry hides a row, the coverage block, the change count, the review question, the reproduction or the advisory that 1.1.0 kept (#819 review, cycles 4 and 6); an allow rule the permission lattice decided another replaced (`widened` or `narrowed`), or the exact rule text that moved between dispositions in one host and source (`moved`), is one entry, never on the routes that redact rule arguments; and `diff` counts entries `from N rows` when one joins rows. Comparable results with entries end with one review question, naming the row count when an entry joins rows, and every result whose comparison names a base commit and a commit or working-tree head — a zero-row result and a refusal included (#812 follow-up, `tests/test_host_comparison_coverage.py`) — ends with the compared commits, the tool version and an `agents-shipgate diff --base ` reproduction, labelled `Inputs:` rather than `Compared:` where the comparison was refused, since that run compared nothing — and a refused comparison publishes no `review` object at all, so those two lines are the only place that run states its provenance, built from the `base_commit` it publishes beside the refusal; `check` and a provided diff print the question alone, and no result without a change asks a question. Every one of those facts is published beside the rows, so a machine consumer reads what a human reads (#795 slice 2, same test file): a row adds `disposition`, the `allow`/`ask`/`deny` list a permission rule is declared under and `null` for any other kind, on every route that publishes rows; and `review` in `diff --json` (capability diff `0.3`) and `host_comparison.review` in `verifier.json` (verifier `0.20`) — one object for one comparison — carry the presented changes, each naming the `row_indexes` it stands for, the `direction` the text uses (`widened`, `narrowed` and `moved` included, which no single row can carry), its cells, its `why` and one `expands`, plus a `summary` of `{rows, changes, widenings}` equal to `diff`'s summary line, the review question and the reproduction command. The block is refused unless its changes stand for every published row exactly once, its counters match and no joined change's two sides read alike, so the routes that redact rule arguments publish their rows alone and never a pair that reads `X → X`; `check`'s boundary result carries rows, with their dispositions, and no block. It is presentation, not a second opinion: it is the one `review_changes` projection the text prints, so the rows, their values, their count and every control answer are what they were. A comparison read back from JSON prints the changes it published, and one whose rows a caller sliced falls back to those rows. Each comparison also says what it established (#812, `tests/test_host_comparison_coverage.py`): `coverage` in `diff --json` (capability diff `0.3`) and `host_comparison.coverage` in `verifier.json` (verifier `0.20`) are the same object, printed as `What this run established` by `diff`, `verify` text and the PR comment. It is read off the grant changes, artifact changes, observed sources and blocking issues the comparator already computed: a file's rows, counting a source inside it (`#profiles.`, `#plugins.`); a file with no row and no artifact change called unchanged (`compared`, `0` rows) only when Git proves its blob identical, as the check `unchanged_limits` uses does, asked privately in one bounded batch and never published, because the artifact digest redacts `env` values and `apiKeyHelper`; a file that changed with no compared grant moving (`changed_without_grant_change`), whose artifact differs only in its digest or whose content Git shows differs while its artifact did not (never a difference a checkout line-ending conversion or a converting attribute explains, and no filter is run), never a plugin manifest or marketplace, a retargeted link or project settings while a hook's loading basis moved, worded as no compared grant changing and never as which fields changed; any other changed file with no row (`changed_without_rows`); a file Git neither proves identical nor shows differs — a provided diff, a link read, a redacted path, a working-tree file a checkout wrote with `CRLF` that Git reports unchanged — as `unchanged_not_proven`, never no change and never a change (#812 review cycle 3); the side that published a source, worded `published by` rather than `read in` for a plugin manifest or marketplace, which is published only while it declares hooks; and on a refused comparison each blocking source and its kind. Outside the bounded candidate rules below, a file no inventory observed is never an item and its absence is no claim, which the block states where it is read — one line under the heading and `read_sources_only` in the JSON — so a true list cannot be taken for the account of the change (#812 follow-up); a source already in `unchanged_limits` is not repeated; the list is capped at ten with `omitted_items`, ordered so what no row shows precedes a file's rows and, among blocking limits, by kind (`unreadable`, `parse_failed`, `unresolved_precedence`, then `unsupported`, `dynamic_source_excluded`, `remote_source_excluded`) — order, not severity, and a ranking of kinds rather than of items, since `unsupported` carries both a file this entry merely does not accept and one whose own text would not parse, so an item behind the count may still be one to repair; total down to every field an item is keyed by, the source name and then its side, limit and status — and counted in text as items not listed, ranked below those listed, and the PR comment lists only what fits in the room its entries, review question, reproduction, advisory, next action and evidence leave, at most 2000 characters, so the block never pushes out a line the comment prints without it (a row list that fills the comment by itself still truncates it, as on `1.0.0`); an instruction file's line carries no redacted-values note; sources are the inventory's redacted paths; `null` means not recorded, which is how a `0.19` verifier reads. It moves no row, reason, digest, baseline, control state or next action, and `check`'s boundary result and text carry none, so neither `check` nor a provided diff asks Git anything for it. The same list names the changed inputs this entry does not read (#821, `tests/test_unread_changed_inputs.py`): capability diff `0.4` and verifier `0.21` add a `changed_not_read` item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set — the committed change, or the working tree's tracked and untracked changes — that a bounded, documented rule set recognises as plausibly agent configuration (`mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, a marketplace entry's external `source`) and that no inventory published; a member is named whatever read its file, because no reader reads it. It is named from the path and, for a manifest or marketplace member, its text: nothing is fetched, run or read as a grant, so it is never a row, a widening, a `check` violation or a loading claim, and an external source is described redacted and never fetched. It ranks right after the blocking limits, inside the same cap; `read_sources_only` is `false` while one is named, and the first line says so instead; `unread_candidates` and `unread_candidates_not_examined` say whether the change set was examined and how many candidates were not — past the bound of 32, or because a file the rule needed was not read or did not parse, one count the text names both causes of. A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, publishes the comparison instead of the setup route, and `verify --preview` then names `audit --host` instead of `init --write`, in an agent-related workspace too; a `0.20` verifier reads with the search not recorded. A comparison refused only by plugin-reference limits, each bounded by its plugin directory, that no compared source depends on, is `partial` instead (#808, `tests/test_partial_host_comparison.py`); any other blocking limit it carries must be one both sides share on an unchanged source, named in `unchanged_limits` as on a comparable result. Capability diff `0.4` and verifier `0.21` publish `comparison_status: partial` with the refusal's `incomparable_reasons`, the rows, review and unchanged limits established outside those directories, and each directory (the outermost, where one holds another) as the reserved `coverage.items[].scope` on the `blocking_limit` items it bounds, and never call a changed project settings file without a row `changed_without_grant_change`, since the hooks whose loading basis it decides are not all compared; `diff`, `verify` text and the PR comment lead with `Partial comparison against …` or `Host capability comparison partial: …` and `Not compared: , …` before any entry, and a partial result with no entry is never printed as no change. Independence is read off the reader's reference graph, never off directory names: any other limit that is not unchanged, a reference leaving its plugin, a plugin at the root or holding project settings, a marketplace elsewhere declaring inline hooks for it, or a directory that does not publish as itself refuses as before. It answers no engine question and moves no control: a partial comparison is not comparable, `verify`'s control and route are the refusal's, the control envelope projects it as `incomparable` with no rows, and `check`, whose boundary result cannot name a directory, refuses its comparison and decides exactly as before. A `0.20` verifier claiming a partial comparison or a scope is refused. | +| `application_diff` | `src/agents_shipgate/cli/application_diff.py` | — | — | Advisory SDK/ADK source-wiring comparison through `diff --application`. Reuses framework observations and the binding graph; its comparator answers no release or merge verdict, activation verdict, executable pin, or declared authority question. Unlike host mode it computes per-agent source differences, not a projection of host drift. Scoped gaps and `not_established` candidates bound the answer; no deployment-root reachability is asserted. `tests/test_application_diff.py` and `tests/test_application_diff_review.py` prove the advisory boundary, isolation, uncertainty and evidence identity. | | `zero_install_detector` | `tools/shipgate-detect.py` | `agent_project_verdict` | `test_detector_verdict_matches_cli` | Emits no `diagnostics[]` and no `next_actions[]`; evidence strings and framework scores are simplified. See the script's own "Intentional simplifications". | | `emitted_ci_workflow` | `src/agents_shipgate/cli/discovery/ci_workflow.py` | `executable_pin` | `tests/test_adopter_pins_resolve.py::test_the_emitted_workflow_pins_the_release_and_not_the_source_tree`, `tests/test_release_source.py::test_candidate_workflow_uses_immutable_source_before_and_after_publication` | Ordinary/source/preview builds use the published fallback; a stamped candidate pins its verified Action SHA and package version. Before publication its smoke substitutes the exact local wheel inputs. Provenance asserts no qualification. | | `prompts` | `prompts/` | `contract_floor`, `executable_pin`, `placeholder_ownership`, `release_decision_vocabulary` | `test_executable_pin_resolves_in_a_published_channel`, `test_surface_enumerations_match_the_engine_vocabulary`, `test_surface_routes_human_owned_placeholders_to_a_human`, `tests/test_adopter_pins_resolve.py::test_every_pin_init_writes_into_an_adopter_repo_names_the_published_release`, `tests/test_adopter_pins_resolve.py::test_the_shipped_floor_is_decided_against_the_release_the_prompts_pin` | — | diff --git a/src/agents_shipgate/cli/application_diff.py b/src/agents_shipgate/cli/application_diff.py index 7bab3c6e2..ad9d8e37e 100644 --- a/src/agents_shipgate/cli/application_diff.py +++ b/src/agents_shipgate/cli/application_diff.py @@ -3,11 +3,13 @@ This entry reads immutable trees and the existing SDK/ADK adapters. It never constructs a release manifest, supplies declarations, or publishes a verdict. """ + from __future__ import annotations import ast import hashlib import json +import sys import tempfile from dataclasses import dataclass, field from pathlib import Path, PurePosixPath @@ -18,7 +20,13 @@ from agents_shipgate.cli.discovery import detect_workspace from agents_shipgate.cli.discovery.artifacts import _candidate_files from agents_shipgate.cli.scan.source_loading import _build_canonical_tools -from agents_shipgate.cli.verify.git import archive_tree, commit_sha, ensure_git_workspace, tree_sha +from agents_shipgate.cli.verify.git import ( + PromisedObjectsMissingError, + archive_fetched_tree, + commit_sha, + ensure_git_workspace, + tree_sha, +) from agents_shipgate.core.agent_bindings import resolve_agent_binding_graph from agents_shipgate.core.artifacts import ArtifactBag from agents_shipgate.core.errors import ConfigError @@ -34,7 +42,9 @@ def _digest(value: Any) -> str: - return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + return hashlib.sha256( + json.dumps(value, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() def _scope(value: str) -> str: @@ -52,12 +62,58 @@ class Observations: bindings: dict[tuple[str, str, str], dict[str, Any]] = field(default_factory=dict) limits: list[str] = field(default_factory=list) sources: list[dict[str, str]] = field(default_factory=list) + coverage_gaps: list[dict[str, Any]] = field(default_factory=list) + + def gap( + self, + message: str, + *, + source: str | None = None, + agent: str | None = None, + tool: str | None = None, + affects: str = "binding_presence", + ) -> None: + self.limits.append(message) + self.status = "partial" + item = { + "source": source, + "agent": agent, + "tool": tool, + "affects": affects, + "reason": message, + } + if item not in self.coverage_gaps: + self.coverage_gaps.append(item) + + def absence_gaps( + self, key: tuple[str, str, str], moves: dict[str, str] | None = None + ) -> list[str]: + reasons = [] + for gap in self.coverage_gaps: + path = gap["source"] + if path is not None: + path = (moves or {}).get(path, path) + if ( + gap["affects"] == "binding_presence" + and (path is None or key[0] == path or key[0].startswith(path + "/")) + and (gap["agent"] is None or gap["agent"] == key[1]) + and (gap["tool"] is None or gap["tool"] == key[2]) + ): + reasons.append(gap["reason"]) + return sorted(set(reasons)) def summary(self) -> dict[str, Any]: - return {"scope": self.scope, "status": self.status, - "sources": self.sources, "agents": list(self.agents.values()), - "binding_count": len(self.bindings), "limits": sorted(set(self.limits))} - + return { + "scope": self.scope, + "status": self.status, + "sources": self.sources, + "agents": list(self.agents.values()), + "binding_count": len(self.bindings), + "limits": sorted(set(self.limits)), + "coverage_gaps": sorted( + self.coverage_gaps, key=lambda gap: json.dumps(gap, sort_keys=True) + ), + } def _source_path(root: Path, ref: str) -> str: @@ -70,18 +126,33 @@ def _source_path(root: Path, ref: str) -> str: return path return ref + def _definition(root: Path, tool: Any) -> dict[str, Any]: """Read a unique function's AST; line movement/comments are not changes.""" path = _source_path(root, tool.source_ref or "") - symbol = (tool.native_locator or "").split("#")[-1] or tool.name - if "#" not in (tool.native_locator or ""): - symbol = (tool.function_signature or tool.name).split("(")[0] + symbol = tool.annotations.get("python_symbol") + if not isinstance(symbol, str): + symbol = ( + (tool.native_locator or "").split("#")[-1] + if "#" in (tool.native_locator or "") + else (tool.function_signature or tool.name).split("(")[0] + ) file = root / path if not file.is_file() or not file.resolve().is_relative_to(root.resolve()): return {"source": path, "line": None, "implementation_sha256": None} try: - nodes = [n for n in ast.walk(ast.parse(file.read_bytes())) - if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef)) and n.name == symbol] + tree = ast.parse(file.read_bytes()) + location = tool.source_location or "" + line = location.rsplit(":", 1)[-1] + nodes = [ + n + for n in ast.walk(tree) + if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef)) + and n.name == symbol + and (not line.isdecimal() or n.lineno == int(line)) + ] + if not line.isdecimal(): + nodes = [n for n in tree.body if n in nodes] except (SyntaxError, ValueError, RecursionError): nodes = [] if len(nodes) != 1: @@ -89,20 +160,40 @@ def _definition(root: Path, tool: Any) -> dict[str, Any]: node = nodes[0] # Docstrings are description evidence, not an implementation change. body = node.body - if body and isinstance(body[0], ast.Expr) and isinstance(body[0].value, ast.Constant) and isinstance(body[0].value.value, str): + if ( + body + and isinstance(body[0], ast.Expr) + and isinstance(body[0].value, ast.Constant) + and isinstance(body[0].value.value, str) + ): body = body[1:] node.body = body # Include defaults and decorators: approval decorators and changed default # bounds are review-relevant even when the executable body is unchanged. - return {"source": path, "line": node.lineno, - "implementation_sha256": _digest(ast.dump(node, include_attributes=False))} + return { + "source": path, + "line": node.lineno, + "implementation_sha256": _digest( + ast.dump( + node, + include_attributes=False, + **({"show_empty": True} if sys.version_info >= (3, 13) else {}), + ) + ), + } def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: result = Observations(scope) root = tree / scope + if root.is_symlink(): + raise ConfigError(f"Application scope is not a regular directory: {scope}") if not root.exists(): result.status = "absent" + result.limits.append( + f"Scope {scope!r} is absent in this tree. If the application moved, " + "select its old path with --base-scope and new path with --scope." + ) return result if not root.is_dir() or root.is_symlink(): raise ConfigError(f"Application scope is not a regular directory: {scope}") @@ -112,55 +203,94 @@ def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: python_files = [p for p in _candidate_files(root) if p.suffix == ".py"] for file in python_files: if file.stat().st_size > MAX_PYTHON_BYTES: - result.limits.append(f"Python input exceeds {MAX_PYTHON_BYTES} bytes: {file.relative_to(root)}") + result.gap(f"Python input exceeds {MAX_PYTHON_BYTES} bytes: {file.relative_to(root)}") if result.limits: result.status = "partial" return result detected = detect_workspace(root, max_python_files=max_python_files) if detected.python_parse_truncated: - result.limits.append(f"Python discovery truncated at {max_python_files} files.") - result.limits.extend(f"Discovery could not read {p}." for p in detected.host_discovery_incomplete_paths) - result.limits.extend(f"Excluded candidate: {item}" for item in detected.excluded_sources) + result.gap(f"Python discovery truncated at {max_python_files} files.") + for path in detected.host_discovery_incomplete_paths: + result.gap(f"Discovery could not read {path}.", source=path) + for item in detected.excluded_sources: + result.gap(f"Excluded candidate: {item}", source=item.get("path")) # Discovery omits malformed Python; preserve that gap rather than an empty # candidate list becoming negative evidence. Bound this second parse too. if len(python_files) > max_python_files: - result.limits.append(f"Python input census exceeds {max_python_files} files.") + result.gap(f"Python input census exceeds {max_python_files} files.") for file in python_files[:max_python_files]: if file.is_symlink(): - result.limits.append(f"Linked Python input: {file.relative_to(root)}") + result.gap( + f"Linked Python input: {file.relative_to(root)}", + source=file.relative_to(root).as_posix(), + ) continue try: ast.parse(file.read_bytes()) except (SyntaxError, ValueError, RecursionError, OSError): - result.limits.append(f"Python input could not be parsed: {file.relative_to(root)}") + result.gap( + f"Python input could not be parsed: {file.relative_to(root)}", + source=file.relative_to(root).as_posix(), + ) for framework in detected.frameworks: if framework.type not in SUPPORTED and framework.candidate_files: - result.limits.append(f"Application comparison does not yet support {framework.type}: {', '.join(framework.candidate_files)}") - entries = sorted({(f.type, p) for f in detected.frameworks if f.type in SUPPORTED - for p in f.candidate_files}) - sources = [ToolSourceConfig(id=f"{kind}:{path}", type=kind, path=path) for kind, path in entries] + for path in framework.candidate_files: + result.gap( + f"Application comparison does not yet support {framework.type}: {path}", + source=path, + ) + entries = sorted( + {(f.type, p) for f in detected.frameworks if f.type in SUPPORTED for p in f.candidate_files} + ) + sources = [ + ToolSourceConfig(id=f"{kind}:{path}", type=kind, path=path) for kind, path in entries + ] result.sources = [{"type": s.type, "path": s.path} for s in sources] - bag = ArtifactBag() - loaded = [] for source in sources: - if source.type == "openai_agents_sdk": - loaded.append(load_openai_sdk_static_tools(source, None, root)) - adk, artifacts = load_google_adk_artifacts(None, root, sources=sources) - loaded.extend(adk) + _observe_source(result, root, source) + return result + + +def _observe_source(result: Observations, root: Path, source: ToolSourceConfig) -> None: + """Each reader's unlocated gaps cover its input file, not other sources. + + A reader's agent-level observations narrow that further. Cross-module + binding resolution remains the reader's responsibility, never a name join. + """ + bag = ArtifactBag() + if source.type == "openai_agents_sdk": + loaded = [load_openai_sdk_static_tools(source, None, root)] + artifacts = None + else: + loaded, artifacts = load_google_adk_artifacts(None, root, sources=[source]) + if artifacts is not None: + bag.set("google_adk", artifacts) + attributed = set() + for item in loaded: + for observation in item.binding_observations: + if not observation.tools_complete or not observation.handoffs_complete: + for message in observation.issues or ["Incomplete observed binding list."]: + result.gap( + message, + source=_source_path(root, observation.source), + agent=observation.agent, + ) + attributed.add(message) + for warning in item.warnings: + if warning not in attributed: + result.gap(warning, source=source.path) + attributed.add(warning) + for omission in item.omissions: + result.gap(f"Omitted source surface: {omission}", source=source.path) if artifacts is not None: - bag.set("google_adk", artifacts) - result.limits.extend(artifacts.warnings) - result.limits.extend(w for item in loaded for w in item.warnings) - result.limits.extend(f"Omitted source surface: {o}" for item in loaded for o in item.omissions) + for warning in artifacts.warnings: + if warning not in attributed: + result.gap(warning, source=source.path) + attributed.add(warning) tools, warnings = _build_canonical_tools(loaded) - result.limits.extend(warnings) + for warning in warnings: + result.gap(warning, source=source.path) graph, _ = resolve_agent_binding_graph(None, tools, bag, loaded) - # Deployment-root selection is irrelevant to observed per-agent edges. - # A root with an explicit empty tool list can also leave unrelated catalog - # definitions unbound. That is a release-scope concern, not missing observed - # wiring. Dynamic/partial lists have their own retained reader/graph issues. - result.limits.extend(i.message for i in graph.issues if i.kind not in - {"ambiguous_root_agent", "missing_binding_evidence"}) tool_by_id = {t.id: t for t in tools} agent_keys = {} ambiguous_agents = set() @@ -168,10 +298,28 @@ def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: key = (_source_path(root, agent.source_ref or ""), agent.name) agent_keys[agent.agent_id] = key if key in result.agents: - result.limits.append(f"Ambiguous agent identity: {key}") + result.gap(f"Ambiguous agent identity: {key}", source=key[0], agent=key[1]) ambiguous_agents.add(key) - result.agents[key] = {"name": agent.name, "source": key[0], - "location": agent.source_pointer} + for binding in list(result.bindings): + if binding[:2] == key: + result.bindings.pop(binding) + result.agents[key] = { + "name": agent.name, + "source": key[0], + "location": agent.source_pointer, + } + for issue in graph.issues: + if ( + issue.kind in {"ambiguous_root_agent", "missing_binding_evidence"} + or issue.message in attributed + ): + continue + agent_key = agent_keys.get(issue.agent_id) + result.gap( + issue.message, + source=agent_key[0] if agent_key else source.path, + agent=agent_key[1] if agent_key else None, + ) ambiguous_bindings = set() for edge in graph.tool_edges: key = agent_keys[edge.agent_id] @@ -180,20 +328,36 @@ def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: tool = tool_by_id[edge.tool_id] definition = _definition(root, tool) if definition["implementation_sha256"] is None: - result.limits.append(f"Implementation location unresolved: {tool.name} ({tool.source_ref}).") + result.gap( + f"Implementation location unresolved: {tool.name} ({tool.source_ref}).", + source=key[0], + agent=key[1], + tool=tool.name, + affects="implementation", + ) binding_key = (*key, tool.name) if binding_key in ambiguous_bindings: continue if binding_key in result.bindings: - result.limits.append(f"Ambiguous tool identity: {binding_key}") + result.gap( + f"Ambiguous tool identity: {binding_key}", + source=key[0], + agent=key[1], + tool=tool.name, + ) result.bindings.pop(binding_key) ambiguous_bindings.add(binding_key) continue result.bindings[binding_key] = { - "agent": key[1], "agent_source": key[0], "tool": tool.name, - "binding_location": edge.source_pointer, "edge_type": edge.edge_type, - "definition": definition, "input_schema": tool.input_schema, - "output_schema": tool.output_schema, "signature": tool.function_signature, + "agent": key[1], + "agent_source": key[0], + "tool": tool.name, + "binding_location": edge.source_pointer, + "edge_type": edge.edge_type, + "definition": definition, + "input_schema": tool.input_schema, + "output_schema": tool.output_schema, + "signature": tool.function_signature, "evidence_basis": edge.provenance_kind, } for edge in graph.handoff_edges: @@ -201,56 +365,99 @@ def observe(tree: Path, scope: str, *, max_python_files: int) -> Observations: if source in ambiguous_agents or target in ambiguous_agents: continue key = (*source, f"handoff:{target[0]}:{target[1]}") - result.bindings[key] = {"agent": source[1], "agent_source": source[0], - "tool": target[1], "edge_type": edge.edge_type, - "target_source": target[0], "binding_location": edge.source_pointer, - "evidence_basis": edge.provenance_kind} - if result.limits: - result.status = "partial" - return result + result.bindings[key] = { + "agent": source[1], + "agent_source": source[0], + "tool": target[1], + "edge_type": edge.edge_type, + "target_source": target[0], + "binding_location": edge.source_pointer, + "evidence_basis": edge.provenance_kind, + } def _meaning(binding: dict[str, Any]) -> dict[str, Any]: - return {k: v for k, v in binding.items() if k not in - {"binding_location", "definition", "evidence_basis", "agent_source"}} | { - "implementation_sha256": binding.get("definition", {}).get("implementation_sha256")} + return { + k: v + for k, v in binding.items() + if k not in {"binding_location", "definition", "evidence_basis", "agent_source"} + } | {"implementation_sha256": binding.get("definition", {}).get("implementation_sha256")} -def compare(base: Observations, head: Observations, *, - target_moves: dict[str, str] | None = None) -> list[dict[str, Any]]: +def compare( + base: Observations, head: Observations, *, target_moves: dict[str, str] | None = None +) -> list[dict[str, Any]]: rows = [] for key in sorted(base.bindings.keys() | head.bindings.keys()): before, after = base.bindings.get(key), head.bindings.get(key) - if before is not None and after is not None: + uncertainty = {} + kind = "added" if before is None else "removed" if after is None else "changed" + if before is None: + reasons = base.absence_gaps(key, target_moves) + if reasons: + uncertainty["base"] = reasons + elif after is None: + reasons = head.absence_gaps(key) + if reasons: + uncertainty["head"] = reasons + else: before_meaning = _meaning(before) if "target_source" in before_meaning: path = before_meaning["target_source"] before_meaning["target_source"] = (target_moves or {}).get(path, path) - if before_meaning == _meaning(after): + for side, value in (("base", before), ("head", after)): + if "definition" in value and value["definition"]["implementation_sha256"] is None: + uncertainty[side] = ["The bound callable's implementation could not be read."] + if before_meaning == _meaning(after) and not uncertainty: continue - if before is None and base.status == "partial": - head.limits.append(f"Cannot establish whether {key[1]}.{key[2]} was absent at base.") - continue - if after is None and head.status == "partial": - head.limits.append(f"Cannot establish removal of {key[1]}.{key[2]} from incomplete head inputs.") - continue - kind = "added" if before is None else "removed" if after is None else "changed" - rows.append({"agent": key[1], "agent_source": key[0], "tool": key[2], - "change": kind, "before": before, "after": after, - "why": {"added": "The source now binds this callable to this agent.", - "removed": "The source no longer binds this callable to this agent.", - "changed": "The bound callable's interface or implementation changed; authority direction is not established."}[kind], - "review_question": f"Should {key[1]} have this {kind} binding to {key[2]}? Review the before/after signature and implementation locations."}) + candidate_change = kind + if uncertainty: + kind = "not_established" + rows.append( + { + "agent": key[1], + "agent_source": key[0], + "tool": key[2], + "change": kind, + "candidate_change": candidate_change if uncertainty else None, + "uncertainty": uncertainty, + "before": before, + "after": after, + "why": { + "added": "The source now binds this callable to this agent.", + "removed": "The selected source path no longer binds this callable to this agent; check scope limits for relocation.", + "not_established": "This candidate change cannot be established from the affected inputs; it is not a no-change result.", + "changed": "The bound callable's interface or implementation changed; authority direction is not established.", + }[kind], + "review_question": ( + f"Resolve the named uncertainty before treating {key[1]}.{key[2]} as {candidate_change}." + if uncertainty + else f"Should {key[1]} have this {kind} binding to {key[2]}? Review the before/after signature and implementation locations." + ), + } + ) return rows - -def _align_exact_moves(workspace: Path, base: str, head: str, - old: Observations, new: Observations) -> list[dict[str, str]]: +def _align_exact_moves( + workspace: Path, base: str, head: str, old: Observations, new: Observations +) -> list[dict[str, str]]: from agents_shipgate.cli.verify.git import _run_git - diff = _run_git(workspace, ["diff", "--no-ext-diff", "--no-textconv", "--name-status", - "-z", "-M100%", base, head, "--"]) + diff = _run_git( + workspace, + [ + "diff", + "--no-ext-diff", + "--no-textconv", + "--name-status", + "-z", + "-M100%", + base, + head, + "--", + ], + ) if diff.returncode: raise ConfigError("Could not establish source relocation evidence.") parts = iter(diff.stdout.split("\0")) @@ -283,20 +490,30 @@ def _align_exact_moves(workspace: Path, base: str, head: str, for key in collisions: aligned.pop(key, None) new.bindings.pop(key, None) - old.limits.append(f"Ambiguous relocated binding: {key}") - old.status = "partial" + old.gap(f"Ambiguous relocated binding: {key}", source=key[0], agent=key[1], tool=key[2]) old.bindings = aligned - old_names = {name for path, name in old.agents if moves.get(path, path) not in - {new_path for new_path, new_name in new.agents if new_name == name}} - new_names = {name for path, name in new.agents if path not in - {moves.get(old_path, old_path) for old_path, old_name in old.agents if old_name == name}} + old_names = { + name + for path, name in old.agents + if moves.get(path, path) + not in {new_path for new_path, new_name in new.agents if new_name == name} + } + new_names = { + name + for path, name in new.agents + if path + not in { + moves.get(old_path, old_path) for old_path, old_name in old.agents if old_name == name + } + } for name in sorted(old_names & new_names): message = f"Agent {name!r} occurs at different unpaired source paths; select --base-scope/--scope or review relocation." - old.limits.append(message) - new.limits.append(message) - old.status = new.status = "partial" - return [{"base_source": a, "head_source": b, "basis": "git_rename_identical_blob"} - for a, b in sorted(moves.items())] + old.gap(message, agent=name) + new.gap(message, agent=name) + return [ + {"base_source": a, "head_source": b, "basis": "git_rename_identical_blob"} + for a, b in sorted(moves.items()) + ] def _location(scope: str, path: str | None) -> str | None: @@ -318,10 +535,18 @@ def _published_binding(binding: dict[str, Any] | None, scope: str) -> dict[str, result["definition"]["source"] = _location(scope, result["definition"]["source"]) return result -def run_application_diff(*, workspace: Path, base: str | None, head: str, - scope: str, base_scope: str | None, - max_python_files: int, json_output: bool) -> int: - from agents_shipgate.cli.diff import _one_line, _resolve_base + +def run_application_diff( + *, + workspace: Path, + base: str | None, + head: str, + scope: str, + base_scope: str | None, + max_python_files: int, + json_output: bool, +) -> int: + from agents_shipgate.cli.diff import _one_line, _refuse_objects_missing, _resolve_base workspace = ensure_git_workspace(workspace) scope, old_scope = _scope(scope), _scope(base_scope if base_scope is not None else scope) @@ -332,8 +557,13 @@ def run_application_diff(*, workspace: Path, base: str | None, head: str, raise typer.BadParameter(f"Head ref {head!r} is unavailable locally. Fetch it first.") from agents_shipgate.cli.verify.git import detect_default_base - base_ref = base if base is not None else detect_default_base( - workspace, head_commit, allow_local_when_no_remote=True, allow_equal_head=True) + base_ref = ( + base + if base is not None + else detect_default_base( + workspace, head_commit, allow_local_when_no_remote=True, allow_equal_head=True + ) + ) if base_ref is None: # Keep the established missing-base diagnostic and recovery wording. _resolve_base(workspace, None, head_commit) @@ -345,10 +575,30 @@ def run_application_diff(*, workspace: Path, base: str | None, head: str, engine = build_engine_requirement(plugins_enabled=False).model_dump(mode="json") with tempfile.TemporaryDirectory(prefix="shipgate-application-diff-") as raw: scratch = Path(raw) - archive_tree(workspace, base_commit, scratch / "base") - archive_tree(workspace, head_commit, scratch / "head") + for side, ref, commit, selected_scope in ( + ("base", base_ref, base_commit, old_scope), + ("head", head, head_commit, scope), + ): + + def in_scope(path: str, selected: str = selected_scope) -> bool: + return path == selected or path.startswith(selected + "/") + + try: + archive_fetched_tree( + workspace, + commit, + scratch / side, + scope=None if selected_scope == "." else in_scope, + ) + except PromisedObjectsMissingError: + _refuse_objects_missing(workspace, ref, commit, side=side) old = observe(scratch / "base", old_scope, max_python_files=max_python_files) new = observe(scratch / "head", scope, max_python_files=max_python_files) + if old.status == new.status == "absent": + raise ConfigError( + f"Neither comparison tree contains the selected scopes: " + f"base={old_scope!r}, head={scope!r}. Check --scope/--base-scope." + ) moves = _align_exact_moves(workspace, base_commit, head_commit, old, new) rows = compare(old, new, target_moves={m["base_source"]: m["head_source"] for m in moves}) for row in rows: @@ -359,41 +609,76 @@ def run_application_diff(*, workspace: Path, base: str | None, head: str, status = "not_established" payload = { "application_comparison_schema_version": SCHEMA_VERSION, - "comparison_status": status, "static_analysis_only": True, + "comparison_status": status, + "static_analysis_only": True, "comparison_basis": "source_observed_per_agent_wiring", - "input_origin": "independent_tree_discovery", "engine": engine, - "base": {"requested_ref": base_ref, "requested_commit": requested_base_commit, - "compared_commit": base_commit, "tree": tree_sha(workspace, base_commit), **old.summary()}, - "head": {"requested_ref": head, "compared_commit": head_commit, - "tree": tree_sha(workspace, head_commit), **new.summary()}, + "input_origin": "independent_tree_discovery", + "engine": engine, + "base": { + "requested_ref": base_ref, + "requested_commit": requested_base_commit, + "compared_commit": base_commit, + "tree": tree_sha(workspace, base_commit), + **old.summary(), + }, + "head": { + "requested_ref": head, + "compared_commit": head_commit, + "tree": tree_sha(workspace, head_commit), + **new.summary(), + }, "options": {"max_python_files": max_python_files, "max_python_bytes": MAX_PYTHON_BYTES}, - "source_correspondence": moves, "rows": rows, - "limits": ["Covers supported OpenAI Agents SDK and Google ADK source wiring only.", - "Deployment-root reachability, runtime behavior, indirect helper effects and business authority are not established.", - "This comparison is advisory evidence and supplies no release verdict or merge permission."], + "source_correspondence": moves, + "rows": rows, + "limits": [ + "Covers supported OpenAI Agents SDK and Google ADK source wiring only.", + "Deployment-root reachability, runtime behavior, indirect helper effects and business authority are not established.", + "This comparison is advisory evidence and supplies no release verdict or merge permission.", + ], } - payload["comparison_id"] = _digest(payload) payload = sanitize_report_payload(payload) + payload["comparison_id"] = _digest(payload) if json_output: typer.echo(json.dumps(payload, ensure_ascii=False, indent=2)) else: typer.echo(f"Application comparison: {status} ({base_commit[:12]} → {head_commit[:12]})") for row in payload["rows"]: - typer.echo(f"{row['change'].upper()} {_one_line(row['agent'])} → {_one_line(row['tool'])}") + typer.echo( + f"{row['change'].upper()} {_one_line(row['agent'])} → {_one_line(row['tool'])}" + ) for side in ("before", "after"): value = row[side] if value is None: - typer.echo(f" {side}: no observed binding") + typer.echo( + f" {side}: " + + ( + "binding presence not established" + if row["uncertainty"] + else "no observed binding" + ) + ) else: definition = value.get("definition", {}) - typer.echo(f" {side}: {_one_line(value.get('signature') or value['tool'])} at {_one_line(value.get('binding_location'))}") + typer.echo( + f" {side}: {_one_line(value.get('signature') or value['tool'])} at {_one_line(value.get('binding_location'))}" + ) if definition: - typer.echo(f" implementation: {_one_line(definition['source'])}:{definition['line']} ({str(definition['implementation_sha256'])[:12]})") + typer.echo( + f" implementation: {_one_line(definition['source'])}:{definition['line']} ({str(definition['implementation_sha256'])[:12]})" + ) typer.echo(f" {_one_line(row['why'])}") + for side, reasons in row["uncertainty"].items(): + for reason in reasons: + typer.echo(f" {side} uncertainty: {_one_line(reason)}") typer.echo(f" Review: {_one_line(row['review_question'])}") if not rows: - typer.echo("No supported application agents were established." if status == "not_established" else - "No established binding/interface/implementation changes in the observed surface.") + typer.echo( + "No supported application agents were established." + if status == "not_established" + else "Incomplete comparison; this is not a no-change result." + if status == "partial" + else "No established binding/interface/implementation changes in the observed surface." + ) for side in ("base", "head"): for limit in payload[side]["limits"]: typer.echo(f" {side} limit: {_one_line(limit)}") diff --git a/src/agents_shipgate/cli/diff.py b/src/agents_shipgate/cli/diff.py index f908a042a..e3379c6e2 100644 --- a/src/agents_shipgate/cli/diff.py +++ b/src/agents_shipgate/cli/diff.py @@ -135,7 +135,7 @@ def refuse_shallow() -> None: return requested, resolved -def _refuse_objects_missing(workspace: Path, base_ref: str, base_commit: str) -> NoReturn: +def _refuse_objects_missing(workspace: Path, base_ref: str, base_commit: str, *, side: str = "base") -> NoReturn: """Name the hydration a partial clone needs instead of a traceback (#817). In a `--filter=blob:none` clone only what was checked out has blobs, and in @@ -162,7 +162,7 @@ def _refuse_objects_missing(workspace: Path, base_ref: str, base_commit: str) -> fetch = ["git", "-C", str(workspace), "fetch", "--refetch", "--no-filter"] refused = ( - f"The base side of this diff, {_one_line(base_ref)} ({base_commit[:8]}), " + f"The {side} side of this diff, {_one_line(base_ref)} ({base_commit[:8]}), " "could not be read (objects_missing)" ) remotes = promisor_remotes(workspace) @@ -175,7 +175,7 @@ def _refuse_objects_missing(workspace: Path, base_ref: str, base_commit: str) -> kind="command", command=command, why=( - f"If `{_one_line(remotes[0])}` cannot supply the base's objects, " + f"If `{_one_line(remotes[0])}` cannot supply the {side}'s objects, " f"hydrate from `{_one_line(remote)}`, another remote this " "partial clone was promised objects by, then rerun." ), @@ -488,7 +488,7 @@ def diff( head: str | None = typer.Option(None, "--head", help="Application comparison head ref; defaults to committed HEAD."), scope: str = typer.Option(".", "--scope", help="Application directory relative to the repository."), base_scope: str | None = typer.Option(None, "--base-scope", help="Old application directory for an explicitly selected scope move."), - max_python_files: int = typer.Option(1000, "--max-python-files", min=1, help="Application discovery parse bound."), + max_python_files: int | None = typer.Option(None, "--max-python-files", min=1, help="Application discovery parse bound."), json_output: bool = typer.Option(False, "--json", help="Emit the rows as JSON."), ) -> None: """Show what this change does to the agent's authority.""" @@ -499,15 +499,15 @@ def diff( try: code = run_application_diff(workspace=workspace, base=base, head=head or "HEAD", scope=scope, base_scope=base_scope, - max_python_files=max_python_files, json_output=json_output) + max_python_files=max_python_files if max_python_files is not None else 1000, json_output=json_output) except (ConfigError, InputParseError) as exc: from agents_shipgate.cli.agent_mode import emit_agent_mode_error typer.echo(f"Application comparison could not read its inputs: {exc}", err=True) - emit_agent_mode_error("input_parse_error", message=str(exc), exit_code=2) + emit_agent_mode_error("input_parse_error" if isinstance(exc, InputParseError) else "config_error", message=str(exc), exit_code=2) raise typer.Exit(2) from exc raise typer.Exit(code) - if head is not None or scope != "." or base_scope is not None: - raise typer.BadParameter("--head, --scope and --base-scope require --application.") + if head is not None or scope != "." or base_scope is not None or max_python_files is not None: + raise typer.BadParameter("--head, --scope, --base-scope and --max-python-files require --application.") raise typer.Exit( run_capability_diff(workspace=workspace, base=base, json_output=json_output) ) diff --git a/src/agents_shipgate/cli/verify/git.py b/src/agents_shipgate/cli/verify/git.py index 5cb88e50c..cdb132726 100644 --- a/src/agents_shipgate/cli/verify/git.py +++ b/src/agents_shipgate/cli/verify/git.py @@ -3100,10 +3100,12 @@ def archive_fetched_tree( A blobless clone fails the object copy with a :class:`ConfigError`, and a treeless one fails the tree lookup before it with a - :class:`subprocess.CalledProcessError`. Either is re-raised unchanged - unless :func:`promised_objects_missing` says the clone was promised - objects it does not hold; then :class:`PromisedObjectsMissingError`, so a - caller can name the hydration instead of a traceback (#817). + :class:`subprocess.CalledProcessError`. Both become + :class:`PromisedObjectsMissingError` when :func:`promised_objects_missing` + proves the clone lacks promised objects, so a caller can name the + hydration instead of a traceback (#817). Other Git + process failures are normalized to ConfigError at this subprocess boundary, + so CLI callers need no process-execution import just to catch an error. """ try: @@ -3111,6 +3113,8 @@ def archive_fetched_tree( except (ConfigError, subprocess.CalledProcessError) as exc: if promised_objects_missing(workspace, commit): raise PromisedObjectsMissingError(commit) from exc + if isinstance(exc, subprocess.CalledProcessError): + raise ConfigError(f"Could not materialize Git tree for {commit}: {exc}") from exc raise diff --git a/tests/test_adapter_static_only.py b/tests/test_adapter_static_only.py index 43626c14a..70990d87f 100644 --- a/tests/test_adapter_static_only.py +++ b/tests/test_adapter_static_only.py @@ -280,7 +280,7 @@ class AllowedException: AllowedException( relative_path="cli/verify/git.py", surface="attr_call:subprocess.run", - line=3272, + line=3276, snippet=( "subprocess.run(cmd, capture_output=capture_output, check=check, " "env=env, input=input, stderr=stderr, stdin=stdin, stdout=stdout, " diff --git a/tests/test_application_diff.py b/tests/test_application_diff.py index cd81be14d..9878bf5fa 100644 --- a/tests/test_application_diff.py +++ b/tests/test_application_diff.py @@ -129,7 +129,8 @@ def test_bad_base_cannot_become_empty_base(repo): head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) result = run(repo, base, head) assert result['comparison_status'] == 'partial' - assert result['rows'] == [] + assert result['rows'] + assert all(row['change'] == 'not_established' for row in result['rows']) assert any('parsed' in x for x in result['base']['limits']) @@ -138,7 +139,8 @@ def test_bad_head_cannot_become_removed_tools(repo): head = commit(repo, {'agent.py': 'from agents import Agent\nAgent(\n'}) result = run(repo, base, head) assert result['comparison_status'] == 'partial' - assert result['rows'] == [] + assert result['rows'] + assert all(row['change'] == 'not_established' for row in result['rows']) def test_truncation_cannot_become_no_change(repo): @@ -269,7 +271,8 @@ def test_unresolved_tools_are_not_an_empty_surface(repo): head = commit(repo, {'agent.py': SDK.replace('TOOLS', '[lookup]')}) result = run(repo, base, head) assert result['comparison_status'] == 'partial' - assert result['rows'] == [] + assert result['rows'] + assert all(row['change'] == 'not_established' for row in result['rows']) def test_nonidentical_unpaired_move_does_not_invent_new_capabilities(repo): @@ -278,7 +281,8 @@ def test_nonidentical_unpaired_move_does_not_invent_new_capabilities(repo): head = commit(repo, {'old/agent.py': None, 'new/agent.py': '# changed comment\n' + source}) result = run(repo, base, head) assert result['comparison_status'] == 'partial' - assert result['rows'] == [] + assert result['rows'] + assert all(row['change'] == 'not_established' for row in result['rows']) assert any('unpaired' in x for x in result['head']['limits']) diff --git a/tests/test_application_diff_review.py b/tests/test_application_diff_review.py new file mode 100644 index 000000000..7ad40e104 --- /dev/null +++ b/tests/test_application_diff_review.py @@ -0,0 +1,328 @@ +"""Review regressions: scoped uncertainty, definition identity and usable recovery.""" + +import json + +import pytest +from test_application_diff import SDK, commit, git, run +from test_application_diff import repo as repo +from typer.testing import CliRunner + +from agents_shipgate.cli.main import app + + +@pytest.mark.parametrize("same_file", [False, True]) +@pytest.mark.parametrize("removing", [False, True]) +def test_unrelated_dynamic_agent_cannot_hide_a_known_change(repo, same_file, removing): + other = '\nfrom agents import Agent\nhelper = Agent(name="helper", tools=make_tools())\n' + before, after = "[lookup]", "[lookup, execute]" + if removing: + before, after = after, before + base = commit( + repo, + { + "agent.py": SDK.replace("TOOLS", before) + (other if same_file else ""), + **({} if same_file else {"other.py": other}), + }, + ) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", after) + (other if same_file else "")}) + result = run(repo, base, head) + assert result["comparison_status"] == "partial" + assert [(r["tool"], r["change"]) for r in result["rows"]] == [ + ("execute", "removed" if removing else "added") + ] + assert all(g["agent"] == "helper" for g in result["base"]["coverage_gaps"]) + + +def test_unrelated_malformed_file_does_not_hide_an_addition(repo): + base = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]"), "broken.py": "def (\n"}) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup, execute]")}) + result = run(repo, base, head) + assert result["comparison_status"] == "partial" + assert [(r["tool"], r["change"]) for r in result["rows"]] == [("execute", "added")] + + +@pytest.mark.parametrize("pattern", ["override", "method"]) +def test_reader_definition_identity_detects_body_changes(repo, pattern): + source = SDK.replace("TOOLS", "[lookup]") + if pattern == "override": + source = source.replace( + "@function_tool\ndef lookup", '@function_tool(name_override="search")\ndef lookup' + ) + else: + source += "\nclass Other:\n def lookup(self, query):\n return query\n" + base = commit(repo, {"agent.py": source}) + head = commit(repo, {"agent.py": source.replace("return query\n", "return query.upper()\n", 1)}) + result = run(repo, base, head) + assert result["comparison_status"] == "compared" + assert len(result["rows"]) == 1 + row = result["rows"][0] + assert row["change"] == "changed" + assert ( + row["before"]["definition"]["implementation_sha256"] + != row["after"]["definition"]["implementation_sha256"] + ) + + +def test_unknown_implementation_does_not_hide_an_observed_addition(repo, monkeypatch): + import agents_shipgate.cli.application_diff as module + + definition = module._definition + + def missing(root, tool): + value = definition(root, tool) + if tool.name == "lookup": + value["implementation_sha256"] = None + return value + + monkeypatch.setattr(module, "_definition", missing) + base = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]")}) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup, execute]")}) + result = run(repo, base, head) + assert {r["tool"]: r["change"] for r in result["rows"]} == { + "lookup": "not_established", + "execute": "added", + } + assert result["base"]["coverage_gaps"][0]["affects"] == "implementation" + + +def test_uncertainty_belongs_to_base_and_is_not_no_change_text(repo): + base = commit(repo, {"agent.py": SDK.replace("TOOLS", "make_tools()")}) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]")}) + result = run(repo, base, head) + assert result["head"]["status"] == "complete" + assert result["head"]["limits"] == [] + assert list(result["rows"][0]["uncertainty"]) == ["base"] + text = CliRunner().invoke( + app, ["diff", "--application", "--workspace", str(repo), "--base", base, "--head", head] + ) + assert text.exit_code == 0 + assert "NOT_ESTABLISHED" in text.output + assert "No established" not in text.output + + +def test_unpaired_agent_move_does_not_hide_another_agents_addition(repo): + base = commit( + repo, + { + "old.py": SDK.replace("TOOLS", "[lookup]"), + "stable.py": SDK.replace("TOOLS", "[lookup]").replace("agent =", "worker ="), + }, + ) + head = commit( + repo, + { + "old.py": None, + "new.py": "# moved\n" + SDK.replace("TOOLS", "[lookup]"), + "stable.py": SDK.replace("TOOLS", "[lookup, execute]").replace("agent =", "worker ="), + }, + ) + rows = run(repo, base, head)["rows"] + assert [(r["agent"], r["tool"]) for r in rows if r["change"] == "added"] == [ + ("worker", "execute") + ] + assert all(r["change"] == "not_established" for r in rows if r["agent"] == "agent") + + +def test_scoped_archive_materializes_only_selected_blobs(repo, monkeypatch): + import agents_shipgate.cli.application_diff as module + + archive = module.archive_fetched_tree + seen = [] + + def scoped(workspace, ref, destination, **kwargs): + archive(workspace, ref, destination, **kwargs) + assert not (destination / "unrelated.txt").exists() + seen.append( + sorted(p.relative_to(destination).as_posix() for p in destination.rglob("*.py")) + ) + + monkeypatch.setattr(module, "archive_fetched_tree", scoped) + source = SDK.replace("TOOLS", "[lookup]") + base = commit(repo, {"old/agent.py": source, "unrelated.txt": "large unrelated artifact"}) + head = commit(repo, {"old/agent.py": None, "new/agent.py": source}) + assert run(repo, base, head, "--base-scope", "old", "--scope", "new")["rows"] == [] + assert seen == [["old/agent.py"], ["new/agent.py"]] + + +def test_moved_scope_names_absence_and_recovery(repo): + source = SDK.replace("TOOLS", "[lookup]") + base = commit(repo, {"backend/agent.py": source}) + head = commit(repo, {"backend/agent.py": None, "server/agent.py": "# moved\n" + source}) + result = CliRunner().invoke( + app, + [ + "diff", + "--application", + "--workspace", + str(repo), + "--base", + base, + "--head", + head, + "--scope", + "backend", + ], + ) + assert result.exit_code == 0, result.output + assert "Scope 'backend' is absent" in result.output + assert "--base-scope" in result.output + + +def test_misspelled_scope_is_an_input_error(repo): + ref = commit(repo, {"backend/agent.py": SDK.replace("TOOLS", "[lookup]")}) + result = CliRunner().invoke( + app, + ["diff", "--application", "--workspace", str(repo), "--base", ref, "--scope", "bakend"], + env={"AGENTS_SHIPGATE_AGENT_MODE": "1"}, + ) + assert result.exit_code == 2 + assert "Neither comparison tree contains" in result.output + assert "config_error" in result.output + + +@pytest.mark.parametrize("bound", ["5", "1000"]) +def test_explicit_application_bound_requires_application_mode(repo, bound): + ref = commit(repo, {"README.md": "x"}) + result = CliRunner().invoke( + app, ["diff", "--workspace", str(repo), "--base", ref, "--max-python-files", bound] + ) + assert result.exit_code == 2 + assert "--application" in result.output and "--max-python-files" in result.output + + +def test_redacted_comparison_identity_is_recomputable(repo, monkeypatch): + import agents_shipgate.cli.application_diff as module + + ref = commit(repo, {"README.md": "x"}) + monkeypatch.setattr( + module, + "observe", + lambda *a, **kw: module.Observations( + ".", status="partial", limits=["sk-privacyaaaaaaaaaaaaaaaa"] + ), + ) + result = run(repo, ref, ref) + identity = result.pop("comparison_id") + assert "sk-privacyaaaaaaaaaaaaaaaa" not in json.dumps(result) + assert identity == module._digest(result) + + +@pytest.mark.parametrize("filter_spec", ["blob:none", "tree:0"]) +@pytest.mark.parametrize("missing_side", ["base", "head"]) +def test_partial_clone_names_side_and_hydration(repo, filter_spec, missing_side): + from test_capability_diff_partial_clone import _missing_objects, _object_store, _partial_clone + + git(repo, "config", "uploadpack.allowFilter", "true") + base = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]")}) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup, execute]")}) + # Check out only one side before cloning, leaving the other commit's objects promised. + if missing_side == "head": + git(repo, "branch", "feature", head) + git(repo, "reset", "--hard", base) + clone = _partial_clone(repo, repo.name + "-partial", filter_spec) + missing_ref = base if missing_side == "base" else head + assert _missing_objects(clone, missing_ref) + before = _object_store(clone) + result = CliRunner().invoke( + app, + [ + "diff", + "--application", + "--workspace", + str(clone), + "--base", + base, + "--head", + head, + "--json", + ], + env={"AGENTS_SHIPGATE_AGENT_MODE": "1"}, + ) + assert result.exit_code == 2, result.output + assert f"The {missing_side} side" in result.output + assert "objects_missing" in result.output + assert "--refetch --no-filter" in result.output + assert "Traceback" not in result.output + assert before == _object_store(clone) + + +def test_imported_tool_gap_is_explicit_and_not_counted_as_complete(repo): + source = SDK.replace("TOOLS", "[lookup]") + tool_source = source[: source.index("agent =")] + agent_source = 'from agents import Agent\nfrom tools import lookup, execute\nagent = Agent(name="app", tools=TOOLS)\n' + base = commit( + repo, {"tools.py": tool_source, "agent.py": agent_source.replace("TOOLS", "[lookup]")} + ) + head = commit(repo, {"agent.py": agent_source.replace("TOOLS", "[lookup, execute]")}) + result = run(repo, base, head) + assert result["comparison_status"] == "partial" + assert any("unresolved tool" in reason for reason in result["head"]["limits"]) + assert result["rows"] == [] + text = CliRunner().invoke( + app, ["diff", "--application", "--workspace", str(repo), "--base", base, "--head", head] + ) + assert "not a no-change result" in text.output + + +def test_name_override_does_not_hide_newly_bound_execution_tool(repo): + source = SDK.replace( + "@function_tool\ndef lookup", '@function_tool(name_override="search")\ndef lookup' + ) + base = commit(repo, {"agent.py": source.replace("TOOLS", "[lookup]")}) + head = commit(repo, {"agent.py": source.replace("TOOLS", "[lookup, execute]")}) + result = run(repo, base, head) + assert result["comparison_status"] == "compared" + assert [(r["tool"], r["change"]) for r in result["rows"]] == [("execute", "added")] + + +def test_excluded_unrelated_artifact_does_not_hide_application_change(repo, monkeypatch): + import agents_shipgate.cli.application_diff as module + + detect = module.detect_workspace + + def with_exclusion(*args, **kwargs): + detected = detect(*args, **kwargs) + detected.excluded_sources.append( + {"path": "unrelated.json", "type": "mcp", "reason": "malformed export"} + ) + return detected + + monkeypatch.setattr(module, "detect_workspace", with_exclusion) + base = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]")}) + head = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup, execute]")}) + result = run(repo, base, head) + assert result["comparison_status"] == "partial" + assert [(r["tool"], r["change"]) for r in result["rows"]] == [("execute", "added")] + + +def test_scoped_materialization_keeps_link_refusal(repo): + ref = commit(repo, {"real/agent.py": SDK.replace("TOOLS", "[lookup]")}) + (repo / "alias").symlink_to("real") + git(repo, "add", "alias") + ref = commit(repo, {}) + result = CliRunner().invoke( + app, ["diff", "--application", "--workspace", str(repo), "--base", ref, "--scope", "alias"] + ) + assert result.exit_code == 2 + assert "not a regular directory" in result.output + + + +def test_non_promisor_git_error_is_normalized_at_materialization_boundary(repo, monkeypatch): + import subprocess + + from agents_shipgate.cli.verify import git as module + + ref = commit(repo, {"agent.py": SDK.replace("TOOLS", "[lookup]")}) + + def broken(*args, **kwargs): + raise subprocess.CalledProcessError(128, ["git", "rev-parse"]) + + monkeypatch.setattr(module, "archive_tree", broken) + monkeypatch.setattr(module, "promised_objects_missing", lambda *a: False) + result = CliRunner().invoke(app, ["diff", "--application", "--workspace", str(repo), + "--base", ref, "--json"], + env={"AGENTS_SHIPGATE_AGENT_MODE": "1"}) + assert result.exit_code == 2 + assert "config_error" in result.output + assert "Traceback" not in result.output diff --git a/tests/test_distribution_surface_parity.py b/tests/test_distribution_surface_parity.py index 86e9dc8c5..342a92c0b 100644 --- a/tests/test_distribution_surface_parity.py +++ b/tests/test_distribution_surface_parity.py @@ -171,6 +171,13 @@ def paths(self) -> list[Path]: ) }, ), + Surface( + "application_diff", + ("src/agents_shipgate/cli/application_diff.py",), + # Advisory source-wiring comparison, not host drift or an engine verdict. + # No release permission, declared authority, pin or root reachability claim. + {}, + ), Surface( "capability_diff", ( @@ -181,6 +188,7 @@ def paths(self) -> list[Path]: "src/agents_shipgate/core/unread_inputs.py", "src/agents_shipgate/cli/verify/changed_inputs.py", ), + # Default host mode only; application_diff is registered separately. # No claims on purpose: this surface restates none of the engine's # answers. It projects the drift payload the engine produced, which # is why it cannot disagree with it. Its text joins only the rows the From 90527433a5f034c372085aa747eeb41c809b0831 Mon Sep 17 00:00:00 2001 From: Pengfei Hu Date: Thu, 24 Sep 2026 18:25:01 -0700 Subject: [PATCH 3/3] test: normalize colored CLI errors in review regressions --- tests/test_application_diff_review.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/test_application_diff_review.py b/tests/test_application_diff_review.py index 7ad40e104..66b70357d 100644 --- a/tests/test_application_diff_review.py +++ b/tests/test_application_diff_review.py @@ -1,6 +1,7 @@ """Review regressions: scoped uncertainty, definition identity and usable recovery.""" import json +import re import pytest from test_application_diff import SDK, commit, git, run @@ -184,10 +185,12 @@ def test_misspelled_scope_is_an_input_error(repo): def test_explicit_application_bound_requires_application_mode(repo, bound): ref = commit(repo, {"README.md": "x"}) result = CliRunner().invoke( - app, ["diff", "--workspace", str(repo), "--base", ref, "--max-python-files", bound] + app, ["diff", "--workspace", str(repo), "--base", ref, "--max-python-files", bound], + env={"FORCE_COLOR": "1"}, ) assert result.exit_code == 2 - assert "--application" in result.output and "--max-python-files" in result.output + output = re.sub(r"\x1b\[[0-?]*[ -/]*[@-~]", "", result.output) + assert "--application" in output and "--max-python-files" in output def test_redacted_comparison_identity_is_recomputable(repo, monkeypatch):