diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 0cdda6e1..d174e80c 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -10,7 +10,7 @@ "name": "agents-shipgate", "source": "./plugins/claude-code", "description": "Review PR changes to declared coding-agent permissions, MCP bindings, hooks and workflow permissions with the agents-shipgate diff CLI command (not routed by the plugin's skill, which covers check, verify and audit --host). This static host comparison needs no manifest, policy, saved baseline or skill; it names coverage limits and implies no verdict. Tool-Use Readiness release verification is a separate workflow for repositories with a declared tool surface. Requires the agents-shipgate CLI (pipx install agents-shipgate); no hook is enabled by this metadata.", - "version": "1.1.0", + "version": "1.2.0", "author": { "name": "Three Moons Lab", "url": "https://threemoonslab.com/" diff --git a/.github/release-channels.json b/.github/release-channels.json index 9bc5295e..d20ff02c 100644 --- a/.github/release-channels.json +++ b/.github/release-channels.json @@ -2,6 +2,7 @@ "schema": "shipgate.release_channels/v1", "channels": { "1.0.0": "advisory", - "1.1.0": "advisory" + "1.1.0": "advisory", + "1.2.0": "advisory" } } diff --git a/.well-known/agents-shipgate.json b/.well-known/agents-shipgate.json index ddee15b2..70e2c0b3 100644 --- a/.well-known/agents-shipgate.json +++ b/.well-known/agents-shipgate.json @@ -3,7 +3,7 @@ "name": "agents-shipgate", "display_name": "Agents Shipgate", "tagline": "The deterministic merge gate for AI-generated agent capability changes", - "version": "1.1.0", + "version": "1.2.0", "license": "Apache-2.0", "publisher": { "name": "Three Moons Lab", diff --git a/CHANGELOG.md b/CHANGELOG.md index 33e63e34..7451eb79 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,296 +1,99 @@ # Changelog -## Unreleased +## 1.2.0 - 2026-09-30 + +An advisory-channel minor release. It adds `diff --application`, which compares +the agent wiring of OpenAI Agents SDK and Google ADK applications between two +Git refs, read from source with no manifest, saved baseline or declarations, +and supplies no release verdict or merge permission. The host comparison gains +three readers — hook script bytes, coding-agent launches inside CI workflows, +and a hook's or MCP server's launch detail — names the changed inputs it does +not read, and closes three fail-open comparisons. + +Measured, as each entry records: on a pinned corpus of 48 application pull +requests, a scope derived from the change establishes the same 296 rows the +repository root does (#875); on 134 open SDK/ADK pull requests, naming what a +tool reaches left rows, statuses and exit codes unchanged, and 9 of 96 row +sides name an outbound call (#872); a root-scope comparison of +TencentCloud/CubeSandbox went from 412 s to 46 s with identical rows (#686). +The limits are worth stating plainly: `diff --application` reads OpenAI Agents +SDK and Google ADK source wiring only and compares what it observes, not what +runs; every construction, import or call it cannot follow is a named gap or +limit; and a tool's reach is read statically, claiming `read` only when every +outbound call was followed and every one reads. + +**The full reviewed prose is in +[`docs/changelog/1.2.0.md`](docs/changelog/1.2.0.md).** This section is the +release note; that file is the record. + +**Migration notes:** read every `Migration Note: 1.2.0` entry in +[`STABILITY.md`](STABILITY.md) before upgrading from `1.1.0`. Schema versions +move: runtime contract 40 → 41, verifier `0.20` → `0.21`, capability diff +`0.3` → `0.4`, and host-grants inventory, baseline and drift `0.6` → `0.7`. +`diff --application` publishes `application_comparison_schema_version` `0.2`, +its first published version. + +### Highlights + +- **`diff --application` compares agent wiring from source**, for OpenAI + Agents SDK and Google ADK, between exact base and head refs, with scoped + coverage gaps and explicit uncertain candidates. Without `--scope`, the scope + is derived from the change. (#871, #875) +- **It follows a tool where it is bound**: from another module, through a + package re-export, from an ADK local factory or a tools list built in the + agent's function. An agent it cannot read is a named gap, never "no change". + (#864, #865, #876) +- **A row names what the bound tool reaches**: the endpoint, the request + fields, the credential it sends by variable name, and which arguments the + model chooses, with effect evidence from the engine's own assessment. (#872) +- **A host comparison names the changed inputs it does not read**, so a + zero-row result is not read as covering them. (#821) +- **New host readers**: hook script bytes (#702), coding-agent launches in CI + workflows (#823), and a hook's matcher, command and timeout and an MCP + server's launch arguments (#819). +- **Three fail-open comparisons closed**: a narrowing in one settings file no + longer hides a grant in another (#858); a plugin hook the project settings + enable is no longer passed (#809); a plugin directory that cannot be compared + no longer hides the rest (#808). ### Application comparison without prior setup -- Add `diff --application` for OpenAI Agents SDK and Google ADK source-observed per-agent wiring, with exact base/head refs and independently selected scopes. No manifest, saved baseline or authored declarations are needed. The advisory `application_comparison_schema_version: "0.1"` result records before/after evidence, scoped coverage gaps and explicit uncertain candidates; it supplies no release verdict or merge permission. Includes scoped Git materialization, partial-clone recovery, and definition lookup using reader-resolved Python symbols and locations. See [application comparison](docs/application-comparison.md). (#871) -- `diff --application` without `--scope` now derives the scope from the change instead of comparing the repository root. Each changed Python file is related to the OpenAI Agents SDK and Google ADK files that import it, or that it imports, within six import hops (a package's `__init__.py` included); a changed test relates only when an agent imports it. Agent files are those that construct an agent class under any name it is imported as, subclass one, copy one with capabilities of its own, or change an agent's capabilities after construction. A changed file not related to a module that builds an agent is named, with why (no import path, a path longer than six hops, only a rewiring module, a file too large to read), and makes the result `partial` even inside a compared scope; so does a changed link to a module or directory, a changed submodule, an agent outside the compared scopes whose imports go past the bound, or a module that imports the change (directly, through a package or agent file's re-export, or through other modules), imports a builder whose agent's capabilities are not a fixed list of its own names, and builds, copies or rewires an agent itself, calls the repository's code with its own arguments, or sets its module state, unless one compared scope holds all three. A module that builds its agent through a builder's factory is an agent file. A relocated application is one comparison, whatever it holds. Each related group is compared in the outermost package that holds it. On mezgoodle/Vesta#58 a change to `backend/app/services/` is compared in `backend/app`, where its agents are, rather than in the changed directory, which holds none. Independent applications in a monorepo are separate comparisons under `comparisons`, never the root. A change that touches no supported agent is an explicit `not_established` answer naming the files considered. A bound reached is named and makes the result `partial`. `--json` records `scope_selection` (`derived` or `explicit`, the scopes and why). `--scope .` keeps the root, an explicit `--scope` always wins, and `--base-scope` now needs `--scope`. On the pinned corpus of 48 application PRs, derived scopes establish the same 296 rows the repository root does, where the changed-wiring directory establishes 18. The derived scope is narrower than the root for 20 PRs. For 7 PRs it answers that the change touches no supported agent; the root reported unrelated agents there. The median run takes 13 s, the same as the root. (#875) -- `diff --application` no longer reads another library's `Agent` or `function_tool` as the OpenAI Agents SDK's, and no longer reports an agent it stopped observing as removed. On speechmatics/speechmatics-academy#142, a LiveKit voice agent (`from livekit.agents import Agent, function_tool`) moved its tools into an `Agent` subclass that passes them through `super().__init__`. The comparison printed `compared` with four `REMOVED agent → …` rows although the same four tools were still bound. It now prints `not_established`: no supported application agent on either side. (#871 follow-up; related #580, #864, #865) - - **Framework identity.** Discovery and the SDK reader count `function_tool` and `Agent` as the SDK's only when the name is imported from the absolute `agents`/`openai_agents` package, or is not imported at all (the existing reading, unless a wildcard import from another module could supply it). The name is resolved in the scope that uses it, as Python resolves it, so an import in a sibling function never decides it, and a parameter or local assignment of that name is not the SDK's. A name imported from anywhere else — `livekit.agents`, a relative `.agents` package — is another library's. A LiveKit file is no longer an SDK candidate, and a file using both reads only the SDK's symbols. A relative import is no longer any framework's import signal. `scan` of a declared SDK source reads the same way. - - **Unobserved is not removed.** One side may observe an agent while the other side's same file still assigns or imports its name (`from factory import agent`), or passes it as `name=`, through a construction the reader does not support: an `Agent` subclass, a factory, `Agent[Context](...)` or `.clone()`. That side then records a coverage gap for the agent. The result is `partial`, and its rows are `not_established` with `candidate_change` `removed` or `added`. An agent referenced only in another agent's `handoffs` is not an observed construction. An agent whose name is gone from the file is still an established removal, and rows for other agents are unaffected. No schema changes. -- `diff --application` no longer reports `compared` with no changes while an agent it could not see gained a tool. On kkmiecik-coder/CRM#5, the quoting agent `Wycena`, built by `return Agent(name="Wycena", tools=NARZEDZIA_WYCENY)` inside a function, gained `wyslij_obraz`; the only agent read was a test double, so the result said `compared` and printed nothing. (#876) - - **Every SDK construction is observed or named.** The OpenAI Agents SDK reader read only `name = Agent(...)`. It now also reads `return Agent(...)`, `self.agent = Agent(...)`, agents inline in a list, `Agent("name")` and `Agent[Context](...)`, identified by the literal `name`. An agent assigned to a plain name keeps that name as its identity, so renaming its `name=` or moving it between scopes changes nothing; only a variable name assigned in more than one function or class body (two builders' local `agent`) gives way to each agent's literal `name`, so those are two agents, and a handoff to such a variable reaches its own. What it cannot read is a named limit on that agent, so other agents' rows stand: one identity constructed at two sites that bind different tools (identical constructions are one agent), including a Google ADK agent name; `**` or extra positional arguments; a copy (`clone`, `dataclasses.replace`, `copy.replace`) passing its own `tools`/`handoffs`/`mcp_servers`, or only `**` on a value known to be an agent; and an agent built from a same-module `Agent` subclass, including one made by `type("X", (Agent,), {})`. A construction without a literal name is a limit on its file. - - **Changes after construction.** A change to an agent's tools, handoffs or MCP servers — assigning, extending or slicing `x.tools`, a list method on it, `setattr`/`delattr` by name, or a handle `t = x.tools` that is itself changed later (a handle only read, like `len(request.tools)`, is nothing) — is read in the scope where it happens: on an agent the reader constructed (directly, through `self.agent`, or through a module function that returns it) it is a limit on that agent, and a change in place is one on every agent built from the same list object; on a value proven not to be an agent (`Settings()`, a literal, `type(...)()`, the instance's own `self.tools`) it is nothing; on anything else it is a limit on the file. In a module that is not read as an SDK source — a Google ADK source included, whose reader does not follow a change after construction either — a copy is a named limit on it, and so is a change to any object's tools once the module imports anything from the scope, or is itself a Google ADK source — through an alias, a loop, a parameter or a call's result, as in an SDK file — unless the object is plainly not an agent (a literal, or an instance of a class the scope defines that is not an `Agent` subclass); a module that imports nothing from the scope has another library's `.tools`. A module that imports an SDK `Agent` subclass from the scope, directly or through a package that re-exports it (`from app.core import *`), is limited too, never one that only spells its name in a string or imports a vendor class of the same name. - - **Google ADK agent subclasses.** A class deriving from an ADK agent class (`class Helper(LlmAgent)`, or a class deriving from that one) is a named limit where it is used: on the module that defines it when that module builds, passes or decorates it, and on every other module in the scope that imports it. An agent built from it was never read, so a tool it gained could leave the result `compared` with no rows. `scan` no longer reports the defining module's tool surface as enumerated when that module uses the class. A subclass nothing uses is not a limit. (Carried over from #880.) - - **Rows and verdicts that change.** A construction that used to be unread and silently absent is now either read (rows appear) or named (the result, and a `scan` of the same code, becomes `partial` / `insufficient_evidence` instead of complete). A Google ADK agent name constructed at two sites in one file with differing tools or handoffs is `insufficient_evidence` for `scan` too, where it could be `passed` or `review_required`; constructions binding the same definitions and handoffs, each read cleanly, are one agent and keep their rows and verdict. A copy passing no capabilities of its own, and a subclass nothing uses, are not limits. - - **No agent source, no census.** The census of copies, changes after construction and subclasses runs only when some side's discovery found an OpenAI Agents SDK or Google ADK source. Without one no agent is read, so `--scope src` of this repository stays `not_established` instead of turning `partial` over its own `artifacts.tools.append(record)`; every module is still parsed, and a parse failure is still a gap. - - **Test files are not the application.** Test files (the discovery convention, relative to the scope: `test`/`tests` directories, `test_*.py`, `*_test.py`, `conftest.py`, `test.py`, `tests.py`) are listed per side in `excluded_tests`, printed in the text output, and not read as agent sources, so a test double cannot establish a scope and a test file's own defects are not application gaps. - - **A duplicate tool definition limits one name.** A file defining one tool name twice refused the whole comparison with exit 2, and the message asked for the test file to be edited. It is now a named limit on that name in that file: no agent binds either definition, and the file's other tools and every other file are still compared. Seen on alliance-genome/agr_ai_curation#842 and usestrix/strix#1103. - - **One identity built twice keeps what both sites bind.** When one agent name is constructed at two sites (tensorflow#128063's root agent and builder), a tool both sites bind to the same callable is still a row of that agent, beside the limit that names the constructions; only a tool they bind differently is an ambiguous identity. - - Additive on the unreleased `0.1` advisory result (`excluded_tests` per side); no schema or contract bump. +- Add `diff --application` for OpenAI Agents SDK and Google ADK source-observed per-agent wiring between exact base and head refs; it needs no manifest, saved baseline or declarations and supplies no release verdict or merge permission. (#871) +- `diff --application` without `--scope` derives the scope from the change instead of comparing the repository root, and reports how it chose. (#875) +- `diff --application` no longer reads another library's `Agent` or `function_tool` as the OpenAI Agents SDK's, and no longer reports an agent it stopped observing as removed. (#871 follow-up) +- `diff --application` no longer reports `compared` with no changes while an agent it could not see gained a tool: an unread construction is a named gap. (#876) - `diff --application` no longer refuses a repository over a link or submodule its application reader never opens, and reads a Google ADK agent imported from the package root. (#871 follow-up, measured for #868) - - **The problem.** On 2026-09-25, against open third-party pull requests that edit SDK/ADK tool wiring, 15 of the first 27 runs at the default root scope exited 2 with `Git tree contains unsupported external binding`. The path named was never application source: `CLAUDE.md -> AGENTS.md` (dlt-hub/dlt#4417, jaegertracing/jaeger#9636, omnigent-ai/omnigent#6611, wandb/weave#7948), a linked `.claude/skills/…` or `.agents/skills/…` directory (asterinas#3834, vllm#57322, kagent-dev/kagent#2788), `agent/VERSION -> ../../VERSION` (TencentCloud/CubeSandbox#1508), `.pylintrc` (tensorflow#128063), and a vendored submodule (temporalio/sdk-python#1868). The root scope took the unscoped archive route, which refuses every link and gitlink. `from google.adk import Agent` was not read as an agent, so CubeSandbox#1508's `root_agent` gave `not_established` ("No supported application agents were established"). - - **Links.** The root scope now uses the scoped materializer a named scope already used, which recreates each link as a link. A link is never read through. Every link under the scope is censused, a dangling one included, and gapped only where it can hide application source: a `*.py` link whose target is not a Python input the scope already reads (an alias of one is compared at the target's path; reading it too made one agent two ambiguous ones and hid the real file's change, as a named scope did on main); a link to a directory outside the scope that holds Python; and a link resolving to nothing in the repository where the other side reads source at or beneath its path. Replacing `agent.py` with a dangling link, or a source directory with an absolute link, is therefore `not_established`, never a removal. A link Python discovery would not read changes nothing, and a scope that is itself a link is still refused. The host-configuration census, which counts every link that could conceal a *host* path, no longer adds application coverage gaps. - - **Submodules.** A gitlink under the scope is materialized as the empty directory a checkout without `--recurse-submodules` leaves; its content is never fetched. The same gitlink commit on both sides is named in `limits` as unchanged and does not make the comparison `partial`, since identical content cannot carry a change. An added, removed or moved gitlink is a coverage gap over its path on each side that has it, and over the whole scope when it is the selected scope itself, so the result is `partial`, never `compared`. A gitlink was refused (exit 2) before. - - **Google ADK.** `google.adk.Agent`, the package-root re-export of `google.adk.agents.llm_agent.Agent`, is read as an agent constructor by the ADK reader, for `from google.adk import Agent` and `import google.adk as adk` alike. CubeSandbox#1508 now establishes `cube_code_agent` and is `partial`, naming the tool it imports from a sibling module (#864), instead of `not_established`. - - **Re-measured.** At the root scope, main `a430e81a` refused CubeSandbox#1508, dlt#4417 and asterinas#3834; this change compares each, with its own limits (a Python census past `--max-python-files`, an unsupported framework, an unresolved import). No application comparison schema change; `application_comparison_schema_version` stays `"0.1"`. -- `diff --application` rows now name what a bound tool reaches: the endpoint, the request fields, the credential it sends and which arguments the model chooses. (#872, part of #868) - - **The problem.** tensorflow/tensorflow#128063 was the only one of 134 open SDK/ADK PRs whose rows were both correct and complete. Even so, each row named only a signature. `submit_pr_code_review` posts a pull-request review whose `event` can be `APPROVE`, using `GITHUB_TOKEN`; `get_pull_request_details` only reads. The existing assessment gave all three tools `write` with unknown evidence. - - **Reach.** Each side of a function-tool row carries `reach`: the outbound `requests`, `httpx`, `aiohttp` and `urllib` calls read from the tool's own code and its same-scope helpers, up to three calls deep. Every call the read cannot follow is a named limit with its location. Each call records: - - the method and the URL template (`{pr_number}`, `{env OWNER}`, `{…}`); - - literal request fields, and the literals a field is chosen from with the parameters that decide; - - `model_supplied`: which parameters flow where; - - `credential_sources`: environment variable names only, never a value; - - for GraphQL, query or mutation. - - **Effect evidence.** `effect_evidence` is the existing `assess_tool_semantics` over the tool, with the reach as one more structural source (`source_http_call`). A write or delete call supports that effect. `read` needs every call followed and every outbound call a read. Read-only method names pass only on plain data or on objects a known library returned. A decorator, a store into an unknown object, or a method on an object the read cannot name blocks `read`. - - **Secrets.** A hard-coded credential is `literal: true` and never printed. In a URL, a query value prints only when it is a short lowercase word or a number, and a token-shaped path piece is withheld. A field value prints only when it is a plain word and its name does not suggest a secret. - - **Two constructions of one ADK name.** Each row's `binding_location` names the first construction that lists the tool, and `construction_sites` lists every one. Before, every row pointed at the first construction. - - **Measured.** On the 134-PR corpus, rows, statuses and exit codes are unchanged, and run time is flat. Of 96 row sides in 19 PRs: - - 9 name outbound calls (3 PRs) and 5 name a credential; - - 7 carry structural effect evidence (3 read, 4 write); - - 65 name at least one limit, mostly database helpers, tracing and SDK clients. - - `reach`, `effect_evidence` and `construction_sites` are evidence outside the compared meaning. `application_comparison_schema_version` is `0.2`. +- `diff --application` rows name what a bound tool reaches: the endpoint, the request fields, the credential it sends and which arguments the model chooses. (#872, part of #868) - Follow a tool an agent binds from another module in the repository. (#864) - - **The problem.** jpka/attest#3 added `memory_bank.remember_firm_finding` and `memory_bank.recall_firm_memory` to an ADK agent's `tools=[...]`, with the module brought in by `from . import memory_bank`; the reader stopped at the module boundary, so `diff --application` showed no row and `scan` catalogued only the unchanged local tools. The same gap left OpenAI Agents SDK tools such as `from ..tools.shop import add_to_cart` unresolved. - - **What resolves.** For Google ADK, a name imported from a sibling module or re-exported by a package, a module-qualified `module.function`, a plain `alias = function`, and `FunctionTool(imported_function)` / `LongRunningFunctionTool(...)`, including a wrapper built in the imported module; for the OpenAI Agents SDK, a name or `module.function` that reaches a definition carrying the SDK's `@function_tool`. The tool is the definition, with its own signature, location and implementation digest, so jpka/attest#3 now shows the two memory tools as `ADDED` and leaves its four unchanged bindings, `scorer.score_answer` included, alone. A definition reached by several spellings is one tool; same-named functions in different modules stay two. - - **The boundary.** Only regular `.py` files inside the directory the read was given — the `--scope` for `diff --application`, the manifest directory for `scan` — are read, through the bounded input reader, and parsed without being imported or run. Symbolic links are not followed and a module name must match a file's exact spelling. Each application row reached through an import adds `import_path`: every module read, the line of the binding followed and that module's SHA-256; it is evidence, not compared meaning. - - **What stays unresolved, by name.** A module the scope does not contain (the scope spelled from the repository root, `svc.app.tools` with scope `svc/app`, is read inside it), a relative import above the scope, more than one matching module location, a name bound twice or only inside an `if`/`try`, a wildcard import, an import cycle, a class or other value, a parameter or other local assignment of the scope that uses the name, a name that scope binds more than once, a module attribute the same module reassigns (`tools.lookup = ...`, `setattr`), an attribute named like a step of the chain on an imported module that code running first reassigns (every module on the chain, the agent's own file included, every enclosing package's `__init__.py`, and every in-scope module those import; the defining module handing its function on does not count), and such a module that cannot be read (a link, or a missing relative module not imported under `except ImportError`), an SDK function without `@function_tool`, a symbolic link, or more than 64 modules read. The gap names the reason (`Not resolved because …`) and is scoped to the agent that lists the tool, so another agent's change in the same file is still established. One agent binding two different functions under one name is named, not resolved. The ADK unresolved-tool warning keeps its wording. No schema or contract change. - - **Identity.** One agent binding two different functions under one name binds neither, in both readers and whatever their order, and names both definitions. `import a.b` then `a.b.f` reads the submodule, as the import system does. A reference is read where it is used: a builder's own import is followed like a module-level one, `nonlocal` follows the outer function, a nested `def` that is the only one of its name is that definition, a module-level agent binds what the module binds at top level (its `def`, its import, its wrapper assignment) and never a same-named `def` or wrapper nested in a function, a module-level list's names are read at module level whatever the building function binds, and a factory's own toolset or wrapper variable is read like a module-level one. An SDK list variable is read only when the scope that binds it binds it once, to a literal list, and every use of it in the file only reads it — iterated, indexed, compared, tested, handed to a read-only builtin, standard-library reader or logger's method (each proven by its binding), to an agent's (or a copy's) own `tools=`, or to a function whose every use of that parameter is such a read. Spreading it (`[*TOOLS, x]`, `f(*TOOLS)`) or testing it (`TOOLS or []` in a condition) is a read. A method call on it, `+=`, a second name (including through `x or y`), a tuple, a return, `*args`, `globals()`, `sys.modules` or importing the module by `__name__` makes it dynamic. For `scan`, a definition that an import reaches and another configured source also reads (spelling the module's path the same way) is one catalog tool, and the binding reaches it through the exact definition the reader resolved; a `{tool: …}` selector for it is not ambiguous, and the dropped copy's guard evidence goes with it. A source an inventory completes keeps its own observation, so when that source imports a definition another source also reads, the catalog holds both and a selector for it is ambiguous. When what the module binds is not established (a name rebound, or bound only inside an `if`) and the ADK reader falls back to a same-named `def` or wrapper, that binding is named but never established: in a comparison its row is `not_established` on whichever side it is present, added, removed or changed, and so is the row of every tool the module's bindings of that name could give the agent instead, each followed to its definition (a function from a module outside the scope by its imported name); every one of the agent's rows is when one of those cannot be followed or a wildcard import could bind the name; for `scan` it stays the medium-confidence shadowed definition it was. `x = FunctionTool(func=x)` right after `def x` wraps that `def`, and is not a guess. Code that runs before the name is used but is not read — a relative import above the scope, an absolute import of a module the repository holds outside the scope (read from the compared commit's tree, or the checkout for `scan`, at the root, under `src/`, or under any directory between the root and the scope, modules the interpreter preloads, a standard-library name at a root that is a regular package, and the scope's own package aside; a linked or submodule entry counts, and a namespace directory only when it holds the named submodule, so SDK apps under `agents/` still import the SDK), a relative module no file provides (a generated `*_pb2` or `_version` aside), or a package `__getattr__` that is not the lazy-submodule idiom (or that the package could subvert through `sys.modules`, `globals()`, `__name__` or a patched `importlib`) — keeps the tool named. The `__init__.py` of every package above the scope, and what each imports (every package on the way to it and each submodule named), are read too; there a reassignment counts unless it sets one attribute of another module file outside the scope, a module of the scope they import is read like the chain's own, a change to `__path__` is a caveat (`pkgutil.extend_path` aside), their files are read in batches, and what they import in turn, or import by name at run time, is not followed. `sys.modules` and `globals()` are read by allow-list: a store whose key names a module on the chain, a package above one or the framework's own modules is a named stop, a module rebinding its own name through them or its own module object (however spelled, including `__dict__` and `vars()` stores) is a reassignment, reads (a comparison, iteration, a spread, `pkgutil.iter_modules(__path__)`, `get_type_hints(globalns=globals())`) are nothing, and any other use (`mods = sys.modules`, `|=`, a computed key or `setattr`, the module object anywhere but an attribute, a plain alias or a reader, `__dict__.update`, a frame's `f_globals`, `builtins.globals` under another name) or a change to `__path__` is a caveat: its row is `not_established` with the reason, including through a `FunctionTool` wrapper, and for `scan` the ADK module stays at medium. Attest's lazy loaders (`importlib.import_module(f".{name}", __name__)`, `if name == "x": from . import x`) stay established; `from . import other as x` is not the idiom. When a definition another source reads under its own name is the one the reader resolved, the binding reaches it through its exact locator, never a same-named local definition. The module-binding walk and the SDK list reader are linear in the tree, and a chain of thousands of attributes no longer crashes the run. - - Read a Google ADK tool a local factory builds, and a tools list built in the agent's function. (#865) - - **The problem.** visulate/visulate-for-oracle#526 bound `save_memory_tool = create_save_memory_tool()` and `read_memory_tool = create_read_memory_tool()` in its root agent's builder; each factory returns `FunctionTool` around a nested async function, and the reader stopped at the local assignment, so the two new memory tools were unresolved like the agent's nine others. MuhammadVT/smart-assignment#46 built `tools = [...]` in the agent's function and appended a triage tool under `if triage_enabled:`; the reader read the whole list as a dynamic tools expression, losing its unconditional tools. - - **Factories.** A factory call — `tool = create_tool()` bound once and unconditionally in the agent's function or at a module's top level, or `tools=[create_tool()]` — is followed, as syntax and never run, to the factory's one unconditional `return`: `FunctionTool(inner)` / `LongRunningFunctionTool(inner)`, a plain function, a name bound once to one of those, or another factory's call, up to four factories deep. The tool is the function wrapped — one nested in the factory and defined before the `return`, or one the factory's module binds — with its own signature and location; `import_path` records the factory as a step, and the factory's code and the values each of the agent's calls gives its parameters (a literal, or a name or module attribute bound once to one; defaults filled in), which its closure holds, are part of its implementation digest, so `make_sql_tool(readonly=False)` in place of `readonly=True` is a changed tool and the same call spelled otherwise is not; a value this read cannot name is a limit on that tool naming the argument, and a `not_established` row only when the calling module changed. - - **What stays named, with why.** A factory that returns from more than one place or only under a condition, calls itself on the way, is decorated, a generator or `async`; a wrapped function that is decorated, a parameter, defined more than once in its file (a tool is known by its name there), or changed or handed on in the factory (`inner.__name__ = ...`, `setattr`, a helper), a module function included; a tool changed or handed on where it is bound (`tool.name = ...`, `rename(tool)`); and a factory the repository holds outside the read scope, named as such. Any other call — a third-party package, a class, a name bound twice or through a wildcard — keeps the answer it had. On visulate#526 with `--scope ai-agent`, the memory tools are read, and the remote-delegate factory, which renames the function it returns per call, leaves the nine delegates named with that reason — so the two memory tools are `not_established` candidate additions, not a complete eleven-tool surface; they were not rows before. - - **A tools list.** `tools = [a, b]` (or a tuple) bound once in the agent's function, with `tools.append(c)`, `.extend([...])`, `.insert(i, c)` or `tools += [...]` statements, is read member by member up to the statement that builds the agent, which copies the list, so an addition after it — or after an early `return` of another agent — is not that agent's. An addition before it under a condition or in a loop is named on the agent, which is then not complete, and is never read as bound. Any other use of the list — handed to a call, aliased, returned, changed another way, read by a nested function, a starred member — and a module-level list keep the dynamic tools expression. On smart-assignment#46 the unconditional tools are read and the triage tool is named as conditional. -### Changes +- Read a materialized Git tree's blobs through a few `git cat-file --batch` processes: a root-scope `diff --application` of TencentCloud/CubeSandbox went from 412 s to 46 s with identical rows. (#686) -- Compare exact repository script bytes for supported, selected Claude Code and Codex hook executable references. A script-only edit produces an attributable hook row without asserting a permission expansion; its `why` names the direction as not established (#820), so the Claude Code Stop hook lists it apart from widenings rather than staying silent; `check` routes the dependency for review from either compared revision. The documented shell spellings resolve (`"$CLAUDE_PROJECT_DIR"/path`, `"${CLAUDE_PROJECT_DIR}"/path`, the variable and path quoted together, or unquoted with a plain path, and `CLAUDE_PLUGIN_ROOT` alike in a plugin's hooks). A script this entry cannot read withholds only its own bytes: an unchanged shared limit, or a script absent from both sides, is an `unchanged_limits` entry, and any other makes the comparison `partial` with every grant still compared. A script whose working-tree bytes differ only by a checkout's line-ending conversion is `unchanged_not_proven`, not a row. A selected hook whose script is not resolved (an interpreter wrapper, a relative path, a conditional expansion, a compound command or a malformed group) is named as a `script_not_resolved` coverage item while the change could touch it, never a row. A pre-#702 baseline is incomparable only for a hook that now binds a script. `check` and a provided diff (`check --diff`) leave out a selected script the change does not touch, so an untouched missing, ignored or linked script changes no `check` decision; a touched one is routed for review, and compared from the diff where the diff also touches its declaration. Verification binds ignored inputs and missing-path observations, and refuses current authority for unsafe reads. Mode-only changes and recursive dependencies are not compared. (#702) -- Note a literal, unconditional inline `PreToolUse` allow on a broad Claude Code hook matcher. Diff, verifier host comparison and PR comments share the note; script and unsupported command behavior remains a named inventory limit. Direction, severity, widening signals and control decisions are unchanged. See the [bounded grammar](docs/engineering/inline-hook-allow-notes.md). (#826) -- Add factual mutable-source notes to supported MCP launcher changes, including established pinned-to-mutable transitions. Keep package text within existing redaction rules; do not change direction, severity or control decisions (#825). -- Add conditional review guidance for supported Claude Code shell-permission changes: a concrete intent question, human choices and a declaration comparison target, shared by CLI, maintained JSON and advisory PR output. Conflicting, redacted or incomplete evidence withholds specific choices. Existing control permissions, rows and release decisions are unchanged. See the [case mapping](docs/engineering/permission-review-guidance.md). (#839) - -- Correct false widening markers on hook/MCP edits, disabled plugins, more restrictive permission modes and enabled sandboxes. Explicit enablement (including a Codex app, enabled by default), added or re-pointed Claude Code marketplaces, less restrictive modes, documented sandbox loosenings (disabling it, `excludedCommands`, `autoAllowBashIfSandboxed`, `enableWeakerNestedSandbox`, `allowedDomains`, Codex `writable_roots` and `web_search: live`) and added loaded hooks/MCP servers still expand, including a handler added to an event that already had one. An edit whose direction is not established is a row whose `why` ends `authority direction is unknown`, and the Claude Code Stop hook names it under its own heading instead of staying silent. Setting replacements use their full comparison context so another setting's expansion cannot mark a tightening; an unknown or ambiguous predecessor is read as a new declaration, so it cannot hide `bypassPermissions`. Claude Code plugin enablement no longer coerces malformed values to booleans. Existing check decisions and severities remain unchanged. See the [migration note](STABILITY.md#grant-direction-820) and [before/after direction benchmark](benchmark/host-config/direction-replay-820.md). (#820) -- A changed Claude Code `git push` prefix allow now names documented command forms that the narrower deny prefixes in the same source do not cover, including a flag placed after the remote (`git push origin main --force`). The shared CLI/JSON/PR explanation considers unchanged head denies too, declines unsupported shapes and preserves all direction, severity and gate decisions; `check`, whose rows redact rule arguments, omits it. It does not claim those commands will execute without prompting. (#829) -- Rate documented arbitrary-code Bash launcher prefixes with trailing wildcards as critical/admin, and any Bash allow rule the lattice decides is wider than one (`Bash(python3 *)`, `Bash(npx*)`), so widening a rule cannot lower its rating; route newly granted forms through the existing wildcard block check, and count them in the `audit --host` Markdown warning. Show them in check evidence in table text only; exact commands and containment comparisons stay unchanged. `check` no longer reports respelling a Bash rule (`Bash(npx:*)` to `Bash(npx *)`) as a new grant (#824). -- Skill frontmatter `metadata` no longer refuses the entire host comparison. A map's boolean, numeric, list and nested values are preserved, and a `metadata` that is not a map, which Claude Code drops, is digested as written, as an undocumented key is (#730). Claude Code documents a free-form map; this static structure reader does not validate portability to hosts that require string values. Metadata changes remain in the structure digest, and malformed frontmatter, ambiguous YAML and a metadata key that is not a string, at any depth, still refuse. A permission widening beside such metadata is now visible in text and JSON, with matching coverage. See the [migration note](STABILITY.md#skill-metadata-848). (#848) -- Version-based Action installs log the existing installed engine content - digest, including host-only runs. It can be compared with a local verifier's - `engine_distribution_sha256`; it is not the wheel ZIP hash. An engine that - cannot compute it, such as any release before `1.0.0`, logs a warning and - the install continues. The explicit wheel-and-hash install route is - unchanged. (#855) -- The PR-comment fallback uses separated Markdown paragraphs and headings, - so fork reviews render correctly in the job summary after a 403 or 404. - Successful comment updates and error handling are unchanged. (#856) -- Host-only Action summaries no longer invent empty scan status or zero scan - severity counts. Artifact links name only existing files, and annotation - metadata omits `source_report` when no scan report was produced. (#854) -- Workflow comparison values label their aggregate access (for example, - `access: write`) so it is not mistaken for an individual token scope. - Direction, severity, explanation and widening signals are unchanged. (#859) -- Host-only verifier headlines and `control.reason` label their raw count as - rows, so a two-row replacement no longer contradicts the review summary - that correctly calls it one change. Partial comparisons use the same - wording; decisions and control routes are unchanged. (#857) -- An unrelated permission rule for another tool no longer splits a decided - replacement into separate review changes. Adding `Read(src/**)` beside - `Bash(npm test *)` → `Bash(npm *)` preserves the paired widening on the - shared diff, verifier and check routes. Multiple candidates for the same - tool or MCP server remain unpaired. MCP server replacements and - narrow-while-denying edits also retain their direction beside unrelated - tool changes. Unpaired removals no longer claim a permission loss when - an added allow rule in the same source decidedly covers them and no deny - or ask rule for that tool arrives in the same edit. (#858) -- Fix a fail-open comparison where a narrowing in `settings.json` suppressed - the expansion signal and ⚠ for the same rule newly allowed in - `settings.local.json`. Suppression now stays within its source, restoring - that independent grant in `expansion_signals` and review widening counts. (#858) - -- Move the published-release pins, examples and adoption prompts to `v1.1.0` (contract 40) now that it is published, re-capture the README and quickstart `diff` answers from the published `1.1.0`, and re-measure the pilot ledger's Route H dry run on it. No schema or contract change. (#778) -- A host comparison names the changed inputs it does not read, so a zero-row result is not read as covering them. (#821; slice 2 of #812) - - **The problem.** A pull request that added a Cursor plugin's `mcp.json`, removed a `beforeShellExecution` guard from `.cursor/hooks.json`, gave a dotfiles package's `claude/.claude/settings.json` `Bash(*)`, or moved a marketplace plugin's pinned `sha` printed `No static host-grant changes detected`, as a docs-only change does. Re-running a 23-PR public corpus after #812 found 11 of 23 pull requests were such coverage gaps: 0 of the 9 comparable zero-row results named the changed relevant file, and 4 of them named a file the pull request did not touch while omitting the one it did. - - **What is named.** `diff`, `verify` and the manifest-free PR comment list, under `What this run established`, each path in the comparison's own changed-file set that a bounded, documented candidate rule recognises and no reader of this entry read: `mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, and an external marketplace plugin source — `plugins/demo/mcp.json (cursor): added, not read by this entry: MCP configuration in a plugin directory; no row, and loading is not established`. An external source names what it now points at, redacted, and is never fetched. The block's first line says the list includes them. Ordinary documentation, an unrelated `*.json` and an unchanged candidate name nothing. - - **What it is not.** Never a row, a widening, a `check` violation or a claim that a host loads the file. Nothing is fetched or run, only plugin manifests and marketplaces are read, and at most 32 candidates are examined; the rest, and any whose rule needed a file that was not read or did not parse, are counted as not examined, on a line that names both causes. The rules are listed in `docs/host-boundary-support.md` under *Changed inputs named but not read*. - - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; host-grants does not move for it (#819 and #823, below, move it to `0.7` in the same contract), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. - - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. -- A hook row now names what changed in the hook, and an MCP row names a change to the server's launch arguments. Before, `diff`, `verify`, the manifest-free PR comment and `check` printed `PostToolUse → PostToolUse` whether the edit was to the hook's matcher, its command or its timeout, and an MCP server whose version pin moved from `example-mcp-server@1.2.3` to `@latest` read `docs: no difference in the command name npx, env key names or header key names; the change is in a detail this output does not show, such as the command's path or arguments`: the grants carried none of it, and only `config_sha256` saw the edit. On five of 23 public pull requests measured on 2026-09-15, the hook rows showed only event names. (#819, slice 2 of #795; direction is #820, an unpinned-launch note #825) - - **The rows a reviewer reads:** `PostToolUse: matcher Edit → Edit|Write|Bash`, `PostToolUse: command changed (lint.sh sha256:d075f5f4772e → curl sha256:a510416cbecc)`, `PostToolUse: timeout 10 → 600` and `docs: package example-mcp-server@1.2.3 → example-mcp-server@latest`, in `diff`, `verify` text, the PR comment and `check` text, and in `review.changes[].change` in `diff --json` and `verifier.json`; any other launch argument edit reads `launch arguments changed (sha256:… → sha256:…)`, a digest printed as its first twelve hex digits. A timeout written as text prints quoted, so it never reads as a number or a boolean: `timeout 5 → "5"`, `timeout true → "true"`. With several handlers under one event the entry names which one (`handler 2 timeout 5 → 50`), and an added or removed handler is listed as such. The same published handlers in another order read `the published handlers in a different order; a detail this output does not show may also differ, such as …`, never that they are the same handlers. An added or removed hook names its handlers, `SessionEnd (command cleanup.sh sha256:18d2c7ec39bc)`, and an added MCP server its package. When none of the published fields differ, the entry says the change is in a detail it does not show — another hook setting such as `async`, or a redacted or shortened matcher or timeout; for an MCP server, the command's path or another setting such as `cwd` — instead of repeating the same values. The row's direction, severity, `why` and loading basis are unchanged: plugin-selected (#714) and Codex hooks read as before, and no entry claims a direction (#820), runtime loading or what a command does. - - **No command or argument text is published, on any surface.** A hook command is published as its executable's name — the last path segment of its first word, only when that is a plain token (`[A-Za-z0-9._+-]`, at most 80 characters) that no redaction rule rewrites, not a shell reserved word such as `if` and not part of a URL (a first word holding `://`), otherwise `` — and the SHA-256 of the whole command as `config_sha256`'s input holds it. An MCP server's arguments are published as at most one package specification of a strict shape (npm `name@version` or `@scope/name@version` with a version of two or three numeric parts, a `^`/`~` range on one or a common dist-tag; PyPI `name==version` with a version of two or more parts; an OCI image reference with a path and a tag or `sha256` digest) that no redaction rule rewrites and that follows no flag but a package runner's own (`-y`, `--from`, `--rm` …), and the SHA-256 of every argument with the package replaced by a marker and its position digested beside them. The matcher passes the #802 published-label redaction and is cut at 120 characters, and a matcher longer than 1,024 characters as `config_sha256`'s input holds it is ``, never redacted or cut; a timeout is the number or boolean as declared, an over-80-digit integer's cut digits, a plain-token string, or `` for anything else, a non-finite float among them. Earlier drafts published redacted command words, and each of four review cycles found a credential the redaction rules missed inside free-form shell text; publishing none of it closes the class. - - **Host-grants `0.7`:** a hook grant adds `handlers[]` — each handler's group `matcher`, its `command` `{executable, sha256}` and its `timeout` — and `omitted_handlers` (at most sixteen handlers are listed); an MCP server grant adds `package` and `args_sha256`, both `null` when no `args` is declared. A declaration outside the documented shape (a list of matcher groups whose `hooks` are objects, each `command` a string where one is declared) publishes `handlers: null`, and its row says the matcher, command and timeout are not shown because `the declaration is not a list of matcher groups whose hooks are objects and whose commands are strings`; when only one side is outside it, the row names that side (`base` or `head`) and lists the other side's handlers. Runtime contract v41. - - **Saved baselines hold none of it:** `audit --host --save-baseline` writes each hook and MCP grant without `handlers`, `package` or `args_sha256`, in either scope, so nothing read from `~/.claude/settings.json`, `~/.cursor/mcp.json`, managed settings or a git-ignored `.claude/settings.local.json` reaches the committed baseline. A saved `0.7` baseline's grants are the ones a `0.6` baseline holds; no comparison, row or digest read them, and `inventory_sha256` is unchanged. - - **The PR comment keeps every line 1.1.0 kept:** its 6,000-character bound cuts at the first line that does not fit, so long entries could hide every row after them, the coverage block, the change count, the review question, the reproduction and the advisory. The lines 1.1.0 printed now get their room first, the coverage block included, and the entries only what is left: an entry is printed whole when the whole comment fits; otherwise every longer entry is cut to the widest length of at least 60 characters at which it does, ending in `…`; and where not even that fits, entries are printed in their shortest form, longest first: a field-level difference cut after its name (`PreToolUse: …`), an added or removed grant as its row (`(absent) → PreToolUse`). No entry in that form is longer than the one 1.1.0 printed, and a permission rule's entry is never shortened. One line after the rows, not one per entry, says entries were shortened and that `verifier.json` holds each whole. On a pull request that moves two hook scripts under 14 events, 1.1.0's comment held every row, the coverage block, the review question, the reproduction and the advisory, and so does this one, within the same 6,000 characters. `verifier.json` and the other routes keep every entry whole. A comment with no readiness report now points to `verifier.json` when it omits detail, not to a `report.md` that route does not write. - - **Unchanged:** grant equality and every inventory digest leave the new members out, so a change is a row exactly when it was one before, through `config_sha256`; a value the digest's own input redacts (after `--token`, `--api-key` or `--password`, a `--password=…` value, an `X-Api-Key:` header value, a URL's path) moves no published digest, so a change confined to it is no row, as on 1.1.0; every row value, the row count, `check`'s boundary result and the control envelope's `capability_rows` publish what they did; verifier `0.21` and capability diff `0.4` do not move for it; the host-config and cold-start benchmark replays reproduce their run-of-record scores. The digest's credential-assignment rule gained a lookahead that removes its quadratic time on a long run of name characters and matches exactly what it matched, so every `config_sha256` is unchanged. Re-running `diff --json` on the 80 vendored benchmark cases with the prepared `1.1.0` commit and this tree on 2026-09-23 gave byte-identical rows on all 80; 23 entries on 22 cases changed, and each of the 7 changed hook or MCP entries (six repositories; one is vendored in both benchmarks) that read `PreToolUse → PreToolUse` or `no difference in the command name …` now names the field that changed, such as `mcp-outline: package mcp-outline==1.10.0 → mcp-outline==1.10.1` or `PreToolUse: handler 2 timeout 30 → 120`. - - **Compatibility:** a `0.6` baseline stays comparable with no new row or reason, and `audit --host --save-baseline` may now replace it; an older one is still refused, as before. Validators pinned to the `0.6` schemas reject a `0.7` inventory, baseline or drift payload; the `0.6` files stay published. See the [migration note](STABILITY.md#hook-mcp-detail-fields-819). - -- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh` (a `shell:` template only when it runs the script alone, never `bash -c '…' {0}`), and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row, and the workflow's line in `What this run established` names what its grant does not read instead of env values and `apiKeyHelper`. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, though never while that job keeps any unread step of that agent or a launch of it with an unread input the rule is read from, nor while any other job but the receiving one holds more of them than before, so renaming a job while quoting its launch, as another job adds that launch plainly, is a widening; one the job's unread step or unread input may already have met is named, not claimed; and an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and the unreleased runtime contract 41 (#821) is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: Unreleased` entry in [`STABILITY.md`](STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) +### Host comparison and the Action +- Compare exact repository script bytes for supported Claude Code and Codex hook executable references: a script-only edit produces an attributable hook row. (#702) +- A host comparison names the changed inputs it does not read, so a zero-row result is not read as covering them. (#821; slice 2 of #812) +- A hook row names what changed in the hook, and an MCP row names a change to the server's launch arguments. (#819, slice 2 of #795) +- Read how a coding agent is launched inside a CI workflow: a documented agent action's permission inputs, the flags of a plain `claude -p` or `codex exec`, and each `actions/checkout` ref. (#823) - A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) - - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. - - **What changes.** When every blocking limit that refused a comparison is a plugin-reference limit bounded by its plugin directory — a reference is followed only inside it, so that is all it can hide — and nothing outside the directory depends on it, the comparison is `partial`: the directory is left uncompared on both sides and named, and the rows outside it are published. `diff` opens with `Partial comparison against main (…) -> working tree: head_inventory_incomplete` and `Not compared: plugins/demo, a plugin directory this entry could not read completely, …` before any row; `verify` and the PR comment open with `Host capability comparison partial: …` and the same line. A partial result with no row says it is not a no-change answer and never prints `No static host-grant changes detected`. - - **Where it still refuses.** Any other limit that is not an unchanged one (an unreadable settings file, an instruction file whose structure could not be established, a link #700 does not read through), a reference that leaves its plugin, a plugin at the repository root or holding `.claude/settings.json`, a marketplace elsewhere declaring inline hooks for that plugin, or nothing read outside the directory: `incomparable`, no row, exactly as before. A hook file another plugin also selects is withheld with the directory. - - **JSON.** `comparison_status` adds `partial`, with the same `incomparable_reasons` the refusal would have named, and `rows`, `review` and `unchanged_limits` for what was compared. The reserved `coverage.items[].scope` now names the withheld directory on each `blocking_limit` item of a partial comparison, and is `null` everywhere else. In a partial comparison a changed project settings file with no row of its own is `changed_without_rows`, never `changed_without_grant_change`: the hooks whose loading basis it decides are not all compared. Verifier `0.21`, capability diff `0.4` and contract 41 are extended in place; a `0.20` verifier that claims a partial comparison or a scope is refused. - - **What does not move.** Every decision and route. `verify`'s control state, permissions, next action and exit code are the incomplete comparison's; the control envelope's `capability_rows` projects a partial comparison as `incomparable` with no rows; `check` names no scope, so it refuses its comparison as before, and the #808 fixture still gives `require_review` with `HOST-PERMISSION-DENY-REMOVED`; the Stop hook still says to treat the change as unreviewed. `audit --host`, digests, baselines and drift payloads are unchanged (host-grants stays `0.6`), and the host-config and cold-start benchmark replays reproduce their run-of-record scores. See the `STABILITY.md` migration note. - -- A Claude Code setting that disables prompts or approves project MCP servers - now carries one rating on every surface that names it. One table in the - engine rates each value, with its documented basis, and the `audit --host` - grant, the `diff`, `verify` and `check` rows, `check`'s violation and - `verify`'s finding all read it. `enableAllProjectMcpServers: true` and - `skipDangerousModePermissionPrompt: true` are no longer recorded as keys that - "could not be parsed" at `medium`: like `defaultMode: bypassPermissions`, - whose grant and row now read `critical`, they are `critical` everywhere and - `check` blocks with `SHIP-HOST-BOUNDARY-PERMISSION-WILDCARD-ALLOW`. - `defaultMode: dontAsk`, which Claude Code documents as denying whatever no - allow rule permits, is `medium` everywhere and no longer blocks; it is - reviewed through `SHIP-HOST-BOUNDARY-PERMISSION-ALLOW-EXPANDED`. `acceptEdits` - and `auto` grants and rows move from `medium` to the `high` `check` already - gave them, and `plan`, `default` and the other modelled settings move in - `check` to the `medium` their rows show. Rows name the setting and value - (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`, - `approval_policy: never`) instead of `True` or `dontAsk` alone, and a Claude - Code setting's row says what the value does. `enabledMcpjsonServers` is read - as one `high` grant and row per approved server, and an entry that names no - server is kept whole at `high` rather than dropped; `disabledMcpjsonServers` - stays unread. `check` evidence carries a setting's value as its grant - publishes it, with credentials redacted and at most 200 characters. A value - moved between `permissions` and the top level is one the change sets, so a - top-level `defaultMode: bypassPermissions` moved into `permissions`, where - Claude Code reads it, still blocks. No schema, contract or check id moves; - which check id fires for these values, and their decisions, do. See the - `STABILITY.md` migration note. (#827) - -- `check` and `verify` no longer pass a change to a Claude Code plugin's hook - that the repository's own project settings enable just because the plugin - keeps the hook outside the registry paths. 1.1.0 published such a hook as - `execute`/`high` and its change as a `widened`, expanding row, yet `check` - gave `allow` with `merge` permitted beside it, and a `verify` with a manifest - gave `passed` / `mergeable`, while the same hook at - `.claude/hooks/hooks.json` got `require_review` and `review_required`. - - **Routed.** A changed hook file an enabled plugin selects, such as - `plugins/demo/cfg/hooks.json`, and a plugin manifest or marketplace whose - inline hooks an enabled plugin loads, now reach the existing - protected-surface review: `SHIP-AGENT-BOUNDARY-PROTECTED-SURFACE-UNCLASSIFIED` - (evidence `hook_loading_basis: project_enabled_plugin`). In `check` that is - `require_review` with `merge` withheld. In a `verify` with a manifest it is - a finding that moves the release gate to `review_required` and the merge - verdict to `human_review_required`, and the PR comment says so. The route - needs both the file's name, one the plugin hook reader opens, and the - reader's finding, in the base or the head, that an enabled plugin loads - hooks from it. `check` and `verify` read both sides the same way, so a hook - file deleted with its reference is routed by both, and a hook a plugin only - selects stays unrouted, as #714 made it. - - **Unread.** A changed hook file an enabled plugin selects under a name the - static reader does not follow (`cfg/lifecycle.json`, not `hooks.json` or - `-hooks.json`), or inside a directory the walk skips, holds hooks the - host loads and nothing read. It is now incomplete input, - `SHIP-AGENT-BOUNDARY-INPUT-INCOMPLETE`, in both: `human_review_required`, - where 1.1.0 gave `allow` with complete input. - - **Broken heads.** A head that leaves a routed hook file or plugin manifest - unreadable is incomplete input, as at a registry path. A marketplace the - head makes unparseable is different, because a marketplace limit never - blocks: the route still gives `require_review`, input stays `complete`, and - the comparison shows the marketplace's inline hooks as `removed`. - - **Unchanged.** Rows, the host inventory, `diff` and the trigger catalog. - A change only to a selector, a manifest's `hooks` reference or a - marketplace entry, that makes an enabled plugin load an existing, - unchanged hook file is still `allow` beside that file's `added` or - `widened` row. `check` gives the same when the selected file sits at a - registry path, so that is a separate limit. No schema, contract or check - id moves; what moves is when two existing check ids fire. See the - `STABILITY.md` migration note. (#809) - -- `verify --preview` in a repository with a `shipgate.yaml` no longer sends - the caller to a human for missing evidence. `verify --preview --json` there - answered `agent_action_required` with the exact `verify` command, while - `verify --preview --format control` answered `human_review_required` with - "The recorded source and dependency inputs are no longer current: input - directory capture is unavailable; re-run verification", and every - `agent control` refresh exited `4` with `workspace_unverifiable`. Re-running - reproduced it, with the default, an in-repository and a sibling `--out` - alike, on `1.0.0` and `1.1.0`. - - **The cause.** A manifest lets the preview record the verification plan a - `verify` would run, and the preview's pointer bound that plan as its - currency evidence. A preview runs no adapter, so none of the plan's inputs - was captured and the plan has no input-directory census, which every - reader refuses. - - **Now.** A preview's pointer binds its verifier route and never the plan, - and is read against the working tree it was run on, exactly as a - manifest-free preview's already was: `--format control` prints the - `--json` state and command, and `agent control` returns them, with the same - `current_control_id`, until a tracked edit, a new untracked file or a - removal refuses it as `workspace_changed`. Restoring the tree makes it - current again. The plan is still written, and `verify-run.json` still - embeds it. - - **Still refused.** A pointer that binds a plan without its census, such - as one a `1.1.0` configured preview left in a reports directory, is - `workspace_unverifiable` as before, with a `verify` command as its next - action; re-running the preview replaces it. The plan gains no census, so - `verification worker` still refuses to replay it. Nothing else reads a - missing census as complete. - - **A tree that cannot be read stays refused.** Under Git configuration the - worktree readers refuse (#813), the overlay a plan-less pointer binds - cannot be read, and such a pointer declared no snapshot at all, so the - refresh compared HEAD alone: a manifest-free preview stayed current over - any later edit, and a configured one would have too once it stopped - binding its plan. Every pointer published inside a repository without a - plan — a preview, or a `verify` that stopped before building one at a - `--config` or `--head` that does not exist — now declares the worktree - snapshot, so it is refused as `workspace_unverifiable` with the cause - first and a `review` next action, like every other pointer there, and the - run's own `--format control` says `human_review_required`, manifest or - not. `--json` is unchanged. - - **One refusal names the tree.** A preview refused as `workspace_changed` - now names up to three paths that differ from HEAD, redacted, instead of - saying the paths it was read from no longer had their content, which was - false for a tracked edit or a new file after a preview of a clean tree. - - No schema, contract, member, error kind, refusal code or exit code moves. - See the `STABILITY.md` migration note. (#807) - -- An unchanged instruction file or plugin manifest reached through an in-tree - link no longer refuses the whole host comparison. A `SKILL.md` whose - `metadata` has a key that is not a string, which this entry cannot resolve, - is named as an unchanged limit when the change leaves it alone (#721). Read - through `.claude/skills -> ../.agents/skills`, a per-skill link or a file - link, the same untouched file made `diff`, `verify` and the manifest-free PR - comment print `Cannot compare against HEAD~1: base_inventory_incomplete; - head_inventory_incomplete` and no row, hiding a removed `deny` rule beside it. - - **The cause.** The unchanged proof asked whether the path the link is read - under, `.claude/skills/review/SKILL.md`, was one regular file in Git, and a - path through a link never is. - - **Now.** The proof follows the link as the reader does (#700), from Git - tree entries, and holds only when both are unchanged: the link, a link at - the same path with the same text at each link on the way, and the file it - lands on, the same blob at the same in-tree path. A working-tree head is - read without following any link, each link by its own text and the file by - its unfiltered hash. The limit is then named and the rest compared, exactly - as when the skill sits at its own path: the issue's reproduction shows - `⚠ low removed claude-code .claude/settings.json`, - `deny: Bash(curl *) → gone` for the directory link and the per-skill link - alike. - - **Still refused.** A skill added or edited behind the link; the link - retargeted, even to an identical copy, or its text rewritten to land on the - same file; a link replaced by a directory holding the same bytes, or the - reverse; and every link the reader does not read through (dangling, - looping, absolute, escaping, past eight links, or inside a linked - directory), which stays `unreadable` and is never an unchanged limit. - Metadata is not coerced: the key `1` is not read as the string `"1"`, so - it stays an `unsupported` structure, and the value `internal: true` is read - as written, with no limit (#811, #848). - - **A plugin manifest behind a link.** The same proof covers a - plugin-reference limit, such as a `parse_failed` - `plugins/demo/.claude-plugin/plugin.json` that is a file link to - `../../../vendor/plugin.json`. Unchanged on both sides, it is now named in - `unchanged_limits` like one at its own path, so a comparison that was - `partial` only because of it (#808, `Not compared: plugins/demo`) is - `comparable` on `diff`, `verify` and the PR comment, with the directory - compared; `1.1.0` refused it. Edited behind the link, it stays `partial`. - - **`check`.** Its boundary result still cannot name a limit, so it does - what it does for the same limit at its own path. An instruction file's - limit, which it cannot leave out, now refuses its comparison with - `unchanged_limits_not_representable` instead of - `base_inventory_incomplete` / `head_inventory_incomplete`, and its rows - stay empty. A plugin-reference limit both sides share on an unchanged - source it leaves out, as it has since #714, so behind a link it now - compares and publishes the rows it finds where it refused with no row: the - plugin manifest above gives `comparable` with the removed `deny` row, in - the boundary result and in the control envelope's `capability_rows`. - Edited behind the link, the limit is kept and `check` refuses as before. - Its decision, violations and control state do not move. - - No schema, contract, member, reason code or check id is added, and - host-grants stays `0.6`. See the `STABILITY.md` migration note. (#822) - -- Read a materialized Git tree's blobs through a few `git cat-file --batch` processes instead of one `git cat-file blob` per file. `diff --application --scope .` materializes the whole tree on both sides, so its run time grew with the file count: measured on 2026-09-25 with #877's root-scope reach, it took 412 s on TencentCloud/CubeSandbox (3,896 files) and 203 s on dlt-hub/dlt (2,042 files), and now takes 46 s and 45 s with identical rows; materializing one side of CubeSandbox went from about 180 s to under 5 s. One `cat-file --batch-check` first types and sizes every object, so a missing object or one that is not a blob refuses with a configuration error before any content is read; the content is then read in batches of at most 64 MiB, so a large tree is never held in memory at once. Every blob is still hashed against its tree entry's object ID before it is written, and the path, containment, link-order and final digest checks are unchanged. The link texts read to resolve in-tree link targets are batched the same way. No output, schema or contract changes. (#686 cost class) +- `check` and `verify` no longer pass a change to a Claude Code plugin's hook that the repository's project settings enable. (#809) +- Fix a fail-open comparison where a narrowing in `settings.json` suppressed the expansion signal for the same rule newly allowed in `settings.local.json`. (#858) +- An unrelated permission rule for another tool no longer splits a decided replacement into separate review changes. (#858) +- Correct false widening markers on hook and MCP edits, disabled plugins, more restrictive permission modes and enabled sandboxes; an edit whose direction is not established says so. (#820) +- A Claude Code setting that disables prompts or approves project MCP servers carries one rating on every surface that names it. (#827) +- Rate documented arbitrary-code Bash launcher prefixes with trailing wildcards as critical/admin, so widening a rule cannot lower its rating; `check` no longer reports respelling a Bash rule as a new grant. (#824) +- A changed Claude Code `git push` prefix allow names the documented command forms the narrower deny prefixes in the same source do not cover. (#829) +- Note a literal, unconditional inline `PreToolUse` allow on a broad Claude Code hook matcher. (#826) +- Add factual mutable-source notes to supported MCP launcher changes, including pinned-to-mutable transitions. (#825) +- Add conditional review guidance for supported Claude Code shell-permission changes. (#839) +- An unchanged instruction file or plugin manifest reached through an in-tree link no longer refuses the whole host comparison. (#822) +- Skill frontmatter `metadata` no longer refuses the entire host comparison. (#848) +- `verify --preview` in a repository with a `shipgate.yaml` no longer sends the caller to a human for missing evidence. (#807) +- Workflow comparison values label their aggregate access, so it is not mistaken for an individual token scope. (#859) +- Host-only verifier headlines and `control.reason` label their raw count as rows. (#857) +- Host-only Action summaries no longer invent an empty scan status or zero scan severity counts. (#854) +- The PR-comment fallback renders correctly in the job summary after a 403 or 404. (#856) +- Version-based Action installs log the installed engine content digest. (#855) +- Move the published-release pins, examples and adoption prompts to `v1.1.0` (contract 40) now that it was published. (#778) ## 1.1.0 - 2026-09-22 diff --git a/README.md b/README.md index a304f921..35436384 100644 --- a/README.md +++ b/README.md @@ -41,8 +41,8 @@ It shows source-observed changes per agent, including before/after signatures and source locations, in the package that holds the changed code's agents unless `--scope` names one. See [application comparison](docs/application-comparison.md) for scoped applications, moves, exact refs and coverage limits. Version availability -is recorded in the [CHANGELOG entry](CHANGELOG.md#application-comparison-without-prior-setup); -while it is under Unreleased, use a source build containing the feature. +is recorded in the [CHANGELOG entry](CHANGELOG.md#application-comparison-without-prior-setup): +it is new in 1.2.0, so until 1.2.0 is published, use a source build containing the feature. ## What did this PR change? @@ -159,7 +159,7 @@ Supported shell changes then add conditional human choices. They establish no intent or runtime access and grant no authority. The `launch source is mutable` note identifies the unversioned `npx` package declared by the added server; it changes neither the row's severity nor the widening count. The source tree still -reports version `1.1.0`; that version string does not make this a published-wheel +reports version `1.2.0`; that version string does not make this a published-wheel capture. When the answer is useful and you want it on every pull request, add diff --git a/STABILITY.md b/STABILITY.md index 1addc4d7..29f3504f 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -1,8 +1,8 @@ -# Stability Contract · 1.1.0 +# Stability Contract · 1.2.0 What agents and CI integrations can rely on across versions of Agents Shipgate. -Unreleased #829 adds a source-local residual-prefix explanation to the existing +New in 1.2.0, #829 adds a source-local residual-prefix explanation to the existing host comparison row `why` text for supported Claude Code `git push` allows. It reads all compared head deny rules in that source, including unchanged ones, and lists only fixed documented examples outside every deny prefix. @@ -13,7 +13,7 @@ redacted prefix. No field, schema, direction, expansion, severity, check decision or authority changes. The note describes pattern coverage, not runtime approval; other rules still apply. -Unreleased, runtime contract v41: a host comparison names the changed inputs it +New in 1.2.0, runtime contract v41: a host comparison names the changed inputs it does not read (#821). Verifier `0.21` and capability diff `0.4` add a `changed_not_read` coverage item, with the `candidate` rule that named it, for each path in the comparison's own changed-file set that a bounded, documented @@ -29,7 +29,7 @@ workspace too. `minimum_control_contract_version` stays `21`. See [the migration note](#unread-changed-inputs-821). -Also in unreleased runtime contract v41, extended in place: the host inventory +Also new in 1.2.0, in runtime contract v41, extended in place: the host inventory reads how a coding agent is launched inside a workflow job (#823). Host-grants `0.6` shipped in 1.1.0, so host-grants inventory, baseline and drift schemas move to `0.7`: a workflow grant adds `agent_launches[]` — a @@ -62,7 +62,7 @@ workflow stays comparable. It moves neither #821's verifier `0.21` nor its capability diff `0.4`, and `minimum_control_contract_version` stays `21`. See [the migration note](#workflow-agent-launches-contract-v41-823). -Also in unreleased runtime contract v41: a hook row names what changed in the +Also new in 1.2.0, in runtime contract v41: a hook row names what changed in the hook, and an MCP row a change to the server's launch arguments (#819). Host-grants inventory, baseline and drift schemas move to `0.7`: a hook grant adds `handlers[]` — each handler's group `matcher`, its `command` as @@ -81,7 +81,7 @@ schema, a `0.6` baseline without workflow grants stays comparable with no new ro the members. `minimum_control_contract_version` stays `21`. See [the migration note](#hook-mcp-detail-fields-819). -Also unreleased, and moving no version of its own: a Claude Code setting that +Also new in 1.2.0, and moving no version of its own: a Claude Code setting that disables prompts or approves project MCP servers carries one rating on every surface (#827). The `audit --host` grant, the `diff`, `verify` and `check` rows, `check`'s violation and `verify`'s finding read one table. @@ -93,7 +93,7 @@ the wildcard check to `SHIP-HOST-BOUNDARY-PERMISSION-ALLOW-EXPANDED` at entries become grants. See [the migration note](#claude-setting-ratings-827). -Also unreleased, and moving no version of its own: `check` and `verify` route +Also new in 1.2.0, and moving no version of its own: `check` and `verify` route a changed hook declaration of a plugin the repository's project settings enable to protected-surface review, wherever the plugin keeps it (#809). A hook file such a plugin selects outside the registry paths, such as @@ -107,7 +107,7 @@ such a plugin selects under a name the reader does not follow fires still not routed. No schema, member, row or check id moves. See [the migration note](#enabled-plugin-hook-routing-809). -Also unreleased, and moving no version of its own: a `verify --preview` pointer +Also new in 1.2.0, and moving no version of its own: a `verify --preview` pointer never binds the verification plan (#807). In a repository with a manifest, the preview's pointer bound the plan a `verify` would run, whose inputs a preview never reads, so it had no input-directory census and every reader refused it: @@ -314,7 +314,7 @@ the Action tag) for reproducible CI. -## Migration Note: Unreleased — selected hook script dependencies (#702) +## Migration Note: 1.2.0 — selected hook script dependencies (#702) Host-grants `0.7` adds `script_inputs` to hook comparison facts and `hook_script` artifacts to the inventory and drift shapes. A supported direct @@ -375,7 +375,7 @@ runtime execution or script semantic interpretation is introduced. See the -## Migration Note: Unreleased — mutable MCP launch source notes (#825) +## Migration Note: 1.2.0 — mutable MCP launch source notes (#825) Host-grants 0.7 / contract 41 are extended in place. MCP inventory grants add optional `launch_source`: `null` when not established, otherwise a `pin` of @@ -395,7 +395,7 @@ The note says mutable, not unsafe; it is not an actionability or outreach claim. -## Migration Note: Unreleased — conditional shell-permission review guidance (#839) +## Migration Note: 1.2.0 — conditional shell-permission review guidance (#839) The existing maintained `review.changes[]` gains optional `guidance`: a bounded projection of raw Claude Code shell-rule evidence, a specific intent question, @@ -407,8 +407,8 @@ CLI and PR output render the same published projection after existing facts, coverage and authority. Whole-item bounds retain an omitted count and the JSON evidence pointer. Existing rows, inventory digests and control are unchanged. -Verifier `0.21`, capability diff `0.4` and runtime contract v41 are unreleased -and extended in place. The frozen `0.20` schema stays unchanged; a payload +Verifier `0.21`, capability diff `0.4` and runtime contract v41 had not shipped +before 1.2.0 and were extended in place. The frozen `0.20` schema stays unchanged; a payload claiming that version cannot carry `guidance`. Current readers accept missing guidance as unrecorded, not as proof that no review is needed; a recorded `guidance: null` is a change that is not a supported shell case, and prints @@ -417,7 +417,7 @@ nothing. See the -## Migration Note: Unreleased — changed grants are not automatically widenings (#820) +## Migration Note: 1.2.0 — changed grants are not automatically widenings (#820) Host-grants `0.7` and contract 41 are extended in place. The issue originally named unreleased `0.6`/40; those shipped in 1.1.0 and remain frozen. No field, @@ -519,7 +519,7 @@ before/after counts separately from the historical row-presence scores. -## Migration Note: Unreleased — arbitrary-code launcher allow rules (#824) +## Migration Note: 1.2.0 — arbitrary-code launcher allow rules (#824) Host-grants 0.7 / contract 41 are extended in place. The documented [launcher table](docs/engineering/exec-equivalent-permissions.md) rates exact @@ -557,7 +557,7 @@ removal and an addition. -## Migration Note: Unreleased — free-form skill metadata (#848) +## Migration Note: 1.2.0 — free-form skill metadata (#848) In 1.1.0 (contract 40), a skill with `metadata: {internal: true}` made the whole host comparison incomparable. The same happened for integer and nested @@ -591,7 +591,7 @@ ignored. -## Migration Note: Unreleased — workflow values label aggregate access (#859) +## Migration Note: 1.2.0 — workflow values label aggregate access (#859) Workflow row `before` and `after` values now begin with `access: read`, `access: write` or the other recorded access value, instead of an unlabelled @@ -603,7 +603,7 @@ published values; no schema changes. -## Migration Note: Unreleased — host control reasons count rows explicitly (#857) +## Migration Note: 1.2.0 — host control reasons count rows explicitly (#857) The host-only verifier headline and `control.reason` now say `N repository-declared host capability row(s)` instead of `change(s)`. @@ -614,7 +614,7 @@ limits. No schema, decision, permission, next action or merge authority changes. -## Migration Note: Unreleased — unrelated tools do not obscure permission replacements (#858) +## Migration Note: 1.2.0 — unrelated tools do not obscure permission replacements (#858) Permission replacement selection still stays within one host, source and disposition and uses the existing lattice. After setting aside moved rules, @@ -642,7 +642,7 @@ is added, and historical artifacts are not rewritten. -## Migration Note: Unreleased — a plugin directory that cannot be compared no longer hides the rest (verifier `0.21`, capability diff `0.4`, contract v41, #808) +## Migration Note: 1.2.0 — a plugin directory that cannot be compared no longer hides the rest (verifier `0.21`, capability diff `0.4`, contract v41, #808) A pull request that broke one plugin manifest — `plugins/demo/.claude-plugin/plugin.json` left as `{not json` — and also dropped a `deny` rule from `.claude/settings.json` @@ -687,15 +687,15 @@ is added. - **Text.** `diff` opens with `Partial comparison against main () -> working tree: head_inventory_incomplete`, then, before any row, `Not compared: plugins/demo, a plugin directory this entry could not read completely, so no change inside it is shown and nothing is claimed about it.` and `The changes below come only from sources outside it, so they are not the whole change; nothing here is a claim that the change is safe.` With no row, that last line reads `No static host-grant change was detected outside it. That is not a no-change answer for this change, and no verdict is implied.`, and `No static host-grant changes detected` is never printed. `verify --format text` and the manifest-free PR comment open with `Host capability comparison partial: head_inventory_incomplete` and the same two lines. The limit's item reads `plugins/demo/.claude-plugin/plugin.json (claude-code): parse_failed in head, so nothing in plugins/demo was compared`. The rows, their summary, the review question and the reproduction lines print as a comparable result prints them. - **Authority and routes do not move.** A partial comparison is not comparable, so every consumer that switches on `comparison_status == "comparable"` reads it as it read the refusal. `verify`'s control state, permissions, next action (`audit --host`), merge verdict and exit code are the incomplete comparison's, and only its headline changes, to `Host comparison is partial: N repository-declared host capability row(s) outside what it could not compare, listed under host_comparison in verifier.json; review its input limits before interpreting changes.` The control envelope's `capability_rows` projects a partial comparison as `incomparable`, with its reasons and no rows, exactly as before: that block cannot name a directory, so its `reason`, which is that headline, points to the rows in `verifier.json`. `check` records no coverage, so it cannot name one either and refuses its comparison as `1.1.0` did; its decision and violations come from its own routing, so the #808 fixture still gives `require_review` with `HOST-PERMISSION-DENY-REMOVED` and no rows. The Stop hook still tells the agent to treat the change as unreviewed. `audit --host`, the inventory digests, saved host-grants baselines and drift payloads are unchanged (host-grants stays `0.6`), so a partial comparison never creates or satisfies a baseline. The host-config and cold-start benchmark replays reproduce their run-of-record scores. `minimum_control_contract_version` stays `21`. -**Compatibility.** Verifier `0.21`, capability diff `0.4` and runtime contract v41 are unreleased, so they are extended in place. `comparison_status` is a closed enumeration, so a reader validating against the frozen [`docs/verifier-schema.v0.20.json`](docs/verifier-schema.v0.20.json) rejects `partial`; the current reader refuses a `0.20` artifact that claims a partial comparison or a `scope`. A consumer that treats every status but `comparable` as not comparable is unaffected. One that expected only `comparable` or `incomparable` should read `partial` as `incomparable` for any decision, and may read its rows as what is known outside the directories named. +**Compatibility.** Verifier `0.21`, capability diff `0.4` and runtime contract v41 had not shipped before 1.2.0, so they were extended in place. `comparison_status` is a closed enumeration, so a reader validating against the frozen [`docs/verifier-schema.v0.20.json`](docs/verifier-schema.v0.20.json) rejects `partial`; the current reader refuses a `0.20` artifact that claims a partial comparison or a `scope`. A consumer that treats every status but `comparable` as not comparable is unaffected. One that expected only `comparable` or `incomparable` should read `partial` as `incomparable` for any decision, and may read its rows as what is known outside the directories named. --- -## Migration Note: Unreleased — workflow agent launches (host-grants `0.7`, contract v41, #823) +## Migration Note: 1.2.0 — workflow agent launches (host-grants `0.7`, contract v41, #823) -Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` rather than extending it in place, and extends in place the unreleased runtime contract `41` that #821 minted ([its note](#unread-changed-inputs-821)). The `0.6` schema files stay published and unchanged. A workflow grant adds three members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: +Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baseline and drift `0.7` rather than extending it in place, and extends in place runtime contract `41`, unpublished before 1.2.0, that #821 minted ([its note](#unread-changed-inputs-821)). The `0.6` schema files stay published and unchanged. A workflow grant adds three members, each present only when a step declares one; in a `0.7` grant their absence means the steps were read and declare none: ```json { @@ -739,7 +739,7 @@ Host-grants `0.6` shipped in 1.1.0, so this mints host-grants inventory, baselin -## Migration Note: Unreleased — the changed inputs a host comparison does not read (verifier `0.21`, capability diff `0.4`, contract v41, #821) +## Migration Note: 1.2.0 — the changed inputs a host comparison does not read (verifier `0.21`, capability diff `0.4`, contract v41, #821) A zero-row comparison could not tell a reviewer that the change touched agent configuration this entry does not read. A pull request that added a Cursor @@ -785,13 +785,13 @@ command, verdict, reader, row or control state is added. **One route moves, on `verify` and `verify --preview` alike.** `verify` without a `shipgate.yaml` returned to the setup route (`Shipgate config not found`, exit `2`) whenever neither side of the comparison held a host artifact, and that route says nothing about the change. A comparison that read no artifact but names a changed input this entry does not read, or counts one or more changed candidate inputs as not examined (`unread_candidates_not_examined` above `0`, the one place that change is mentioned), is now published instead, on the existing manifest-free host route: advisory, exit `0`, `control.state` `agent_action_required` with the `audit --host` next action that route already names. `verify --preview` runs the same comparison and moves the same way: where its next action was `initialize` (`init --write`) with `host_comparison: null`, it is now `discover` (`audit --host`) with the comparison published and the host route's headline; `control.state` stays `agent_action_required` and the exit stays `0`. That includes an agent-related workspace, such as one whose change also adds a tool: a published host comparison takes the preview route whenever one exists, exactly as it already did when the change edits a host file this entry reads, such as the root `.claude/settings.json`. A comparison that reads no artifact, names nothing and counts nothing as not examined still takes the setup route on `verify` and `initialize` on `verify --preview`, as before; so does one whose changed files could not be listed (`unread_candidates: not_examined`), which says nothing about whether a candidate changed. -**What does not change.** `comparison_status`, `incomparable_reasons`, `rows` and every row value, `review`, `unchanged_limits`, every other coverage item, the inventory digests, saved host-grants baselines and drift payloads (this change moves no host-grants schema; the unreleased host-grants `0.7` is #823's, [its note](#workflow-agent-launches-contract-v41-823)), `audit --host`, `check`'s decision, rows and text, the control envelope's `capability_rows`, and every control state, permission and next action on a comparison that reads a host artifact. The host-config and cold-start benchmark replays reproduce their run-of-record scores. `minimum_control_contract_version` stays `21`. +**What does not change.** `comparison_status`, `incomparable_reasons`, `rows` and every row value, `review`, `unchanged_limits`, every other coverage item, the inventory digests, saved host-grants baselines and drift payloads (this change moves no host-grants schema; host-grants `0.7`, new in 1.2.0, is #823's, [its note](#workflow-agent-launches-contract-v41-823)), `audit --host`, `check`'s decision, rows and text, the control envelope's `capability_rows`, and every control state, permission and next action on a comparison that reads a host artifact. The host-config and cold-start benchmark replays reproduce their run-of-record scores. `minimum_control_contract_version` stays `21`. **Compatibility.** `coverage` and its items are closed objects, so a reader validating against the published [`docs/verifier-schema.v0.20.json`](docs/verifier-schema.v0.20.json) rejects a `0.21` artifact's new members; that schema stays frozen. The current reader reads a `0.20` artifact as `0.21` with `unread_candidates: null`, which is what that build knew, and refuses one that claims a `changed_not_read` item, a `candidate`, `read_sources_only: false` or either `unread_candidates` member. A `diff --json` consumer sees `capability_diff_schema_version: "0.4"`. A consumer switching on `coverage.items[].status` should treat an unknown status as a change it must read, not as no change. -## Migration Note: Unreleased — unconditional inline hook approvals (host-grants `0.7`, contract v41, #826) +## Migration Note: 1.2.0 — unconditional inline hook approvals (host-grants `0.7`, contract v41, #826) Host inventory `0.7` and runtime contract `41` are extended in place with two optional, display-only members of a hook handler, published only on a Claude @@ -808,7 +808,7 @@ unchanged. -## Migration Note: Unreleased — hook matcher, command and timeout, and MCP launch arguments (host-grants `0.7`, contract v41, #819) +## Migration Note: 1.2.0 — hook matcher, command and timeout, and MCP launch arguments (host-grants `0.7`, contract v41, #819) A hook row read `PostToolUse → PostToolUse` whether the edit was to the hook's matcher, its command or its timeout, and an MCP server whose version pin moved @@ -847,10 +847,10 @@ baselines** below): -## Migration Note: Unreleased — one rating per Claude Code setting (#827) +## Migration Note: 1.2.0 — one rating per Claude Code setting (#827) This change moves no version of its own: no schema, member, check id or -`minimum_control_contract_version` moves. Of the unreleased tree's versions, +`minimum_control_contract_version` moves. Of the versions new in 1.2.0, host-grants `0.7` is #823's ([migration note](#workflow-agent-launches-contract-v41-823)), capability diff `0.4` and verifier `0.21` are #821's @@ -938,11 +938,11 @@ finding — all read it. The ratings and their basis are in -## Migration Note: Unreleased — an enabled plugin's hook is routed wherever it lives (#809) +## Migration Note: 1.2.0 — an enabled plugin's hook is routed wherever it lives (#809) No schema, member, check id or `minimum_control_contract_version` moves. This change adds no version of its own: host-grants, capability diff, verifier and -the runtime contract are what the rest of the unreleased tree carries. What +the runtime contract are what the rest of 1.2.0 carries. What moves is when two existing checks fire, and so `check`'s decision and control, and a manifest-backed `verify`'s release decision, merge verdict and PR comment. @@ -1013,7 +1013,7 @@ gave a `widened`, `expands: true` row beside `decision: allow` and -## Migration Note: Unreleased — a preview is read against the working tree it read, configured or not (#807) +## Migration Note: 1.2.0 — a preview is read against the working tree it read, configured or not (#807) No schema, contract, member, error kind, refusal code, exit code or `minimum_control_contract_version` moves. What changes is which artifacts a @@ -1091,12 +1091,12 @@ Outside a repository a preview still declares no snapshot and still reads. -## Migration Note: Unreleased — an unchanged limit reached through an in-tree link is named, not a refusal (#822) +## Migration Note: 1.2.0 — an unchanged limit reached through an in-tree link is named, not a refusal (#822) No schema, member, reason code, check id or `minimum_control_contract_version` moves, and host-grants stays `0.6`. This change adds no version of its own: -capability diff, verifier and the runtime contract are what the rest of the -unreleased tree carries. What moves is which comparisons name an unchanged +capability diff, verifier and the runtime contract are what the rest of 1.2.0 +carries. What moves is which comparisons name an unchanged limit instead of refusing, and so which are `comparable`, and which `check` compares. @@ -1182,8 +1182,8 @@ accept, and a change that only drops `deny: Bash(curl *)` from Either way, `check`'s decision, violations and control state do not move: the change above is still `require_review` with `HOST-PERMISSION-DENY-REMOVED`. -- **`partial` becomes `comparable`.** On this unreleased tree, a comparison - refused only by such a plugin-reference limit is `partial` (#808): `diff`, +- **`partial` becomes `comparable`.** Without this change, a comparison + refused only by such a plugin-reference limit is `partial` (#808, also new in 1.2.0): `diff`, `verify` and the PR comment withhold its plugin directory, as `Not compared: plugins/demo`, and publish the rows outside it. Once the proof holds, the limit is an unchanged one, so a comparison that was @@ -1202,9 +1202,9 @@ accept, and a change that only drops `deny: Bash(curl *)` from benchmark replays reproduce their run-of-record scores. **Compatibility.** No field changes shape. A consumer that switches on -`comparison_status` reads `comparable` where it read `incomparable` for these -layouts, or `partial` on this unreleased tree for a plugin-reference limit, -with the limit in `unchanged_limits` exactly as a limit at its own path has +`comparison_status` reads `comparable` where `1.1.0` read `incomparable` for these +layouts (the intermediate `partial` for a plugin-reference limit, #808, never +shipped), with the limit in `unchanged_limits` exactly as a limit at its own path has been published since `1.0.0`. Where such a limit was its only refusal, `check`'s boundary result reads `unchanged_limits_not_representable` for a limit it cannot leave out, and for a shared plugin-reference limit it diff --git a/docs/INDEX.md b/docs/INDEX.md index 5ddc067e..75765595 100644 --- a/docs/INDEX.md +++ b/docs/INDEX.md @@ -226,6 +226,7 @@ remain separate, pending work. - [`release-runbook.md`](release-runbook.md) — cutting a tag: mandatory rehearsal, the two-job publication transaction, provenance bindings, and the recovery path when PyPI succeeds but finalization fails - [`release-evidence-policy-decision.md`](release-evidence-policy-decision.md) — the approved release evidence bar: the 38-case `pre_1_0` policy for `0.x` tags, the 80-case `beta` policy from 1.0 on, and the promotion path between them (decided 2026-08-29, #341; amended 2026-09-04, #520) - [`release-evidence-policy-decision.md`](release-evidence-policy-decision.md) § Amendment 2 — the unqualified preview channel: the admissibility finding, the five conditions it holds under, and what would have made it inadmissible (decided 2026-09-02, #491) +- [`changelog/1.2.0.md`](changelog/1.2.0.md) — the full reviewed prose for every 1.2.0 change; `CHANGELOG.md` carries the one-line-per-change release note - [`changelog/1.1.0.md`](changelog/1.1.0.md) — the full reviewed prose for every 1.1.0 change; `CHANGELOG.md` carries the one-line-per-change release note - [`changelog/1.0.0.md`](changelog/1.0.0.md) — the full reviewed prose for every 1.0.0 change made after the `0.16.0` section was cut; `CHANGELOG.md` carries the one-line-per-change release note - [`changelog/0.16.0.md`](changelog/0.16.0.md) — the full reviewed prose for every 0.16.0 change; `CHANGELOG.md` carries the one-line-per-change release note diff --git a/docs/agent-boundary-result-schema.v3.json b/docs/agent-boundary-result-schema.v3.json index 0e5e8308..7c922597 100644 --- a/docs/agent-boundary-result-schema.v3.json +++ b/docs/agent-boundary-result-schema.v3.json @@ -484,7 +484,7 @@ "type": "string" }, "version": { - "default": "1.1.0", + "default": "1.2.0", "title": "Version", "type": "string" } diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index 00c6e059..464e4273 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -776,7 +776,7 @@ Downstream repos generated with `.shipgate/agent-contract.json`. - Latest release: `v1.1.0` -- In-tree runtime: `1.1.0` — see [pyproject.toml](../pyproject.toml) +- In-tree runtime: `1.2.0` — see [pyproject.toml](../pyproject.toml) - Runtime contract: `41` (minimum control contract: `21`) - Current report schema: `1.0`, frozen, superseding `0.43` — [`docs/report-schema.v1.0.json`](report-schema.v1.0.json); the `1.x` rules are in [`docs/report-1-0-contract.md`](report-1-0-contract.md) - Current packet schema: `0.18` — [`docs/packet-schema.v0.18.json`](packet-schema.v0.18.json) diff --git a/docs/agent-handoff-schema.v9.json b/docs/agent-handoff-schema.v9.json index fd379c4b..1ac5c136 100644 --- a/docs/agent-handoff-schema.v9.json +++ b/docs/agent-handoff-schema.v9.json @@ -595,7 +595,7 @@ "type": "string" }, "version": { - "default": "1.1.0", + "default": "1.2.0", "title": "Version", "type": "string" } diff --git a/docs/agent-result-schema.v3.json b/docs/agent-result-schema.v3.json index c4d794f7..1c498301 100644 --- a/docs/agent-result-schema.v3.json +++ b/docs/agent-result-schema.v3.json @@ -416,7 +416,7 @@ "type": "string" }, "version": { - "default": "1.1.0", + "default": "1.2.0", "title": "Version", "type": "string" } diff --git a/docs/ai-search-summary.md b/docs/ai-search-summary.md index deeb49cc..25549e13 100644 --- a/docs/ai-search-summary.md +++ b/docs/ai-search-summary.md @@ -113,7 +113,7 @@ Per-agent guides cover [Codex](agents/use-with-codex.md), [Claude Code](agents/use-with-claude-code.md), and [Cursor](agents/use-with-cursor.md). -The current source tree is `1.1.0` (runtime contract 41, unreleased). The +The current source tree is `1.2.0` (runtime contract 41, unreleased). The latest published release is `v1.1.0` (runtime contract 40), on the advisory channel with no qualification claim. In report v1.0, `passed` is an evidence-backed static verdict: the configured root has a diff --git a/docs/application-comparison.md b/docs/application-comparison.md index 63ed763a..c9d411a3 100644 --- a/docs/application-comparison.md +++ b/docs/application-comparison.md @@ -1,8 +1,8 @@ # Application comparison without prior setup -**Availability:** see the [CHANGELOG entry](../CHANGELOG.md#application-comparison-without-prior-setup). -While that entry is under Unreleased, run the source checkout's `./shipgate` -or a build containing the feature. +**Availability:** new in 1.2.0; see the [CHANGELOG entry](../CHANGELOG.md#application-comparison-without-prior-setup). +Until 1.2.0 is published, run the source checkout's `./shipgate` or a build +containing the feature. For an OpenAI Agents SDK or Google ADK application, compare committed PR refs: diff --git a/docs/changelog/1.2.0.md b/docs/changelog/1.2.0.md new file mode 100644 index 00000000..ff9ae2c8 --- /dev/null +++ b/docs/changelog/1.2.0.md @@ -0,0 +1,305 @@ +# 1.2.0 — full change record + +The reviewed prose for every change in 1.2.0: the rationale, the trap each fix +closed, and the invariants each one established. This is the record. The +`1.1.0` record is [`1.1.0.md`](1.1.0.md). + +[`CHANGELOG.md`](../../CHANGELOG.md) carries the *release note*, one line per +change, because that section is published verbatim as the GitHub release body, +and verification refuses a body over 125,000 characters. See +[`release-runbook.md`](../release-runbook.md) § Release-note hygiene. + +Entries are grouped as in the release note: the new application comparison +first, then the host comparison and the Action. + +--- + +### Application comparison without prior setup + +- Add `diff --application` for OpenAI Agents SDK and Google ADK source-observed per-agent wiring, with exact base/head refs and independently selected scopes. No manifest, saved baseline or authored declarations are needed. The advisory result (`application_comparison_schema_version` `"0.1"` here; 1.2.0 publishes `"0.2"`, after #872) records before/after evidence, scoped coverage gaps and explicit uncertain candidates; it supplies no release verdict or merge permission. Includes scoped Git materialization, partial-clone recovery, and definition lookup using reader-resolved Python symbols and locations. See [application comparison](../application-comparison.md). (#871) +- `diff --application` without `--scope` now derives the scope from the change instead of comparing the repository root. Each changed Python file is related to the OpenAI Agents SDK and Google ADK files that import it, or that it imports, within six import hops (a package's `__init__.py` included); a changed test relates only when an agent imports it. Agent files are those that construct an agent class under any name it is imported as, subclass one, copy one with capabilities of its own, or change an agent's capabilities after construction. A changed file not related to a module that builds an agent is named, with why (no import path, a path longer than six hops, only a rewiring module, a file too large to read), and makes the result `partial` even inside a compared scope; so does a changed link to a module or directory, a changed submodule, an agent outside the compared scopes whose imports go past the bound, or a module that imports the change (directly, through a package or agent file's re-export, or through other modules), imports a builder whose agent's capabilities are not a fixed list of its own names, and builds, copies or rewires an agent itself, calls the repository's code with its own arguments, or sets its module state, unless one compared scope holds all three. A module that builds its agent through a builder's factory is an agent file. A relocated application is one comparison, whatever it holds. Each related group is compared in the outermost package that holds it. On mezgoodle/Vesta#58 a change to `backend/app/services/` is compared in `backend/app`, where its agents are, rather than in the changed directory, which holds none. Independent applications in a monorepo are separate comparisons under `comparisons`, never the root. A change that touches no supported agent is an explicit `not_established` answer naming the files considered. A bound reached is named and makes the result `partial`. `--json` records `scope_selection` (`derived` or `explicit`, the scopes and why). `--scope .` keeps the root, an explicit `--scope` always wins, and `--base-scope` now needs `--scope`. On the pinned corpus of 48 application PRs, derived scopes establish the same 296 rows the repository root does, where the changed-wiring directory establishes 18. The derived scope is narrower than the root for 20 PRs. For 7 PRs it answers that the change touches no supported agent; the root reported unrelated agents there. The median run takes 13 s, the same as the root. (#875) +- `diff --application` no longer reads another library's `Agent` or `function_tool` as the OpenAI Agents SDK's, and no longer reports an agent it stopped observing as removed. On speechmatics/speechmatics-academy#142, a LiveKit voice agent (`from livekit.agents import Agent, function_tool`) moved its tools into an `Agent` subclass that passes them through `super().__init__`. The comparison printed `compared` with four `REMOVED agent → …` rows although the same four tools were still bound. It now prints `not_established`: no supported application agent on either side. (#871 follow-up; related #580, #864, #865) + - **Framework identity.** Discovery and the SDK reader count `function_tool` and `Agent` as the SDK's only when the name is imported from the absolute `agents`/`openai_agents` package, or is not imported at all (the existing reading, unless a wildcard import from another module could supply it). The name is resolved in the scope that uses it, as Python resolves it, so an import in a sibling function never decides it, and a parameter or local assignment of that name is not the SDK's. A name imported from anywhere else — `livekit.agents`, a relative `.agents` package — is another library's. A LiveKit file is no longer an SDK candidate, and a file using both reads only the SDK's symbols. A relative import is no longer any framework's import signal. `scan` of a declared SDK source reads the same way. + - **Unobserved is not removed.** One side may observe an agent while the other side's same file still assigns or imports its name (`from factory import agent`), or passes it as `name=`, through a construction the reader does not support: an `Agent` subclass, a factory, `Agent[Context](...)` or `.clone()`. That side then records a coverage gap for the agent. The result is `partial`, and its rows are `not_established` with `candidate_change` `removed` or `added`. An agent referenced only in another agent's `handoffs` is not an observed construction. An agent whose name is gone from the file is still an established removal, and rows for other agents are unaffected. No schema changes. +- `diff --application` no longer reports `compared` with no changes while an agent it could not see gained a tool. On kkmiecik-coder/CRM#5, the quoting agent `Wycena`, built by `return Agent(name="Wycena", tools=NARZEDZIA_WYCENY)` inside a function, gained `wyslij_obraz`; the only agent read was a test double, so the result said `compared` and printed nothing. (#876) + - **Every SDK construction is observed or named.** The OpenAI Agents SDK reader read only `name = Agent(...)`. It now also reads `return Agent(...)`, `self.agent = Agent(...)`, agents inline in a list, `Agent("name")` and `Agent[Context](...)`, identified by the literal `name`. An agent assigned to a plain name keeps that name as its identity, so renaming its `name=` or moving it between scopes changes nothing; only a variable name assigned in more than one function or class body (two builders' local `agent`) gives way to each agent's literal `name`, so those are two agents, and a handoff to such a variable reaches its own. What it cannot read is a named limit on that agent, so other agents' rows stand: one identity constructed at two sites that bind different tools (identical constructions are one agent), including a Google ADK agent name; `**` or extra positional arguments; a copy (`clone`, `dataclasses.replace`, `copy.replace`) passing its own `tools`/`handoffs`/`mcp_servers`, or only `**` on a value known to be an agent; and an agent built from a same-module `Agent` subclass, including one made by `type("X", (Agent,), {})`. A construction without a literal name is a limit on its file. + - **Changes after construction.** A change to an agent's tools, handoffs or MCP servers — assigning, extending or slicing `x.tools`, a list method on it, `setattr`/`delattr` by name, or a handle `t = x.tools` that is itself changed later (a handle only read, like `len(request.tools)`, is nothing) — is read in the scope where it happens: on an agent the reader constructed (directly, through `self.agent`, or through a module function that returns it) it is a limit on that agent, and a change in place is one on every agent built from the same list object; on a value proven not to be an agent (`Settings()`, a literal, `type(...)()`, the instance's own `self.tools`) it is nothing; on anything else it is a limit on the file. In a module that is not read as an SDK source — a Google ADK source included, whose reader does not follow a change after construction either — a copy is a named limit on it, and so is a change to any object's tools once the module imports anything from the scope, or is itself a Google ADK source — through an alias, a loop, a parameter or a call's result, as in an SDK file — unless the object is plainly not an agent (a literal, or an instance of a class the scope defines that is not an `Agent` subclass); a module that imports nothing from the scope has another library's `.tools`. A module that imports an SDK `Agent` subclass from the scope, directly or through a package that re-exports it (`from app.core import *`), is limited too, never one that only spells its name in a string or imports a vendor class of the same name. + - **Google ADK agent subclasses.** A class deriving from an ADK agent class (`class Helper(LlmAgent)`, or a class deriving from that one) is a named limit where it is used: on the module that defines it when that module builds, passes or decorates it, and on every other module in the scope that imports it. An agent built from it was never read, so a tool it gained could leave the result `compared` with no rows. `scan` no longer reports the defining module's tool surface as enumerated when that module uses the class. A subclass nothing uses is not a limit. (Carried over from #880.) + - **Rows and verdicts that change.** A construction that used to be unread and silently absent is now either read (rows appear) or named (the result, and a `scan` of the same code, becomes `partial` / `insufficient_evidence` instead of complete). A Google ADK agent name constructed at two sites in one file with differing tools or handoffs is `insufficient_evidence` for `scan` too, where it could be `passed` or `review_required`; constructions binding the same definitions and handoffs, each read cleanly, are one agent and keep their rows and verdict. A copy passing no capabilities of its own, and a subclass nothing uses, are not limits. + - **No agent source, no census.** The census of copies, changes after construction and subclasses runs only when some side's discovery found an OpenAI Agents SDK or Google ADK source. Without one no agent is read, so `--scope src` of this repository stays `not_established` instead of turning `partial` over its own `artifacts.tools.append(record)`; every module is still parsed, and a parse failure is still a gap. + - **Test files are not the application.** Test files (the discovery convention, relative to the scope: `test`/`tests` directories, `test_*.py`, `*_test.py`, `conftest.py`, `test.py`, `tests.py`) are listed per side in `excluded_tests`, printed in the text output, and not read as agent sources, so a test double cannot establish a scope and a test file's own defects are not application gaps. + - **A duplicate tool definition limits one name.** A file defining one tool name twice refused the whole comparison with exit 2, and the message asked for the test file to be edited. It is now a named limit on that name in that file: no agent binds either definition, and the file's other tools and every other file are still compared. Seen on alliance-genome/agr_ai_curation#842 and usestrix/strix#1103. + - **One identity built twice keeps what both sites bind.** When one agent name is constructed at two sites (tensorflow#128063's root agent and builder), a tool both sites bind to the same callable is still a row of that agent, beside the limit that names the constructions; only a tool they bind differently is an ambiguous identity. + - Additive on the `0.1` advisory result, which never shipped (`excluded_tests` per side); no schema or contract bump. +- `diff --application` no longer refuses a repository over a link or submodule its application reader never opens, and reads a Google ADK agent imported from the package root. (#871 follow-up, measured for #868) + - **The problem.** On 2026-09-25, against open third-party pull requests that edit SDK/ADK tool wiring, 15 of the first 27 runs at the default root scope exited 2 with `Git tree contains unsupported external binding`. The path named was never application source: `CLAUDE.md -> AGENTS.md` (dlt-hub/dlt#4417, jaegertracing/jaeger#9636, omnigent-ai/omnigent#6611, wandb/weave#7948), a linked `.claude/skills/…` or `.agents/skills/…` directory (asterinas#3834, vllm#57322, kagent-dev/kagent#2788), `agent/VERSION -> ../../VERSION` (TencentCloud/CubeSandbox#1508), `.pylintrc` (tensorflow#128063), and a vendored submodule (temporalio/sdk-python#1868). The root scope took the unscoped archive route, which refuses every link and gitlink. `from google.adk import Agent` was not read as an agent, so CubeSandbox#1508's `root_agent` gave `not_established` ("No supported application agents were established"). + - **Links.** The root scope now uses the scoped materializer a named scope already used, which recreates each link as a link. A link is never read through. Every link under the scope is censused, a dangling one included, and gapped only where it can hide application source: a `*.py` link whose target is not a Python input the scope already reads (an alias of one is compared at the target's path; reading it too made one agent two ambiguous ones and hid the real file's change, as a named scope did on main); a link to a directory outside the scope that holds Python; and a link resolving to nothing in the repository where the other side reads source at or beneath its path. Replacing `agent.py` with a dangling link, or a source directory with an absolute link, is therefore `not_established`, never a removal. A link Python discovery would not read changes nothing, and a scope that is itself a link is still refused. The host-configuration census, which counts every link that could conceal a *host* path, no longer adds application coverage gaps. + - **Submodules.** A gitlink under the scope is materialized as the empty directory a checkout without `--recurse-submodules` leaves; its content is never fetched. The same gitlink commit on both sides is named in `limits` as unchanged and does not make the comparison `partial`, since identical content cannot carry a change. An added, removed or moved gitlink is a coverage gap over its path on each side that has it, and over the whole scope when it is the selected scope itself, so the result is `partial`, never `compared`. A gitlink was refused (exit 2) before. + - **Google ADK.** `google.adk.Agent`, the package-root re-export of `google.adk.agents.llm_agent.Agent`, is read as an agent constructor by the ADK reader, for `from google.adk import Agent` and `import google.adk as adk` alike. CubeSandbox#1508 now establishes `cube_code_agent` and is `partial`, naming the tool it imports from a sibling module (#864), instead of `not_established`. + - **Re-measured.** At the root scope, main `a430e81a` refused CubeSandbox#1508, dlt#4417 and asterinas#3834; this change compares each, with its own limits (a Python census past `--max-python-files`, an unsupported framework, an unresolved import). No application comparison schema change of its own; the result was still `"0.1"` then, and 1.2.0 publishes `"0.2"` (#872). +- `diff --application` rows now name what a bound tool reaches: the endpoint, the request fields, the credential it sends and which arguments the model chooses. (#872, part of #868) + - **The problem.** tensorflow/tensorflow#128063 was the only one of 134 open SDK/ADK PRs whose rows were both correct and complete. Even so, each row named only a signature. `submit_pr_code_review` posts a pull-request review whose `event` can be `APPROVE`, using `GITHUB_TOKEN`; `get_pull_request_details` only reads. The existing assessment gave all three tools `write` with unknown evidence. + - **Reach.** Each side of a function-tool row carries `reach`: the outbound `requests`, `httpx`, `aiohttp` and `urllib` calls read from the tool's own code and its same-scope helpers, up to three calls deep. Every call the read cannot follow is a named limit with its location. Each call records: + - the method and the URL template (`{pr_number}`, `{env OWNER}`, `{…}`); + - literal request fields, and the literals a field is chosen from with the parameters that decide; + - `model_supplied`: which parameters flow where; + - `credential_sources`: environment variable names only, never a value; + - for GraphQL, query or mutation. + - **Effect evidence.** `effect_evidence` is the existing `assess_tool_semantics` over the tool, with the reach as one more structural source (`source_http_call`). A write or delete call supports that effect. `read` needs every call followed and every outbound call a read. Read-only method names pass only on plain data or on objects a known library returned. A decorator, a store into an unknown object, or a method on an object the read cannot name blocks `read`. + - **Secrets.** A hard-coded credential is `literal: true` and never printed. In a URL, a query value prints only when it is a short lowercase word or a number, and a token-shaped path piece is withheld. A field value prints only when it is a plain word and its name does not suggest a secret. + - **Two constructions of one ADK name.** Each row's `binding_location` names the first construction that lists the tool, and `construction_sites` lists every one. Before, every row pointed at the first construction. + - **Measured.** On the 134-PR corpus, rows, statuses and exit codes are unchanged, and run time is flat. Of 96 row sides in 19 PRs: + - 9 name outbound calls (3 PRs) and 5 name a credential; + - 7 carry structural effect evidence (3 read, 4 write); + - 65 name at least one limit, mostly database helpers, tracing and SDK clients. + - `reach`, `effect_evidence` and `construction_sites` are evidence outside the compared meaning. `application_comparison_schema_version` is `0.2`. +- Follow a tool an agent binds from another module in the repository. (#864) + - **The problem.** jpka/attest#3 added `memory_bank.remember_firm_finding` and `memory_bank.recall_firm_memory` to an ADK agent's `tools=[...]`, with the module brought in by `from . import memory_bank`; the reader stopped at the module boundary, so `diff --application` showed no row and `scan` catalogued only the unchanged local tools. The same gap left OpenAI Agents SDK tools such as `from ..tools.shop import add_to_cart` unresolved. + - **What resolves.** For Google ADK, a name imported from a sibling module or re-exported by a package, a module-qualified `module.function`, a plain `alias = function`, and `FunctionTool(imported_function)` / `LongRunningFunctionTool(...)`, including a wrapper built in the imported module; for the OpenAI Agents SDK, a name or `module.function` that reaches a definition carrying the SDK's `@function_tool`. The tool is the definition, with its own signature, location and implementation digest, so jpka/attest#3 now shows the two memory tools as `ADDED` and leaves its four unchanged bindings, `scorer.score_answer` included, alone. A definition reached by several spellings is one tool; same-named functions in different modules stay two. + - **The boundary.** Only regular `.py` files inside the directory the read was given — the `--scope` for `diff --application`, the manifest directory for `scan` — are read, through the bounded input reader, and parsed without being imported or run. Symbolic links are not followed and a module name must match a file's exact spelling. Each application row reached through an import adds `import_path`: every module read, the line of the binding followed and that module's SHA-256; it is evidence, not compared meaning. + - **What stays unresolved, by name.** A module the scope does not contain (the scope spelled from the repository root, `svc.app.tools` with scope `svc/app`, is read inside it), a relative import above the scope, more than one matching module location, a name bound twice or only inside an `if`/`try`, a wildcard import, an import cycle, a class or other value, a parameter or other local assignment of the scope that uses the name, a name that scope binds more than once, a module attribute the same module reassigns (`tools.lookup = ...`, `setattr`), an attribute named like a step of the chain on an imported module that code running first reassigns (every module on the chain, the agent's own file included, every enclosing package's `__init__.py`, and every in-scope module those import; the defining module handing its function on does not count), and such a module that cannot be read (a link, or a missing relative module not imported under `except ImportError`), an SDK function without `@function_tool`, a symbolic link, or more than 64 modules read. The gap names the reason (`Not resolved because …`) and is scoped to the agent that lists the tool, so another agent's change in the same file is still established. One agent binding two different functions under one name is named, not resolved. The ADK unresolved-tool warning keeps its wording. No schema or contract change. + - **Identity.** One agent binding two different functions under one name binds neither, in both readers and whatever their order, and names both definitions. `import a.b` then `a.b.f` reads the submodule, as the import system does. A reference is read where it is used: a builder's own import is followed like a module-level one, `nonlocal` follows the outer function, a nested `def` that is the only one of its name is that definition, a module-level agent binds what the module binds at top level (its `def`, its import, its wrapper assignment) and never a same-named `def` or wrapper nested in a function, a module-level list's names are read at module level whatever the building function binds, and a factory's own toolset or wrapper variable is read like a module-level one. An SDK list variable is read only when the scope that binds it binds it once, to a literal list, and every use of it in the file only reads it — iterated, indexed, compared, tested, handed to a read-only builtin, standard-library reader or logger's method (each proven by its binding), to an agent's (or a copy's) own `tools=`, or to a function whose every use of that parameter is such a read. Spreading it (`[*TOOLS, x]`, `f(*TOOLS)`) or testing it (`TOOLS or []` in a condition) is a read. A method call on it, `+=`, a second name (including through `x or y`), a tuple, a return, `*args`, `globals()`, `sys.modules` or importing the module by `__name__` makes it dynamic. For `scan`, a definition that an import reaches and another configured source also reads (spelling the module's path the same way) is one catalog tool, and the binding reaches it through the exact definition the reader resolved; a `{tool: …}` selector for it is not ambiguous, and the dropped copy's guard evidence goes with it. A source an inventory completes keeps its own observation, so when that source imports a definition another source also reads, the catalog holds both and a selector for it is ambiguous. When what the module binds is not established (a name rebound, or bound only inside an `if`) and the ADK reader falls back to a same-named `def` or wrapper, that binding is named but never established: in a comparison its row is `not_established` on whichever side it is present, added, removed or changed, and so is the row of every tool the module's bindings of that name could give the agent instead, each followed to its definition (a function from a module outside the scope by its imported name); every one of the agent's rows is when one of those cannot be followed or a wildcard import could bind the name; for `scan` it stays the medium-confidence shadowed definition it was. `x = FunctionTool(func=x)` right after `def x` wraps that `def`, and is not a guess. Code that runs before the name is used but is not read — a relative import above the scope, an absolute import of a module the repository holds outside the scope (read from the compared commit's tree, or the checkout for `scan`, at the root, under `src/`, or under any directory between the root and the scope, modules the interpreter preloads, a standard-library name at a root that is a regular package, and the scope's own package aside; a linked or submodule entry counts, and a namespace directory only when it holds the named submodule, so SDK apps under `agents/` still import the SDK), a relative module no file provides (a generated `*_pb2` or `_version` aside), or a package `__getattr__` that is not the lazy-submodule idiom (or that the package could subvert through `sys.modules`, `globals()`, `__name__` or a patched `importlib`) — keeps the tool named. The `__init__.py` of every package above the scope, and what each imports (every package on the way to it and each submodule named), are read too; there a reassignment counts unless it sets one attribute of another module file outside the scope, a module of the scope they import is read like the chain's own, a change to `__path__` is a caveat (`pkgutil.extend_path` aside), their files are read in batches, and what they import in turn, or import by name at run time, is not followed. `sys.modules` and `globals()` are read by allow-list: a store whose key names a module on the chain, a package above one or the framework's own modules is a named stop, a module rebinding its own name through them or its own module object (however spelled, including `__dict__` and `vars()` stores) is a reassignment, reads (a comparison, iteration, a spread, `pkgutil.iter_modules(__path__)`, `get_type_hints(globalns=globals())`) are nothing, and any other use (`mods = sys.modules`, `|=`, a computed key or `setattr`, the module object anywhere but an attribute, a plain alias or a reader, `__dict__.update`, a frame's `f_globals`, `builtins.globals` under another name) or a change to `__path__` is a caveat: its row is `not_established` with the reason, including through a `FunctionTool` wrapper, and for `scan` the ADK module stays at medium. Attest's lazy loaders (`importlib.import_module(f".{name}", __name__)`, `if name == "x": from . import x`) stay established; `from . import other as x` is not the idiom. When a definition another source reads under its own name is the one the reader resolved, the binding reaches it through its exact locator, never a same-named local definition. The module-binding walk and the SDK list reader are linear in the tree, and a chain of thousands of attributes no longer crashes the run. + +- Read a Google ADK tool a local factory builds, and a tools list built in the agent's function. (#865) + - **The problem.** visulate/visulate-for-oracle#526 bound `save_memory_tool = create_save_memory_tool()` and `read_memory_tool = create_read_memory_tool()` in its root agent's builder; each factory returns `FunctionTool` around a nested async function, and the reader stopped at the local assignment, so the two new memory tools were unresolved like the agent's nine others. MuhammadVT/smart-assignment#46 built `tools = [...]` in the agent's function and appended a triage tool under `if triage_enabled:`; the reader read the whole list as a dynamic tools expression, losing its unconditional tools. + - **Factories.** A factory call — `tool = create_tool()` bound once and unconditionally in the agent's function or at a module's top level, or `tools=[create_tool()]` — is followed, as syntax and never run, to the factory's one unconditional `return`: `FunctionTool(inner)` / `LongRunningFunctionTool(inner)`, a plain function, a name bound once to one of those, or another factory's call, up to four factories deep. The tool is the function wrapped — one nested in the factory and defined before the `return`, or one the factory's module binds — with its own signature and location; `import_path` records the factory as a step, and the factory's code and the values each of the agent's calls gives its parameters (a literal, or a name or module attribute bound once to one; defaults filled in), which its closure holds, are part of its implementation digest, so `make_sql_tool(readonly=False)` in place of `readonly=True` is a changed tool and the same call spelled otherwise is not; a value this read cannot name is a limit on that tool naming the argument, and a `not_established` row only when the calling module changed. + - **What stays named, with why.** A factory that returns from more than one place or only under a condition, calls itself on the way, is decorated, a generator or `async`; a wrapped function that is decorated, a parameter, defined more than once in its file (a tool is known by its name there), or changed or handed on in the factory (`inner.__name__ = ...`, `setattr`, a helper), a module function included; a tool changed or handed on where it is bound (`tool.name = ...`, `rename(tool)`); and a factory the repository holds outside the read scope, named as such. Any other call — a third-party package, a class, a name bound twice or through a wildcard — keeps the answer it had. On visulate#526 with `--scope ai-agent`, the memory tools are read, and the remote-delegate factory, which renames the function it returns per call, leaves the nine delegates named with that reason — so the two memory tools are `not_established` candidate additions, not a complete eleven-tool surface; they were not rows before. + - **A tools list.** `tools = [a, b]` (or a tuple) bound once in the agent's function, with `tools.append(c)`, `.extend([...])`, `.insert(i, c)` or `tools += [...]` statements, is read member by member up to the statement that builds the agent, which copies the list, so an addition after it — or after an early `return` of another agent — is not that agent's. An addition before it under a condition or in a loop is named on the agent, which is then not complete, and is never read as bound. Any other use of the list — handed to a call, aliased, returned, changed another way, read by a nested function, a starred member — and a module-level list keep the dynamic tools expression. On smart-assignment#46 the unconditional tools are read and the triage tool is named as conditional. +### Changes + +- Compare exact repository script bytes for supported, selected Claude Code and Codex hook executable references. A script-only edit produces an attributable hook row without asserting a permission expansion; its `why` names the direction as not established (#820), so the Claude Code Stop hook lists it apart from widenings rather than staying silent; `check` routes the dependency for review from either compared revision. The documented shell spellings resolve (`"$CLAUDE_PROJECT_DIR"/path`, `"${CLAUDE_PROJECT_DIR}"/path`, the variable and path quoted together, or unquoted with a plain path, and `CLAUDE_PLUGIN_ROOT` alike in a plugin's hooks). A script this entry cannot read withholds only its own bytes: an unchanged shared limit, or a script absent from both sides, is an `unchanged_limits` entry, and any other makes the comparison `partial` with every grant still compared. A script whose working-tree bytes differ only by a checkout's line-ending conversion is `unchanged_not_proven`, not a row. A selected hook whose script is not resolved (an interpreter wrapper, a relative path, a conditional expansion, a compound command or a malformed group) is named as a `script_not_resolved` coverage item while the change could touch it, never a row. A pre-#702 baseline is incomparable only for a hook that now binds a script. `check` and a provided diff (`check --diff`) leave out a selected script the change does not touch, so an untouched missing, ignored or linked script changes no `check` decision; a touched one is routed for review, and compared from the diff where the diff also touches its declaration. Verification binds ignored inputs and missing-path observations, and refuses current authority for unsafe reads. Mode-only changes and recursive dependencies are not compared. (#702) +- Note a literal, unconditional inline `PreToolUse` allow on a broad Claude Code hook matcher. Diff, verifier host comparison and PR comments share the note; script and unsupported command behavior remains a named inventory limit. Direction, severity, widening signals and control decisions are unchanged. See the [bounded grammar](../engineering/inline-hook-allow-notes.md). (#826) +- Add factual mutable-source notes to supported MCP launcher changes, including established pinned-to-mutable transitions. Keep package text within existing redaction rules; do not change direction, severity or control decisions (#825). +- Add conditional review guidance for supported Claude Code shell-permission changes: a concrete intent question, human choices and a declaration comparison target, shared by CLI, maintained JSON and advisory PR output. Conflicting, redacted or incomplete evidence withholds specific choices. Existing control permissions, rows and release decisions are unchanged. See the [case mapping](../engineering/permission-review-guidance.md). (#839) + +- Correct false widening markers on hook/MCP edits, disabled plugins, more restrictive permission modes and enabled sandboxes. Explicit enablement (including a Codex app, enabled by default), added or re-pointed Claude Code marketplaces, less restrictive modes, documented sandbox loosenings (disabling it, `excludedCommands`, `autoAllowBashIfSandboxed`, `enableWeakerNestedSandbox`, `allowedDomains`, Codex `writable_roots` and `web_search: live`) and added loaded hooks/MCP servers still expand, including a handler added to an event that already had one. An edit whose direction is not established is a row whose `why` ends `authority direction is unknown`, and the Claude Code Stop hook names it under its own heading instead of staying silent. Setting replacements use their full comparison context so another setting's expansion cannot mark a tightening; an unknown or ambiguous predecessor is read as a new declaration, so it cannot hide `bypassPermissions`. Claude Code plugin enablement no longer coerces malformed values to booleans. Existing check decisions and severities remain unchanged. See the [migration note](../../STABILITY.md#grant-direction-820) and [before/after direction benchmark](benchmark/host-config/direction-replay-820.md). (#820) +- A changed Claude Code `git push` prefix allow now names documented command forms that the narrower deny prefixes in the same source do not cover, including a flag placed after the remote (`git push origin main --force`). The shared CLI/JSON/PR explanation considers unchanged head denies too, declines unsupported shapes and preserves all direction, severity and gate decisions; `check`, whose rows redact rule arguments, omits it. It does not claim those commands will execute without prompting. (#829) +- Rate documented arbitrary-code Bash launcher prefixes with trailing wildcards as critical/admin, and any Bash allow rule the lattice decides is wider than one (`Bash(python3 *)`, `Bash(npx*)`), so widening a rule cannot lower its rating; route newly granted forms through the existing wildcard block check, and count them in the `audit --host` Markdown warning. Show them in check evidence in table text only; exact commands and containment comparisons stay unchanged. `check` no longer reports respelling a Bash rule (`Bash(npx:*)` to `Bash(npx *)`) as a new grant (#824). +- Skill frontmatter `metadata` no longer refuses the entire host comparison. A map's boolean, numeric, list and nested values are preserved, and a `metadata` that is not a map, which Claude Code drops, is digested as written, as an undocumented key is (#730). Claude Code documents a free-form map; this static structure reader does not validate portability to hosts that require string values. Metadata changes remain in the structure digest, and malformed frontmatter, ambiguous YAML and a metadata key that is not a string, at any depth, still refuse. A permission widening beside such metadata is now visible in text and JSON, with matching coverage. See the [migration note](../../STABILITY.md#skill-metadata-848). (#848) +- Version-based Action installs log the existing installed engine content + digest, including host-only runs. It can be compared with a local verifier's + `engine_distribution_sha256`; it is not the wheel ZIP hash. An engine that + cannot compute it, such as any release before `1.0.0`, logs a warning and + the install continues. The explicit wheel-and-hash install route is + unchanged. (#855) +- The PR-comment fallback uses separated Markdown paragraphs and headings, + so fork reviews render correctly in the job summary after a 403 or 404. + Successful comment updates and error handling are unchanged. (#856) +- Host-only Action summaries no longer invent empty scan status or zero scan + severity counts. Artifact links name only existing files, and annotation + metadata omits `source_report` when no scan report was produced. (#854) +- Workflow comparison values label their aggregate access (for example, + `access: write`) so it is not mistaken for an individual token scope. + Direction, severity, explanation and widening signals are unchanged. (#859) +- Host-only verifier headlines and `control.reason` label their raw count as + rows, so a two-row replacement no longer contradicts the review summary + that correctly calls it one change. Partial comparisons use the same + wording; decisions and control routes are unchanged. (#857) +- An unrelated permission rule for another tool no longer splits a decided + replacement into separate review changes. Adding `Read(src/**)` beside + `Bash(npm test *)` → `Bash(npm *)` preserves the paired widening on the + shared diff, verifier and check routes. Multiple candidates for the same + tool or MCP server remain unpaired. MCP server replacements and + narrow-while-denying edits also retain their direction beside unrelated + tool changes. Unpaired removals no longer claim a permission loss when + an added allow rule in the same source decidedly covers them and no deny + or ask rule for that tool arrives in the same edit. (#858) +- Fix a fail-open comparison where a narrowing in `settings.json` suppressed + the expansion signal and ⚠ for the same rule newly allowed in + `settings.local.json`. Suppression now stays within its source, restoring + that independent grant in `expansion_signals` and review widening counts. (#858) + +- Move the published-release pins, examples and adoption prompts to `v1.1.0` (contract 40) now that it is published, re-capture the README and quickstart `diff` answers from the published `1.1.0`, and re-measure the pilot ledger's Route H dry run on it. No schema or contract change. (#778) +- A host comparison names the changed inputs it does not read, so a zero-row result is not read as covering them. (#821; slice 2 of #812) + - **The problem.** A pull request that added a Cursor plugin's `mcp.json`, removed a `beforeShellExecution` guard from `.cursor/hooks.json`, gave a dotfiles package's `claude/.claude/settings.json` `Bash(*)`, or moved a marketplace plugin's pinned `sha` printed `No static host-grant changes detected`, as a docs-only change does. Re-running a 23-PR public corpus after #812 found 11 of 23 pull requests were such coverage gaps: 0 of the 9 comparable zero-row results named the changed relevant file, and 4 of them named a file the pull request did not touch while omitting the one it did. + - **What is named.** `diff`, `verify` and the manifest-free PR comment list, under `What this run established`, each path in the comparison's own changed-file set that a bounded, documented candidate rule recognises and no reader of this entry read: `mcp.json` in a plugin directory, a plugin manifest's `mcpServers`, a Codex, Cursor or Copilot manifest's `hooks` and the hook files it names, a manifest or marketplace that does not parse, `.cursor/hooks.json`, host settings below the repository root, and an external marketplace plugin source — `plugins/demo/mcp.json (cursor): added, not read by this entry: MCP configuration in a plugin directory; no row, and loading is not established`. An external source names what it now points at, redacted, and is never fetched. The block's first line says the list includes them. Ordinary documentation, an unrelated `*.json` and an unchanged candidate name nothing. + - **What it is not.** Never a row, a widening, a `check` violation or a claim that a host loads the file. Nothing is fetched or run, only plugin manifests and marketplaces are read, and at most 32 candidates are examined; the rest, and any whose rule needed a file that was not read or did not parse, are counted as not examined, on a line that names both causes. The rules are listed in `docs/host-boundary-support.md` under *Changed inputs named but not read*. + - **JSON.** A `changed_not_read` coverage item with its `candidate` rule, ranked right after the blocking limits and inside the existing cap; `read_sources_only` is `false` while one is named, and `unread_candidates` / `unread_candidates_not_examined` say whether the change set was examined. Verifier `0.20` → `0.21`, capability diff `0.3` → `0.4`, runtime contract 40 → 41; host-grants does not move for it (#819 and #823, below, move it to `0.7` in the same contract), and `minimum_control_contract_version` stays `21`. A `0.20` verifier reads with the search not recorded. + - **One route moves, on `verify` and `verify --preview` alike.** A manifest-free `verify` whose only host-relevant change is such an input, or a changed candidate it counts as not examined, now publishes the host comparison (advisory, exit `0`) instead of the setup route, which said nothing about the change. `verify --preview` moves the same way: its next action is now `discover` (`audit --host`) with the comparison published, where it was `initialize` (`init --write`) with none. That includes an agent-related workspace, as it already did when the change edited a host file this entry reads. The pilot ledger's source-tree column was re-measured for contract 41. Rows, digests, baselines, `audit --host`, `check` and the benchmark replays are unchanged. +- A hook row now names what changed in the hook, and an MCP row names a change to the server's launch arguments. Before, `diff`, `verify`, the manifest-free PR comment and `check` printed `PostToolUse → PostToolUse` whether the edit was to the hook's matcher, its command or its timeout, and an MCP server whose version pin moved from `example-mcp-server@1.2.3` to `@latest` read `docs: no difference in the command name npx, env key names or header key names; the change is in a detail this output does not show, such as the command's path or arguments`: the grants carried none of it, and only `config_sha256` saw the edit. On five of 23 public pull requests measured on 2026-09-15, the hook rows showed only event names. (#819, slice 2 of #795; direction is #820, an unpinned-launch note #825) + - **The rows a reviewer reads:** `PostToolUse: matcher Edit → Edit|Write|Bash`, `PostToolUse: command changed (lint.sh sha256:d075f5f4772e → curl sha256:a510416cbecc)`, `PostToolUse: timeout 10 → 600` and `docs: package example-mcp-server@1.2.3 → example-mcp-server@latest`, in `diff`, `verify` text, the PR comment and `check` text, and in `review.changes[].change` in `diff --json` and `verifier.json`; any other launch argument edit reads `launch arguments changed (sha256:… → sha256:…)`, a digest printed as its first twelve hex digits. A timeout written as text prints quoted, so it never reads as a number or a boolean: `timeout 5 → "5"`, `timeout true → "true"`. With several handlers under one event the entry names which one (`handler 2 timeout 5 → 50`), and an added or removed handler is listed as such. The same published handlers in another order read `the published handlers in a different order; a detail this output does not show may also differ, such as …`, never that they are the same handlers. An added or removed hook names its handlers, `SessionEnd (command cleanup.sh sha256:18d2c7ec39bc)`, and an added MCP server its package. When none of the published fields differ, the entry says the change is in a detail it does not show — another hook setting such as `async`, or a redacted or shortened matcher or timeout; for an MCP server, the command's path or another setting such as `cwd` — instead of repeating the same values. The row's direction, severity, `why` and loading basis are unchanged: plugin-selected (#714) and Codex hooks read as before, and no entry claims a direction (#820), runtime loading or what a command does. + - **No command or argument text is published, on any surface.** A hook command is published as its executable's name — the last path segment of its first word, only when that is a plain token (`[A-Za-z0-9._+-]`, at most 80 characters) that no redaction rule rewrites, not a shell reserved word such as `if` and not part of a URL (a first word holding `://`), otherwise `` — and the SHA-256 of the whole command as `config_sha256`'s input holds it. An MCP server's arguments are published as at most one package specification of a strict shape (npm `name@version` or `@scope/name@version` with a version of two or three numeric parts, a `^`/`~` range on one or a common dist-tag; PyPI `name==version` with a version of two or more parts; an OCI image reference with a path and a tag or `sha256` digest) that no redaction rule rewrites and that follows no flag but a package runner's own (`-y`, `--from`, `--rm` …), and the SHA-256 of every argument with the package replaced by a marker and its position digested beside them. The matcher passes the #802 published-label redaction and is cut at 120 characters, and a matcher longer than 1,024 characters as `config_sha256`'s input holds it is ``, never redacted or cut; a timeout is the number or boolean as declared, an over-80-digit integer's cut digits, a plain-token string, or `` for anything else, a non-finite float among them. Earlier drafts published redacted command words, and each of four review cycles found a credential the redaction rules missed inside free-form shell text; publishing none of it closes the class. + - **Host-grants `0.7`:** a hook grant adds `handlers[]` — each handler's group `matcher`, its `command` `{executable, sha256}` and its `timeout` — and `omitted_handlers` (at most sixteen handlers are listed); an MCP server grant adds `package` and `args_sha256`, both `null` when no `args` is declared. A declaration outside the documented shape (a list of matcher groups whose `hooks` are objects, each `command` a string where one is declared) publishes `handlers: null`, and its row says the matcher, command and timeout are not shown because `the declaration is not a list of matcher groups whose hooks are objects and whose commands are strings`; when only one side is outside it, the row names that side (`base` or `head`) and lists the other side's handlers. Runtime contract v41. + - **Saved baselines hold none of it:** `audit --host --save-baseline` writes each hook and MCP grant without `handlers`, `package` or `args_sha256`, in either scope, so nothing read from `~/.claude/settings.json`, `~/.cursor/mcp.json`, managed settings or a git-ignored `.claude/settings.local.json` reaches the committed baseline. A saved `0.7` baseline's grants are the ones a `0.6` baseline holds; no comparison, row or digest read them, and `inventory_sha256` is unchanged. + - **The PR comment keeps every line 1.1.0 kept:** its 6,000-character bound cuts at the first line that does not fit, so long entries could hide every row after them, the coverage block, the change count, the review question, the reproduction and the advisory. The lines 1.1.0 printed now get their room first, the coverage block included, and the entries only what is left: an entry is printed whole when the whole comment fits; otherwise every longer entry is cut to the widest length of at least 60 characters at which it does, ending in `…`; and where not even that fits, entries are printed in their shortest form, longest first: a field-level difference cut after its name (`PreToolUse: …`), an added or removed grant as its row (`(absent) → PreToolUse`). No entry in that form is longer than the one 1.1.0 printed, and a permission rule's entry is never shortened. One line after the rows, not one per entry, says entries were shortened and that `verifier.json` holds each whole. On a pull request that moves two hook scripts under 14 events, 1.1.0's comment held every row, the coverage block, the review question, the reproduction and the advisory, and so does this one, within the same 6,000 characters. `verifier.json` and the other routes keep every entry whole. A comment with no readiness report now points to `verifier.json` when it omits detail, not to a `report.md` that route does not write. + - **Unchanged:** grant equality and every inventory digest leave the new members out, so a change is a row exactly when it was one before, through `config_sha256`; a value the digest's own input redacts (after `--token`, `--api-key` or `--password`, a `--password=…` value, an `X-Api-Key:` header value, a URL's path) moves no published digest, so a change confined to it is no row, as on 1.1.0; every row value, the row count, `check`'s boundary result and the control envelope's `capability_rows` publish what they did; verifier `0.21` and capability diff `0.4` do not move for it; the host-config and cold-start benchmark replays reproduce their run-of-record scores. The digest's credential-assignment rule gained a lookahead that removes its quadratic time on a long run of name characters and matches exactly what it matched, so every `config_sha256` is unchanged. Re-running `diff --json` on the 80 vendored benchmark cases with the prepared `1.1.0` commit and this tree on 2026-09-23 gave byte-identical rows on all 80; 23 entries on 22 cases changed, and each of the 7 changed hook or MCP entries (six repositories; one is vendored in both benchmarks) that read `PreToolUse → PreToolUse` or `no difference in the command name …` now names the field that changed, such as `mcp-outline: package mcp-outline==1.10.0 → mcp-outline==1.10.1` or `PreToolUse: handler 2 timeout 30 → 120`. + - **Compatibility:** a `0.6` baseline stays comparable with no new row or reason, and `audit --host --save-baseline` may now replace it; an older one is still refused, as before. Validators pinned to the `0.6` schemas reject a `0.7` inventory, baseline or drift payload; the `0.6` files stay published. See the [migration note](../../STABILITY.md#hook-mcp-detail-fields-819). + +- Read how a coding agent is launched inside a CI workflow. A documented agent action's permission inputs (`anthropics/claude-code-action`, `anthropics/claude-code-base-action`, `openai/codex-action`), the permission flags of a `run:` that is one plain `claude -p` or `codex exec` command, and each `actions/checkout` step's `with.ref` are listed on the workflow grant and compared as text, never executed. Changing plain `claude_args` from `--allowedTools Read` to `--permission-mode bypassPermissions --allowedTools Bash`, adding a `claude -p --permission-mode acceptEdits Summarize` step, or checking out `${{ github.event.pull_request.head.sha }}` in a `pull_request_target` job each gave no row and now gives one naming `job/step` and both values in `diff`, `check`, manifest-free `verify` and the PR comment. Only a documented rule a job's launches gain widens — bypassed permission checks (a flag, or JSON settings whose `defaultMode` is `bypassPermissions`), a bypassed or `danger-full-access` sandbox (`permission-profile: :danger-full-access` included), `safety-strategy: unsafe`, or a user gate opened to `*` — and every other edit is `changed`; a workflow row that runs an agent now names the untrusted-input trigger, write scopes, secrets and pull request checkout beside each agent step, so a move to `issue_comment` with `pull-requests: write` says which agent step it reaches. Shell is not parsed: a `run:` is read only when it is one line of plain words (no quote, expansion, operator, redirection, comment or continuation) run by `bash` or `sh` (a `shell:` template only when it runs the script alone, never `bash -c '…' {0}`), and `claude_args` / `codex-args` only when they are a plain list of words with no `--settings` or `--mcp-config` flag. Any other `run:` that mentions `claude` or `codex` is listed in `unread_agent_runs` and named as a non-blocking limit in `audit --host`; it publishes none of its text, is never compared and gives no row, and the workflow's line in `What this run established` names what its grant does not read instead of env values and `apiKeyHelper`. Any other argument input, the #823 reproduction's quoted `--allowedTools "Read"` included, is `unread_arguments`: compared by a digest, so editing it is a `changed` row, and read for no rule. The rules each launch meets are read from the declared text and published as `widening_rules`, so redaction never hides one; one a launch already met in a job it left (a renamed job, a moved step) is moved, not gained, though never while that job keeps any unread step of that agent or a launch of it with an unread input the rule is read from, nor while any other job but the receiving one holds more of them than before, so renaming a job while quoting its launch, as another job adds that launch plainly, is a widening; one the job's unread step or unread input may already have met is named, not claimed; and an unread step that remains takes no gain from another launch. A launch that becomes one this audit does not read (`npx`, quoting, `codex` options before `exec`) is worded as no longer declaring a launch this audit reads, never as no longer starting an agent. A JSON object in a `settings` or `mcp_config` input publishes its shape and none of its free text: key names, with `env` and `headers` values and `apiKeyHelper` redacted and every other string a `` digest, except the strings a host reader publishes (a permission rule, a documented setting's value, an MCP server's command name and URL host), so an MCP server's arguments and a hook's command are compared but never published, and a value there that is neither a JSON object nor a plain file path publishes only a digest; a codex `--config` override publishes its key and, except for the sandbox, permission profile, approval policy and model, only a digest or `` for its value; a URL elsewhere publishes its scheme and host; other credential-shaped text in a setting, prose such as "never print bearer tokens" included, is published redacted, compared as published and named as a non-blocking limit, while a redacted checkout ref refuses as a redacted step reference does. Every published value goes through the #802 label redaction. Host-grants inventory, baseline and drift move to `0.7`, because `0.6` shipped in 1.1.0, and runtime contract 41, unpublished before 1.2.0 (#821), is extended in place; a `0.4`–`0.6` baseline holding a workflow grant is incomparable (`baseline_workflow_agent_launches_unavailable`), and the verifier `0.21` and capability diff `0.4` #821 minted do not move here. See the `Migration Note: 1.2.0` entry in [`STABILITY.md`](../../STABILITY.md#workflow-agent-launches-contract-v41-823). (#823) + +- A plugin directory that cannot be compared no longer hides the host changes outside it. (#808) + - **The problem.** A pull request that broke `plugins/demo/.claude-plugin/plugin.json` and also dropped a `deny` rule from `.claude/settings.json` printed `Cannot compare against main: head_inventory_incomplete` and no row on `diff`, `verify` and the manifest-free PR comment, where the published `1.0.0` showed the removed denial. The plugin-reference limit #714 introduced refused the whole comparison, including files that plugin cannot reach. + - **What changes.** When every blocking limit that refused a comparison is a plugin-reference limit bounded by its plugin directory — a reference is followed only inside it, so that is all it can hide — and nothing outside the directory depends on it, the comparison is `partial`: the directory is left uncompared on both sides and named, and the rows outside it are published. `diff` opens with `Partial comparison against main (…) -> working tree: head_inventory_incomplete` and `Not compared: plugins/demo, a plugin directory this entry could not read completely, …` before any row; `verify` and the PR comment open with `Host capability comparison partial: …` and the same line. A partial result with no row says it is not a no-change answer and never prints `No static host-grant changes detected`. + - **Where it still refuses.** Any other limit that is not an unchanged one (an unreadable settings file, an instruction file whose structure could not be established, a link #700 does not read through), a reference that leaves its plugin, a plugin at the repository root or holding `.claude/settings.json`, a marketplace elsewhere declaring inline hooks for that plugin, or nothing read outside the directory: `incomparable`, no row, exactly as before. A hook file another plugin also selects is withheld with the directory. + - **JSON.** `comparison_status` adds `partial`, with the same `incomparable_reasons` the refusal would have named, and `rows`, `review` and `unchanged_limits` for what was compared. The reserved `coverage.items[].scope` now names the withheld directory on each `blocking_limit` item of a partial comparison, and is `null` everywhere else. In a partial comparison a changed project settings file with no row of its own is `changed_without_rows`, never `changed_without_grant_change`: the hooks whose loading basis it decides are not all compared. Verifier `0.21`, capability diff `0.4` and contract 41 are extended in place; a `0.20` verifier that claims a partial comparison or a scope is refused. + - **What does not move.** Every decision and route. `verify`'s control state, permissions, next action and exit code are the incomplete comparison's; the control envelope's `capability_rows` projects a partial comparison as `incomparable` with no rows; `check` names no scope, so it refuses its comparison as before, and the #808 fixture still gives `require_review` with `HOST-PERMISSION-DENY-REMOVED`; the Stop hook still says to treat the change as unreviewed. `audit --host`, digests, baselines and drift payloads are unchanged (host-grants stays `0.6`), and the host-config and cold-start benchmark replays reproduce their run-of-record scores. See the `STABILITY.md` migration note. + +- A Claude Code setting that disables prompts or approves project MCP servers + now carries one rating on every surface that names it. One table in the + engine rates each value, with its documented basis, and the `audit --host` + grant, the `diff`, `verify` and `check` rows, `check`'s violation and + `verify`'s finding all read it. `enableAllProjectMcpServers: true` and + `skipDangerousModePermissionPrompt: true` are no longer recorded as keys that + "could not be parsed" at `medium`: like `defaultMode: bypassPermissions`, + whose grant and row now read `critical`, they are `critical` everywhere and + `check` blocks with `SHIP-HOST-BOUNDARY-PERMISSION-WILDCARD-ALLOW`. + `defaultMode: dontAsk`, which Claude Code documents as denying whatever no + allow rule permits, is `medium` everywhere and no longer blocks; it is + reviewed through `SHIP-HOST-BOUNDARY-PERMISSION-ALLOW-EXPANDED`. `acceptEdits` + and `auto` grants and rows move from `medium` to the `high` `check` already + gave them, and `plan`, `default` and the other modelled settings move in + `check` to the `medium` their rows show. Rows name the setting and value + (`enableAllProjectMcpServers: true`, `defaultMode: dontAsk`, + `approval_policy: never`) instead of `True` or `dontAsk` alone, and a Claude + Code setting's row says what the value does. `enabledMcpjsonServers` is read + as one `high` grant and row per approved server, and an entry that names no + server is kept whole at `high` rather than dropped; `disabledMcpjsonServers` + stays unread. `check` evidence carries a setting's value as its grant + publishes it, with credentials redacted and at most 200 characters. A value + moved between `permissions` and the top level is one the change sets, so a + top-level `defaultMode: bypassPermissions` moved into `permissions`, where + Claude Code reads it, still blocks. No schema, contract or check id moves; + which check id fires for these values, and their decisions, do. See the + `STABILITY.md` migration note. (#827) + +- `check` and `verify` no longer pass a change to a Claude Code plugin's hook + that the repository's own project settings enable just because the plugin + keeps the hook outside the registry paths. 1.1.0 published such a hook as + `execute`/`high` and its change as a `widened`, expanding row, yet `check` + gave `allow` with `merge` permitted beside it, and a `verify` with a manifest + gave `passed` / `mergeable`, while the same hook at + `.claude/hooks/hooks.json` got `require_review` and `review_required`. + - **Routed.** A changed hook file an enabled plugin selects, such as + `plugins/demo/cfg/hooks.json`, and a plugin manifest or marketplace whose + inline hooks an enabled plugin loads, now reach the existing + protected-surface review: `SHIP-AGENT-BOUNDARY-PROTECTED-SURFACE-UNCLASSIFIED` + (evidence `hook_loading_basis: project_enabled_plugin`). In `check` that is + `require_review` with `merge` withheld. In a `verify` with a manifest it is + a finding that moves the release gate to `review_required` and the merge + verdict to `human_review_required`, and the PR comment says so. The route + needs both the file's name, one the plugin hook reader opens, and the + reader's finding, in the base or the head, that an enabled plugin loads + hooks from it. `check` and `verify` read both sides the same way, so a hook + file deleted with its reference is routed by both, and a hook a plugin only + selects stays unrouted, as #714 made it. + - **Unread.** A changed hook file an enabled plugin selects under a name the + static reader does not follow (`cfg/lifecycle.json`, not `hooks.json` or + `-hooks.json`), or inside a directory the walk skips, holds hooks the + host loads and nothing read. It is now incomplete input, + `SHIP-AGENT-BOUNDARY-INPUT-INCOMPLETE`, in both: `human_review_required`, + where 1.1.0 gave `allow` with complete input. + - **Broken heads.** A head that leaves a routed hook file or plugin manifest + unreadable is incomplete input, as at a registry path. A marketplace the + head makes unparseable is different, because a marketplace limit never + blocks: the route still gives `require_review`, input stays `complete`, and + the comparison shows the marketplace's inline hooks as `removed`. + - **Unchanged.** Rows, the host inventory, `diff` and the trigger catalog. + A change only to a selector, a manifest's `hooks` reference or a + marketplace entry, that makes an enabled plugin load an existing, + unchanged hook file is still `allow` beside that file's `added` or + `widened` row. `check` gives the same when the selected file sits at a + registry path, so that is a separate limit. No schema, contract or check + id moves; what moves is when two existing check ids fire. See the + `STABILITY.md` migration note. (#809) + +- `verify --preview` in a repository with a `shipgate.yaml` no longer sends + the caller to a human for missing evidence. `verify --preview --json` there + answered `agent_action_required` with the exact `verify` command, while + `verify --preview --format control` answered `human_review_required` with + "The recorded source and dependency inputs are no longer current: input + directory capture is unavailable; re-run verification", and every + `agent control` refresh exited `4` with `workspace_unverifiable`. Re-running + reproduced it, with the default, an in-repository and a sibling `--out` + alike, on `1.0.0` and `1.1.0`. + - **The cause.** A manifest lets the preview record the verification plan a + `verify` would run, and the preview's pointer bound that plan as its + currency evidence. A preview runs no adapter, so none of the plan's inputs + was captured and the plan has no input-directory census, which every + reader refuses. + - **Now.** A preview's pointer binds its verifier route and never the plan, + and is read against the working tree it was run on, exactly as a + manifest-free preview's already was: `--format control` prints the + `--json` state and command, and `agent control` returns them, with the same + `current_control_id`, until a tracked edit, a new untracked file or a + removal refuses it as `workspace_changed`. Restoring the tree makes it + current again. The plan is still written, and `verify-run.json` still + embeds it. + - **Still refused.** A pointer that binds a plan without its census, such + as one a `1.1.0` configured preview left in a reports directory, is + `workspace_unverifiable` as before, with a `verify` command as its next + action; re-running the preview replaces it. The plan gains no census, so + `verification worker` still refuses to replay it. Nothing else reads a + missing census as complete. + - **A tree that cannot be read stays refused.** Under Git configuration the + worktree readers refuse (#813), the overlay a plan-less pointer binds + cannot be read, and such a pointer declared no snapshot at all, so the + refresh compared HEAD alone: a manifest-free preview stayed current over + any later edit, and a configured one would have too once it stopped + binding its plan. Every pointer published inside a repository without a + plan — a preview, or a `verify` that stopped before building one at a + `--config` or `--head` that does not exist — now declares the worktree + snapshot, so it is refused as `workspace_unverifiable` with the cause + first and a `review` next action, like every other pointer there, and the + run's own `--format control` says `human_review_required`, manifest or + not. `--json` is unchanged. + - **One refusal names the tree.** A preview refused as `workspace_changed` + now names up to three paths that differ from HEAD, redacted, instead of + saying the paths it was read from no longer had their content, which was + false for a tracked edit or a new file after a preview of a clean tree. + - No schema, contract, member, error kind, refusal code or exit code moves. + See the `STABILITY.md` migration note. (#807) + +- An unchanged instruction file or plugin manifest reached through an in-tree + link no longer refuses the whole host comparison. A `SKILL.md` whose + `metadata` has a key that is not a string, which this entry cannot resolve, + is named as an unchanged limit when the change leaves it alone (#721). Read + through `.claude/skills -> ../.agents/skills`, a per-skill link or a file + link, the same untouched file made `diff`, `verify` and the manifest-free PR + comment print `Cannot compare against HEAD~1: base_inventory_incomplete; + head_inventory_incomplete` and no row, hiding a removed `deny` rule beside it. + - **The cause.** The unchanged proof asked whether the path the link is read + under, `.claude/skills/review/SKILL.md`, was one regular file in Git, and a + path through a link never is. + - **Now.** The proof follows the link as the reader does (#700), from Git + tree entries, and holds only when both are unchanged: the link, a link at + the same path with the same text at each link on the way, and the file it + lands on, the same blob at the same in-tree path. A working-tree head is + read without following any link, each link by its own text and the file by + its unfiltered hash. The limit is then named and the rest compared, exactly + as when the skill sits at its own path: the issue's reproduction shows + `⚠ low removed claude-code .claude/settings.json`, + `deny: Bash(curl *) → gone` for the directory link and the per-skill link + alike. + - **Still refused.** A skill added or edited behind the link; the link + retargeted, even to an identical copy, or its text rewritten to land on the + same file; a link replaced by a directory holding the same bytes, or the + reverse; and every link the reader does not read through (dangling, + looping, absolute, escaping, past eight links, or inside a linked + directory), which stays `unreadable` and is never an unchanged limit. + Metadata is not coerced: the key `1` is not read as the string `"1"`, so + it stays an `unsupported` structure, and the value `internal: true` is read + as written, with no limit (#811, #848). + - **A plugin manifest behind a link.** The same proof covers a + plugin-reference limit, such as a `parse_failed` + `plugins/demo/.claude-plugin/plugin.json` that is a file link to + `../../../vendor/plugin.json`. Unchanged on both sides, it is now named in + `unchanged_limits` like one at its own path, so a comparison that was + `partial` only because of it (#808, `Not compared: plugins/demo`) is + `comparable` on `diff`, `verify` and the PR comment, with the directory + compared; `1.1.0` refused it. Edited behind the link, it stays `partial`. + - **`check`.** Its boundary result still cannot name a limit, so it does + what it does for the same limit at its own path. An instruction file's + limit, which it cannot leave out, now refuses its comparison with + `unchanged_limits_not_representable` instead of + `base_inventory_incomplete` / `head_inventory_incomplete`, and its rows + stay empty. A plugin-reference limit both sides share on an unchanged + source it leaves out, as it has since #714, so behind a link it now + compares and publishes the rows it finds where it refused with no row: the + plugin manifest above gives `comparable` with the removed `deny` row, in + the boundary result and in the control envelope's `capability_rows`. + Edited behind the link, the limit is kept and `check` refuses as before. + Its decision, violations and control state do not move. + - No schema, contract, member, reason code or check id is added, and + host-grants stays `0.6`. See the `STABILITY.md` migration note. (#822) + +- Read a materialized Git tree's blobs through a few `git cat-file --batch` processes instead of one `git cat-file blob` per file. `diff --application --scope .` materializes the whole tree on both sides, so its run time grew with the file count: measured on 2026-09-25 with #877's root-scope reach, it took 412 s on TencentCloud/CubeSandbox (3,896 files) and 203 s on dlt-hub/dlt (2,042 files), and now takes 46 s and 45 s with identical rows; materializing one side of CubeSandbox went from about 180 s to under 5 s. One `cat-file --batch-check` first types and sizes every object, so a missing object or one that is not a blob refuses with a configuration error before any content is read; the content is then read in batches of at most 64 MiB, so a large tree is never held in memory at once. Every blob is still hashed against its tree entry's object ID before it is written, and the path, containment, link-order and final digest checks are unchanged. The link texts read to resolve in-tree link targets are batched the same way. No output, schema or contract changes. (#686 cost class) diff --git a/docs/design-partner-pilot-results.md b/docs/design-partner-pilot-results.md index ec7f614b..1015b6a4 100644 --- a/docs/design-partner-pilot-results.md +++ b/docs/design-partner-pilot-results.md @@ -64,7 +64,8 @@ allow list from `Bash(npm test)` / `Read(src/**)` to `Bash(*)` / `Read(**)` / capability-change class this pilot exists to observe. **Published build measured: `1.1.0`.** Preview measured: -`0.16.0+preview.20260903.gb61aca7`. Source tree: `1.1.0`, runtime contract 41. +`0.16.0+preview.20260903.gb61aca7`. Source tree: `1.2.0`, runtime contract 41: only this label moved with the +1.2.0 version bump, and the source-tree cells below were not re-measured for it. The released and source-tree columns were both rerun on 2026-09-22, after `v1.1.0` was published: the released column from `pip install agents-shipgate==1.1.0` in a clean virtualenv outside any checkout (the wheel diff --git a/docs/determinism-boundary.json b/docs/determinism-boundary.json index 931852c7..3a64120f 100644 --- a/docs/determinism-boundary.json +++ b/docs/determinism-boundary.json @@ -7,7 +7,7 @@ "literal_registration": "Each action is written out where the parser can read it: a decorated function, a tool class, a literal list of tool objects, a config entry naming the tool." }, "description": "What each built-in input can establish, per declaration shape: the extraction-confidence ceiling and what that ceiling means for a release verdict. Generated from agents_shipgate.inputs.coverage.build_boundary_matrix(). Do not edit by hand.", - "generated_for_version": "1.1.0", + "generated_for_version": "1.2.0", "outcomes": { "low_confidence": "Every action from this route raises `low_confidence_tool` and either `unattested_surface` (when its set was enumerated) or `incomplete_surface` (when it was not), so none can be pass-eligible. Only the latter adds an exclusion-ledger row for an unread remainder. Semantic evidence gaps are zero-tolerance, so **one** such action is already enough to put the run below the evidence threshold and withhold a verdict. A reviewed tool inventory is the route out.", "not_applicable": "No such declaration exists for this input.", diff --git a/docs/determinism-boundary.md b/docs/determinism-boundary.md index 9004692b..bda43a7d 100644 --- a/docs/determinism-boundary.md +++ b/docs/determinism-boundary.md @@ -426,7 +426,7 @@ boundary rather than an accident of it. ### Which release this describes -This page describes **agents-shipgate 1.1.0**, and its rows move as the adapters do — `schema_version` (`shipgate.determinism_boundary/v1`) versions the shape of the machine-readable companion, not the routes. +This page describes **agents-shipgate 1.2.0**, and its rows move as the adapters do — `schema_version` (`shipgate.determinism_boundary/v1`) versions the shape of the machine-readable companion, not the routes. If you arrived from a link in a stored report, check that version against the scanner that produced the report before trusting a row: a boundary is only a specification of the release it was generated from. Every released version is tagged, so the matrix your scanner implemented is at `https://github.com/ThreeMoonsLab/agents-shipgate/blob/v/docs/determinism-boundary.md`. diff --git a/docs/engineering/permission-review-guidance.md b/docs/engineering/permission-review-guidance.md index 209973d8..64edb74f 100644 --- a/docs/engineering/permission-review-guidance.md +++ b/docs/engineering/permission-review-guidance.md @@ -61,8 +61,8 @@ coverage, authority and evidence first; guidance uses only remaining room. A comment with no room retains its existing evidence pointer. All source-derived strings use the existing single-line/Markdown literal display protections. -Verifier `0.21`, capability diff `0.4` and contract v41 are unreleased and are -extended in place. `guidance: null` means the change is not a supported shell +Verifier `0.21`, capability diff `0.4` and contract v41 had not shipped before +1.2.0 and were extended in place. `guidance: null` means the change is not a supported shell case, so the renderers print nothing for it; it does not establish absence of risk. Frozen verifier `0.20` is unchanged and cannot claim the new field. Current readers accept historical artifacts whose changes carry no `guidance` diff --git a/docs/examples/capability-lock.v0.8.example.json b/docs/examples/capability-lock.v0.8.example.json index 4135816a..cdb3a13b 100644 --- a/docs/examples/capability-lock.v0.8.example.json +++ b/docs/examples/capability-lock.v0.8.example.json @@ -1,7 +1,7 @@ { "capability_lock_schema_version": "0.8", "experimental": false, - "cli_version": "1.1.0", + "cli_version": "1.2.0", "source": { "config_path": "shipgate.yaml", "manifest_dir": ".", diff --git a/docs/quickstart.md b/docs/quickstart.md index bf9819a3..943ac3d5 100644 --- a/docs/quickstart.md +++ b/docs/quickstart.md @@ -130,7 +130,7 @@ comparison facts and coverage, but does not append the conditional permission review guidance shown in the change example, nor the `launch source is mutable` note on the added MCP server. That note identifies its unversioned `npx` package and changes neither severity nor the widening count. The source tree still -reports version `1.1.0`; its version string alone is not published-wheel +reports version `1.2.0`; its version string alone is not published-wheel provenance. On the remote's `main`, `.claude/settings.json` allows `Bash(npm test:*)` and denies `Bash(rm -rf:*)`, and `.mcp.json` configures one server, `docs`. The PR branch allows diff --git a/llms-full.txt b/llms-full.txt index 4a1d4096..df64fa0c 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -2350,7 +2350,7 @@ Downstream repos generated with `.shipgate/agent-contract.json`. - Latest release: `v1.1.0` -- In-tree runtime: `1.1.0` — see [pyproject.toml](../pyproject.toml) +- In-tree runtime: `1.2.0` — see [pyproject.toml](../pyproject.toml) - Runtime contract: `41` (minimum control contract: `21`) - Current report schema: `1.0`, frozen, superseding `0.43` — [`docs/report-schema.v1.0.json`](report-schema.v1.0.json); the `1.x` rules are in [`docs/report-1-0-contract.md`](report-1-0-contract.md) - Current packet schema: `0.18` — [`docs/packet-schema.v0.18.json`](packet-schema.v0.18.json) diff --git a/llms.txt b/llms.txt index 1f088484..7003f6a8 100644 --- a/llms.txt +++ b/llms.txt @@ -13,7 +13,7 @@ - Publisher URL: https://threemoonslab.com/ - License: Apache-2.0 - Latest public release: v1.1.0 (runtime contract 40; advisory channel, no qualification claim) -- Current source-tree runtime: 1.1.0 on `main` (contract 41; unreleased, ahead of the latest public release) +- Current source-tree runtime: 1.2.0 on `main` (contract 41; unreleased, ahead of the latest public release) - Canonical repository: https://github.com/ThreeMoonsLab/agents-shipgate - Do not use: Agent Shipcheck, Agent Shipgate, agents shipgate, Agents-Shipgate diff --git a/plugins/agents-shipgate/.codex-plugin/plugin.json b/plugins/agents-shipgate/.codex-plugin/plugin.json index a2332a4c..0538c2a0 100644 --- a/plugins/agents-shipgate/.codex-plugin/plugin.json +++ b/plugins/agents-shipgate/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agents-shipgate", - "version": "1.1.0", + "version": "1.2.0", "description": "Review declared coding-agent configuration changes with the agents-shipgate diff CLI command; the bundled skill verifies tool-using agent releases.", "author": { "name": "Three Moons Lab", diff --git a/plugins/claude-code/.claude-plugin/plugin.json b/plugins/claude-code/.claude-plugin/plugin.json index 0bbe854f..d5f89ef4 100644 --- a/plugins/claude-code/.claude-plugin/plugin.json +++ b/plugins/claude-code/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agents-shipgate", - "version": "1.1.0", + "version": "1.2.0", "description": "Review PR changes to declared coding-agent permissions, MCP bindings, hooks and workflow permissions with the agents-shipgate diff CLI command (not routed by this plugin's skill, which covers check, verify and audit --host). This static host comparison needs no manifest, policy, saved baseline or skill; it names coverage limits and implies no verdict. For tool-using application repositories, the existing Tool-Use Readiness workflow verifies a declared tool surface. The plugin supplies optional workflow instructions; the scanner runs through the agents-shipgate CLI installed locally (pipx install agents-shipgate). Installing this metadata does not enable hooks or establish runtime authority.", "author": { "name": "Three Moons Lab", diff --git a/pyproject.toml b/pyproject.toml index 50afa108..14b4025b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "agents-shipgate" -version = "1.1.0" +version = "1.2.0" description = "The deterministic merge gate for AI-generated agent capability changes. Shows what a PR changes in coding-agent configuration (Claude Code settings, .mcp.json, hooks, Codex, Cursor, VS Code MCP, GitHub workflow permissions) with no manifest. Agent release readiness for tool-using AI agents. CLI + GitHub Action. Scans MCP, OpenAPI, OpenAI Agents SDK, Anthropic, Google ADK, LangChain, CrewAI, OpenAI API, Codex config, Codex plugin, n8n, Conductor OSS workflow JSON." readme = "README.md" requires-python = ">=3.12" diff --git a/src/agents_shipgate/__init__.py b/src/agents_shipgate/__init__.py index 61f4b760..b424c5d6 100644 --- a/src/agents_shipgate/__init__.py +++ b/src/agents_shipgate/__init__.py @@ -1,3 +1,3 @@ """Agents Shipgate package.""" -__version__ = "1.1.0" +__version__ = "1.2.0" diff --git a/tests/golden/codex_boundary_result/agents_requirement_removed.json b/tests/golden/codex_boundary_result/agents_requirement_removed.json index 07832305..28fb2440 100644 --- a/tests/golden/codex_boundary_result/agents_requirement_removed.json +++ b/tests/golden/codex_boundary_result/agents_requirement_removed.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/docs_only.json b/tests/golden/codex_boundary_result/docs_only.json index e32cc7f6..50c563e4 100644 --- a/tests/golden/codex_boundary_result/docs_only.json +++ b/tests/golden/codex_boundary_result/docs_only.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/github_action_removed.json b/tests/golden/codex_boundary_result/github_action_removed.json index 64915430..aeebd8e2 100644 --- a/tests/golden/codex_boundary_result/github_action_removed.json +++ b/tests/golden/codex_boundary_result/github_action_removed.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/malformed_toml.json b/tests/golden/codex_boundary_result/malformed_toml.json index d4cd79ed..64107a1f 100644 --- a/tests/golden/codex_boundary_result/malformed_toml.json +++ b/tests/golden/codex_boundary_result/malformed_toml.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/mcp_auto_approve_write.json b/tests/golden/codex_boundary_result/mcp_auto_approve_write.json index ec0e1ab3..7011c0c2 100644 --- a/tests/golden/codex_boundary_result/mcp_auto_approve_write.json +++ b/tests/golden/codex_boundary_result/mcp_auto_approve_write.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/network_wildcard.json b/tests/golden/codex_boundary_result/network_wildcard.json index 12ede411..83c1fa08 100644 --- a/tests/golden/codex_boundary_result/network_wildcard.json +++ b/tests/golden/codex_boundary_result/network_wildcard.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/python_refactor.json b/tests/golden/codex_boundary_result/python_refactor.json index 49dc79e0..9b6a0446 100644 --- a/tests/golden/codex_boundary_result/python_refactor.json +++ b/tests/golden/codex_boundary_result/python_refactor.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/golden/codex_boundary_result/unknown_permission_key.json b/tests/golden/codex_boundary_result/unknown_permission_key.json index 63acc268..41f40422 100644 --- a/tests/golden/codex_boundary_result/unknown_permission_key.json +++ b/tests/golden/codex_boundary_result/unknown_permission_key.json @@ -3,7 +3,7 @@ "agent": "codex", "tool": { "name": "agents-shipgate", - "version": "1.1.0" + "version": "1.2.0" }, "subject": { "agent": "codex" diff --git a/tests/test_release_channel.py b/tests/test_release_channel.py index f066e33d..71faf046 100644 --- a/tests/test_release_channel.py +++ b/tests/test_release_channel.py @@ -28,6 +28,7 @@ def test_the_committed_declaration_records_the_advisory_1_0_decision() -> None: assert rc.load_declaration(REPO_ROOT / rc.DECLARATION_PATH) == { "1.0.0": "advisory", "1.1.0": "advisory", + "1.2.0": "advisory", } diff --git a/tests/test_v07_metadata_roundtrip.py b/tests/test_v07_metadata_roundtrip.py index d1065b6a..651effcd 100644 --- a/tests/test_v07_metadata_roundtrip.py +++ b/tests/test_v07_metadata_roundtrip.py @@ -240,7 +240,7 @@ def test_package_version_is_current_in_tree_runtime(): """Guard against bumping schemas while leaving package metadata behind.""" import agents_shipgate - assert agents_shipgate.__version__ == "1.1.0", ( + assert agents_shipgate.__version__ == "1.2.0", ( f"package version is {agents_shipgate.__version__!r}; " - "expected 1.1.0 for the current in-tree runtime" + "expected 1.2.0 for the current in-tree runtime" )