From 2a123c67ca2d326e37368ab52ac3398076b85041 Mon Sep 17 00:00:00 2001 From: Pengfei Hu Date: Sat, 3 Oct 2026 02:11:37 -0700 Subject: [PATCH 1/2] feat(openshell): bind optional native containment to trusted inputs --- .well-known/agents-shipgate.json | 4 + STABILITY.md | 11 + docs/agent-contract-current.md | 1 + docs/checks.json | 44 ++ docs/checks.md | 18 + docs/checks/verify.yaml | 20 + docs/openshell-native-evidence-schema.v1.json | 357 +++++++++++++++ docs/openshell-native-trust-schema.v1.json | 111 +++++ docs/openshell-support.md | 94 +++- llms-full.txt | 19 + scripts/generate_schemas.py | 22 + src/agents_shipgate/checks/verify.py | 4 + .../cli/_artifact_lifecycle.py | 1 + src/agents_shipgate/cli/verification.py | 4 + src/agents_shipgate/cli/verify/command.py | 13 + .../cli/verify/orchestrator.py | 41 +- src/agents_shipgate/core/openshell_native.py | 429 ++++++++++++++++++ .../core/verification_input_currency.py | 4 + .../schemas/openshell_native.py | 107 +++++ src/agents_shipgate/schemas/verification.py | 3 + tests/test_adapter_static_only.py | 30 ++ tests/test_openshell_native.py | 393 ++++++++++++++++ 22 files changed, 1728 insertions(+), 2 deletions(-) create mode 100644 docs/openshell-native-evidence-schema.v1.json create mode 100644 docs/openshell-native-trust-schema.v1.json create mode 100644 src/agents_shipgate/core/openshell_native.py create mode 100644 src/agents_shipgate/schemas/openshell_native.py create mode 100644 tests/test_openshell_native.py diff --git a/.well-known/agents-shipgate.json b/.well-known/agents-shipgate.json index 640427d5..b7d11a1e 100644 --- a/.well-known/agents-shipgate.json +++ b/.well-known/agents-shipgate.json @@ -312,6 +312,8 @@ "host_grants_inventory_schema_version": "0.9", "host_grants_baseline_schema_version": "0.9", "host_grants_drift_schema_version": "0.9", + "openshell_native_trust_schema_version": "1", + "openshell_native_evidence_schema_version": "1", "trigger_catalog_schema_version": "0.4", "capability_standard_version": "0.5", "governance_benchmark_catalog_schema_version": "0.2", @@ -528,6 +530,8 @@ "host_grants_inventory": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-inventory-schema.v0.9.json", "host_grants_baseline": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-baseline-schema.v0.9.json", "host_grants_drift": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/host-grants-drift-schema.v0.9.json", + "openshell_native_trust": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-trust-schema.v1.json", + "openshell_native_evidence": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-evidence-schema.v1.json", "scenario": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/scenario-schema.v0.1.json", "checks_catalog": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/checks.json", "determinism_boundary": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/determinism-boundary.json", diff --git a/STABILITY.md b/STABILITY.md index d08d37c4..00548fc1 100644 --- a/STABILITY.md +++ b/STABILITY.md @@ -4273,6 +4273,17 @@ tests on every CI run, not by convention: sanitized environment, isolated/fsck-validated object graph, fixed list argv, exact force-with-lease, and no shell. They are explicit operational executor surfaces, not part of static tool extraction. + - **`core/openshell_native.py`** — one `subprocess` import and one + `subprocess.Popen` call serve only explicit + `verify --openshell-proof-config` requests. The operator selects an + external configuration that pins the executable and containment boundary + by hash. Captured bytes run from a private directory with fixed list argv, + no shell, a sanitized environment, bounded output and process-group + lifetime. Linux additionally bounds virtual memory; Darwin does not. + Default static extraction and verification never invoke this boundary. + There is no executable discovery, installation, download or network call. + The result describes modeled containment and grants no merge authority; + see [the OpenShell trust contract](docs/openshell-support.md). - **`cli/fixture.py`** — one `subprocess.run` helper invokes local `git init`, `git config`, `git add`, `git commit`, and `git update-ref` against a temporary bundled fixture copy so diff --git a/docs/agent-contract-current.md b/docs/agent-contract-current.md index 59e82122..757e1e8e 100644 --- a/docs/agent-contract-current.md +++ b/docs/agent-contract-current.md @@ -821,6 +821,7 @@ Downstream repos generated with - Current registry schema: `0.4` — [`docs/registry-schema.v0.4.json`](registry-schema.v0.4.json) - Current org evidence bundle schema: `shipgate.org_evidence_bundle/v2` — [`docs/org-evidence-bundle-schema.v2.json`](org-evidence-bundle-schema.v2.json) - Current host-grants inventory, baseline, and drift schemas: `0.9` — [`inventory`](host-grants-inventory-schema.v0.9.json), [`baseline`](host-grants-baseline-schema.v0.9.json), [`drift`](host-grants-drift-schema.v0.9.json). Version 0.9 adds explicit local OpenShell composition, provider profile provenance and effective-snapshot metadata; historical host schemas remain frozen. See [OpenShell support](openshell-support.md). +- Optional OpenShell native trust/evidence schemas: `1` — [`trust`](openshell-native-trust-schema.v1.json), [`evidence`](openshell-native-evidence-schema.v1.json). Only explicit external trust inputs enable local execution; the default remains static. Native containment never grants merge authority. See [OpenShell support](openshell-support.md#optional-native-containment). - Current trigger catalog schema: `0.4` — [`docs/triggers.json`](triggers.json) - Current governance benchmark catalog schema: `0.2` — [`docs/governance-benchmark-catalog-schema.v0.2.json`](governance-benchmark-catalog-schema.v0.2.json) - Current governance benchmark result schema: `0.2` — [`docs/governance-benchmark-result-schema.v0.2.json`](governance-benchmark-result-schema.v0.2.json) diff --git a/docs/checks.json b/docs/checks.json index 3baa0645..48646709 100644 --- a/docs/checks.json +++ b/docs/checks.json @@ -3051,6 +3051,50 @@ "requires_human_review_regardless_of_patch": false, "suggested_patch_kind": "manual" }, + { + "autofix_safe": false, + "category": "verify", + "default_severity": "critical", + "description": "Trusted native OpenShell execution found a modeled candidate action outside the independently pinned maximum boundary.", + "docs_url": "https://github.com/ThreeMoonsLab/agents-shipgate/blob/main/docs/checks.md#ship-verify-openshell-boundary-exceeded", + "dynamic_default": false, + "evidence_fields": [ + "status", + "reason_code", + "required" + ], + "fires_when": "Opt-in local execution returns a validated exceeds_boundary envelope with matching actual exit status and all required domains.", + "floor_severity": "critical", + "id": "SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED", + "mvp_tier": "lifecycle", + "rationale": "A known containment counterexample cannot be treated as a passing release input. Native evidence never approves other review obligations.", + "recommendation": "Narrow the candidate and rerun trusted native proof; do not widen the trusted boundary from the candidate PR.", + "requires_human_review": true, + "requires_human_review_regardless_of_patch": false, + "suggested_patch_kind": "manual" + }, + { + "autofix_safe": false, + "category": "verify", + "default_severity": "medium", + "description": "Requested native OpenShell containment is unavailable, unsupported, invalid or inconclusive.", + "docs_url": "https://github.com/ThreeMoonsLab/agents-shipgate/blob/main/docs/checks.md#ship-verify-openshell-proof-unavailable", + "dynamic_default": false, + "evidence_fields": [ + "status", + "reason_code", + "required" + ], + "fires_when": "Opt-in execution cannot establish current modeled containment, or required proof has not been supplied. Optional absent proof is reported only as absent.", + "floor_severity": "medium", + "id": "SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE", + "mvp_tier": "lifecycle", + "rationale": "A missing or failed proof establishes no containment. Required proof also creates an unsuppressible evidence gap through the existing release decision.", + "recommendation": "Repair trusted external inputs or unsupported proof context and rerun verify with the same proof options.", + "requires_human_review": true, + "requires_human_review_regardless_of_patch": false, + "suggested_patch_kind": "manual" + }, { "autofix_safe": false, "category": "verify", diff --git a/docs/checks.md b/docs/checks.md index 8fdd5458..83ab4e56 100644 --- a/docs/checks.md +++ b/docs/checks.md @@ -1427,3 +1427,21 @@ the Conductor finding above. HTTP/custom-worker/A2A/provider-native execution is recorded as an unsupported capability and source warning in v1, so partial coverage cannot silently produce `passed`. A `HUMAN` task is structural pause evidence only; it does not prove reviewer identity or approval. + +| `SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED` | critical | Opt-in trusted native containment found the candidate outside the pinned maximum boundary; narrow it and rerun proof. | +| `SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE` | medium | Requested native containment is absent, invalid, unsupported or inconclusive; required proof retains an evidence gap. Repair trust inputs and rerun the same request. | + +### SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED + +Opt-in trusted native execution found a modeled action outside the independently +pinned maximum. The critical verify finding is suppression-immune. Narrow the +candidate and rerun proof; a baseline or proof success cannot approve other +release obligations. Counterexample wording is excluded from finding identity. + +### SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE + +Requested modeled containment is absent, invalid, unsupported, inconclusive, +failed, cancelled or timed out. Required non-success retains an unsuppressible +evidence gap. Optional missing proof is an absent observation, with no claim of +containment. Repair external trust inputs and rerun the same verification +request. See [OpenShell support](openshell-support.md#optional-native-containment). diff --git a/docs/checks/verify.yaml b/docs/checks/verify.yaml index 0019180b..5ec9d1c1 100644 --- a/docs/checks/verify.yaml +++ b/docs/checks/verify.yaml @@ -263,3 +263,23 @@ checks: recommendation: A human must review the removal; restore the config binding or an explicit literal allowlist, or declare an explicit local tool inventory for the toolkit so the mounted surface is statically enumerable. +- id: SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED + default_severity: critical + floor_severity: critical + mvp_tier: lifecycle + requires_human_review: true + description: Trusted native OpenShell execution found a modeled candidate action outside the independently pinned maximum boundary. + rationale: A known containment counterexample cannot be treated as a passing release input. Native evidence never approves other review obligations. + fires_when: Opt-in local execution returns a validated exceeds_boundary envelope with matching actual exit status and all required domains. + evidence_fields: [status, reason_code, required] + recommendation: Narrow the candidate and rerun trusted native proof; do not widen the trusted boundary from the candidate PR. +- id: SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE + default_severity: medium + floor_severity: medium + mvp_tier: lifecycle + requires_human_review: true + description: Requested native OpenShell containment is unavailable, unsupported, invalid or inconclusive. + rationale: A missing or failed proof establishes no containment. Required proof also creates an unsuppressible evidence gap through the existing release decision. + fires_when: Opt-in execution cannot establish current modeled containment, or required proof has not been supplied. Optional absent proof is reported only as absent. + evidence_fields: [status, reason_code, required] + recommendation: Repair trusted external inputs or unsupported proof context and rerun verify with the same proof options. diff --git a/docs/openshell-native-evidence-schema.v1.json b/docs/openshell-native-evidence-schema.v1.json new file mode 100644 index 00000000..816b7a92 --- /dev/null +++ b/docs/openshell-native-evidence-schema.v1.json @@ -0,0 +1,357 @@ +{ + "$defs": { + "ExternalIdentity": { + "additionalProperties": false, + "properties": { + "path": { + "title": "Path", + "type": "string" + }, + "sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Sha256" + }, + "size_bytes": { + "default": 0, + "maximum": 67108864, + "minimum": 0, + "title": "Size Bytes", + "type": "integer" + }, + "state": { + "enum": [ + "file", + "absent", + "unconfirmable" + ], + "title": "State", + "type": "string" + } + }, + "required": [ + "path", + "state" + ], + "title": "ExternalIdentity", + "type": "object" + }, + "NativeObservation": { + "additionalProperties": false, + "properties": { + "actual_exit_code": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Actual Exit Code" + }, + "candidate": { + "additionalProperties": true, + "title": "Candidate", + "type": "object" + }, + "deployment_enforcement_verified": { + "const": false, + "default": false, + "title": "Deployment Enforcement Verified", + "type": "boolean" + }, + "external_inputs": { + "items": { + "$ref": "#/$defs/ExternalIdentity" + }, + "maxItems": 3, + "title": "External Inputs", + "type": "array" + }, + "grants_merge_authority": { + "const": false, + "default": false, + "title": "Grants Merge Authority", + "type": "boolean" + }, + "invocation": { + "items": { + "type": "string" + }, + "title": "Invocation", + "type": "array" + }, + "modeled_containment_only": { + "const": true, + "default": true, + "title": "Modeled Containment Only", + "type": "boolean" + }, + "provenance": { + "default": "not_executed", + "enum": [ + "not_executed", + "local_trusted_execution" + ], + "title": "Provenance", + "type": "string" + }, + "raw_result": { + "anyOf": [ + { + "maxLength": 262144, + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Raw Result" + }, + "raw_result_sha256": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Raw Result Sha256" + }, + "reason_code": { + "title": "Reason Code", + "type": "string" + }, + "required": { + "title": "Required", + "type": "boolean" + }, + "required_domains": { + "items": { + "type": "string" + }, + "title": "Required Domains", + "type": "array" + }, + "status": { + "enum": [ + "absent", + "within_boundary", + "exceeds_boundary", + "unsupported", + "inconclusive", + "error", + "invalid", + "timeout", + "cancelled" + ], + "title": "Status", + "type": "string" + }, + "version": { + "const": 1, + "default": 1, + "title": "Version", + "type": "integer" + } + }, + "required": [ + "status", + "reason_code", + "required" + ], + "title": "NativeObservation", + "type": "object" + }, + "VerificationGitSubject": { + "additionalProperties": false, + "properties": { + "base_commit_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Base Commit Sha" + }, + "base_ref": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Base Ref" + }, + "base_tree_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Base Tree Sha" + }, + "head_commit_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Head Commit Sha" + }, + "head_ref": { + "title": "Head Ref", + "type": "string" + }, + "head_tree_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Head Tree Sha" + }, + "merge_base_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Merge Base Sha" + }, + "repository_id": { + "title": "Repository Id", + "type": "string" + }, + "snapshot_kind": { + "enum": [ + "committed_tree", + "worktree_overlay" + ], + "title": "Snapshot Kind", + "type": "string" + }, + "source_head_commit_sha": { + "anyOf": [ + { + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Source Head Commit Sha" + }, + "worktree_overlay_sha256": { + "anyOf": [ + { + "pattern": "^sha256:[0-9a-f]{64}$", + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Worktree Overlay Sha256" + } + }, + "required": [ + "repository_id", + "head_ref", + "snapshot_kind" + ], + "title": "VerificationGitSubject", + "type": "object" + }, + "VerificationSubject": { + "additionalProperties": false, + "properties": { + "git": { + "$ref": "#/$defs/VerificationGitSubject" + }, + "subject_id": { + "pattern": "^sha256:[0-9a-f]{64}$", + "title": "Subject Id", + "type": "string" + } + }, + "required": [ + "subject_id", + "git" + ], + "title": "VerificationSubject", + "type": "object" + } + }, + "$id": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-evidence-schema.v1.json", + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "evidence_schema_version": { + "const": "1", + "default": "1", + "title": "Evidence Schema Version", + "type": "string" + }, + "observation": { + "$ref": "#/$defs/NativeObservation" + }, + "request_id": { + "title": "Request Id", + "type": "string" + }, + "subject": { + "$ref": "#/$defs/VerificationSubject" + } + }, + "required": [ + "observation", + "request_id", + "subject" + ], + "title": "Agents Shipgate modeled OpenShell native containment evidence v1", + "type": "object" +} diff --git a/docs/openshell-native-trust-schema.v1.json b/docs/openshell-native-trust-schema.v1.json new file mode 100644 index 00000000..37b46d9f --- /dev/null +++ b/docs/openshell-native-trust-schema.v1.json @@ -0,0 +1,111 @@ +{ + "$defs": { + "CandidateSelection": { + "additionalProperties": false, + "properties": { + "composition": { + "default": "", + "title": "Composition", + "type": "string" + }, + "path": { + "default": "", + "title": "Path", + "type": "string" + }, + "registration": { + "title": "Registration", + "type": "string" + } + }, + "required": [ + "registration" + ], + "title": "CandidateSelection", + "type": "object" + }, + "PinnedFile": { + "additionalProperties": false, + "properties": { + "path": { + "maxLength": 4096, + "minLength": 1, + "title": "Path", + "type": "string" + }, + "sha256": { + "pattern": "^sha256:[a-f0-9]{64}$", + "title": "Sha256", + "type": "string" + } + }, + "required": [ + "path", + "sha256" + ], + "title": "PinnedFile", + "type": "object" + } + }, + "$id": "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-trust-schema.v1.json", + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "boundary": { + "$ref": "#/$defs/PinnedFile" + }, + "candidate": { + "$ref": "#/$defs/CandidateSelection" + }, + "executable": { + "$ref": "#/$defs/PinnedFile" + }, + "prover_version": { + "const": "0.1.2", + "title": "Prover Version", + "type": "string" + }, + "required": { + "title": "Required", + "type": "boolean" + }, + "required_domains": { + "items": { + "type": "string" + }, + "maxItems": 5, + "minItems": 5, + "title": "Required Domains", + "type": "array" + }, + "runtime_version": { + "const": "0.1.2", + "title": "Runtime Version", + "type": "string" + }, + "timeout_seconds": { + "default": 10, + "maximum": 30, + "minimum": 1, + "title": "Timeout Seconds", + "type": "integer" + }, + "version": { + "const": 1, + "title": "Version", + "type": "integer" + } + }, + "required": [ + "version", + "required", + "runtime_version", + "prover_version", + "executable", + "boundary", + "candidate", + "required_domains" + ], + "title": "Agents Shipgate optional OpenShell native trust configuration v1", + "type": "object" +} diff --git a/docs/openshell-support.md b/docs/openshell-support.md index b2761862..e6a1117f 100644 --- a/docs/openshell-support.md +++ b/docs/openshell-support.md @@ -47,7 +47,7 @@ Defaults follow the pinned [OpenShell v0.1.2 authored schema](https://github.com and [conversion code](https://github.com/NVIDIA/OpenShell/blob/v0.1.2/crates/openshell-policy/src/lib.rs). The [upstream schema reference](https://docs.nvidia.com/openshell/how-it-works/policies/schema) describes runtime constraints beyond document inventory. This reader is not a -substitute for upstream policy validation. Native proof remains an optional, separate implementation stage (#948). +substitute for upstream policy validation. Native containment is a separate opt-in verifier stage. ## Conservative declared-authority comparison @@ -271,3 +271,95 @@ Filesystem and Landlock fields are startup-bound, process identity is fixed at sandbox creation, and network policy/middleware fields may update dynamically. A composed proposal and an effective snapshot retain distinct roles. Neither establishes installed policy or live workload behavior. + +## Optional native containment + +A default `scan`, `check`, preview or flagless verifier never executes or installs +OpenShell or its prover. Native execution requires an operator-owned external +trust configuration, passed explicitly to configured `verify`: + +```bash +agents-shipgate verify --workspace . --config shipgate.yaml --base origin/main \ + --openshell-proof-config /operator/openshell/trust.json --json +``` + +The operator supplies the pinned executable and maximum boundary independently +of the candidate PR. All three trust inputs must be absolute, outside the +candidate workspace, without symlink traversal or hard links, owned by the +running user or root, and not writable by group or others. Output directories +cannot overlap them. External location is a containment boundary, not proof of +human approval: the operator or trusted CI must independently approve and +protect these files. Never copy trust configuration or executables from the +candidate checkout to manufacture this provenance. A trusted CI job must load +its trust inputs and invocation from operator-controlled infrastructure rather +than candidate-authored workflow code. + +Example trust configuration ([schema v1](openshell-native-trust-schema.v1.json)): + +```json +{ + "version": 1, + "required": true, + "runtime_version": "0.1.2", + "prover_version": "0.1.2", + "executable": {"path": "/operator/openshell/openshell-prover", "sha256": "sha256:"}, + "boundary": {"path": "/operator/openshell/maximum.yaml", "sha256": "sha256:"}, + "candidate": {"registration": ".shipgate/openshell.json", "path": "configs/worker-policy.yaml"}, + "required_domains": ["filesystem", "network_l4", "network_rest", "process", "landlock"], + "timeout_seconds": 10 +} +``` + +Replace digest placeholders with the independently approved SHA-256 identities. +Select either a registered document `path`, or a named local `composition` from +a version 2 registration. Composed candidates retain contributor provenance and +reconstruct the selected authored fields, preserving omitted defaults. The +result must normalize to the same policy as the static composer before it runs. +Every original dependency +remains bound by the static read session. A selected effective snapshot still +has unverified deployment freshness. No credential values are read. + +The supported contract is [OpenShell v0.1.2 JSON schema 1](https://github.com/NVIDIA/OpenShell/blob/v0.1.2/crates/openshell-prover-cli/src/main.rs). +All five [modeled domains](https://github.com/NVIDIA/OpenShell/blob/v0.1.2/crates/openshell-prover/src/containment.rs) +are required. The wrapper checks schema/version, input names, actual process +exit status, domain coverage, outcome consistency and stable reason identifiers. +It supplies captured immutable bytes to a copied, hash-pinned executable in a +private directory, with no shell, repository executable discovery or inherited +credential/loader environment. The installation's approved runtime and shared +libraries remain part of the operator's execution trust; the executable hash +does not attest an entire operating system. The executable must remain runnable +after copying into the private directory without loader environment overrides. +Use a release-stamped prover reporting `0.1.2`; an unstamped source build inherits +the upstream development Cargo version `0.0.0` and is rejected by this contract. + +Execution supports POSIX descriptor reads and process groups. Bounds include +1–30 seconds of solver budget plus two seconds of wall-time overhead, CPU time, +256 KiB per output stream, file size, descriptor count and input size. Linux +also applies a 2 GiB address-space limit. Darwin does not support that limit; +its time/output/file bounds still apply. Unsupported platforms cannot produce +passing execution evidence. The complete process group is terminated at the +end, including timeout, output overflow and cancellation. + +`openshell-native.json` ([evidence schema v1](openshell-native-evidence-schema.v1.json)) +records the observation, immutable candidate hash, trusted config/boundary/prover +identities, invocation options, actual exit status, available raw validated JSON, +composition provenance and the verifier's Git subject/request identity. The +same observation is bound in plan options and the existing terminal receipt. +Current-control reads revalidate every external origin as well as candidate +inputs; a change to ignored policy/profile bytes, trust configuration, boundary +or prover makes that receipt stale. A local receipt is not a portable signature +of approved execution. Repository-authored result JSON has no import route; +workers and assembly refuse native-proof plans and require a fresh trusted run. + +`within_boundary` establishes only modeled containment of those inputs. It +clears no static host expansion, review, purpose, effect, authority or binding +obligation. `exceeds_boundary` produces a critical, suppression-immune verify +finding. Unsupported, inconclusive, error, invalid, cancelled and timeout +outcomes remain distinct observations. Required non-success also adds an +unsuppressible evidence gap through the existing release decision, so it cannot +complete. If optional proof is missing, its observation is `absent`, with +`provenance: "not_executed"`; it is never described as successful containment. +Use `--openshell-proof-required` to require proof even when no configuration +is available. Required status is scoped to that verification request, and +recovery commands preserve both proof flags. Native execution is refused on +preview and never supplies a manifest-free release verdict. diff --git a/llms-full.txt b/llms-full.txt index 05bf8f0e..64d09e32 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -2400,6 +2400,7 @@ Downstream repos generated with - Current registry schema: `0.4` — [`docs/registry-schema.v0.4.json`](registry-schema.v0.4.json) - Current org evidence bundle schema: `shipgate.org_evidence_bundle/v2` — [`docs/org-evidence-bundle-schema.v2.json`](org-evidence-bundle-schema.v2.json) - Current host-grants inventory, baseline, and drift schemas: `0.9` — [`inventory`](host-grants-inventory-schema.v0.9.json), [`baseline`](host-grants-baseline-schema.v0.9.json), [`drift`](host-grants-drift-schema.v0.9.json). Version 0.9 adds explicit local OpenShell composition, provider profile provenance and effective-snapshot metadata; historical host schemas remain frozen. See [OpenShell support](openshell-support.md). +- Optional OpenShell native trust/evidence schemas: `1` — [`trust`](openshell-native-trust-schema.v1.json), [`evidence`](openshell-native-evidence-schema.v1.json). Only explicit external trust inputs enable local execution; the default remains static. Native containment never grants merge authority. See [OpenShell support](openshell-support.md#optional-native-containment). - Current trigger catalog schema: `0.4` — [`docs/triggers.json`](triggers.json) - Current governance benchmark catalog schema: `0.2` — [`docs/governance-benchmark-catalog-schema.v0.2.json`](governance-benchmark-catalog-schema.v0.2.json) - Current governance benchmark result schema: `0.2` — [`docs/governance-benchmark-result-schema.v0.2.json`](governance-benchmark-result-schema.v0.2.json) @@ -4834,6 +4835,24 @@ is recorded as an unsupported capability and source warning in v1, so partial coverage cannot silently produce `passed`. A `HUMAN` task is structural pause evidence only; it does not prove reviewer identity or approval. +| `SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED` | critical | Opt-in trusted native containment found the candidate outside the pinned maximum boundary; narrow it and rerun proof. | +| `SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE` | medium | Requested native containment is absent, invalid, unsupported or inconclusive; required proof retains an evidence gap. Repair trust inputs and rerun the same request. | + +### SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED + +Opt-in trusted native execution found a modeled action outside the independently +pinned maximum. The critical verify finding is suppression-immune. Narrow the +candidate and rerun proof; a baseline or proof success cannot approve other +release obligations. Counterexample wording is excluded from finding identity. + +### SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE + +Requested modeled containment is absent, invalid, unsupported, inconclusive, +failed, cancelled or timed out. Required non-success retains an unsuppressible +evidence gap. Optional missing proof is an absent observation, with no claim of +containment. Repair external trust inputs and rerun the same verification +request. See [OpenShell support](openshell-support.md#optional-native-containment). + diff --git a/scripts/generate_schemas.py b/scripts/generate_schemas.py index a3e37453..f0c80e37 100644 --- a/scripts/generate_schemas.py +++ b/scripts/generate_schemas.py @@ -2557,6 +2557,26 @@ def build_host_grants_drift_schema() -> tuple[Path, str]: return target, _canonical_json(schema) +def build_openshell_native_trust_schema() -> tuple[Path, str]: + from agents_shipgate.schemas.openshell_native import NativeTrustConfig + + schema = NativeTrustConfig.model_json_schema() + schema["$schema"] = "https://json-schema.org/draft/2020-12/schema" + schema["$id"] = "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-trust-schema.v1.json" + schema["title"] = "Agents Shipgate optional OpenShell native trust configuration v1" + return DOCS / "openshell-native-trust-schema.v1.json", _canonical_json(schema) + + +def build_openshell_native_evidence_schema() -> tuple[Path, str]: + from agents_shipgate.schemas.openshell_native import NativeEvidence + + schema = NativeEvidence.model_json_schema() + schema["$schema"] = "https://json-schema.org/draft/2020-12/schema" + schema["$id"] = "https://raw.githubusercontent.com/ThreeMoonsLab/agents-shipgate/main/docs/openshell-native-evidence-schema.v1.json" + schema["title"] = "Agents Shipgate modeled OpenShell native containment evidence v1" + return DOCS / "openshell-native-evidence-schema.v1.json", _canonical_json(schema) + + # Public ordered list of (name, builder) pairs. Tests and the CLI iterate this # instead of hardcoding individual calls, so adding a new schema is one edit. # --- Determinism boundary --------------------------------------------------- @@ -3002,6 +3022,8 @@ def write_determinism_boundary_page( ("host_grants_inventory", build_host_grants_inventory_schema), ("host_grants_baseline", build_host_grants_baseline_schema), ("host_grants_drift", build_host_grants_drift_schema), + ("openshell_native_trust", build_openshell_native_trust_schema), + ("openshell_native_evidence", build_openshell_native_evidence_schema), ("governance_benchmark_catalog", build_governance_benchmark_catalog_schema), ("governance_benchmark_result", build_governance_benchmark_result_schema), ("determinism_boundary_matrix", build_determinism_boundary_matrix), diff --git a/src/agents_shipgate/checks/verify.py b/src/agents_shipgate/checks/verify.py index d4361ba1..6edd2378 100644 --- a/src/agents_shipgate/checks/verify.py +++ b/src/agents_shipgate/checks/verify.py @@ -54,6 +54,10 @@ def run(context: ScanContext) -> list[Finding]: assessment = assessment_for_scan_context(context) findings: list[Finding] = [] + if verification.openshell_native is not None: + from agents_shipgate.core.openshell_native import native_findings + + findings.extend(native_findings(verification.openshell_native, context)) seen: set[str] = set() for raw in verification.changed_files: path = raw.replace("\\", "/") diff --git a/src/agents_shipgate/cli/_artifact_lifecycle.py b/src/agents_shipgate/cli/_artifact_lifecycle.py index 3fbbf6bc..b95055c0 100644 --- a/src/agents_shipgate/cli/_artifact_lifecycle.py +++ b/src/agents_shipgate/cli/_artifact_lifecycle.py @@ -19,6 +19,7 @@ "verification-unit-result.json", "verification-artifacts.json", "verification-receipt.json", + "openshell-native.json", "human-authorization.json", # Identity-bearing: it names the ``input_set_id`` of the run that # produced it, so one left beside a later run's receipt would offer a diff --git a/src/agents_shipgate/cli/verification.py b/src/agents_shipgate/cli/verification.py index 5a0e9dca..af6f476c 100644 --- a/src/agents_shipgate/cli/verification.py +++ b/src/agents_shipgate/cli/verification.py @@ -523,6 +523,8 @@ def worker( require_workspace(workspace) plan = VerificationPlan.model_validate(_load_json(plan_path)) + if "openshell_native" in plan.inputs.options: + raise InputParseError("Native containment cannot be imported or replayed by a worker; rerun trusted local verify.") resolved_diff = diff_path or plan_path.with_name(plan.inputs.diff.path) try: validate_engine_requirement( @@ -568,6 +570,8 @@ def assemble( """Validate worker IR and close the sole-engine verifier projections.""" plan = VerificationPlan.model_validate(_load_json(plan_path)) + if "openshell_native" in plan.inputs.options: + raise InputParseError("Native containment cannot be imported or replayed by a worker; rerun trusted local verify.") units = [VerificationUnitResult.model_validate(_load_json(path)) for path in unit_paths] verifier = VerifierArtifact.model_validate(_load_json(verifier_path)) try: diff --git a/src/agents_shipgate/cli/verify/command.py b/src/agents_shipgate/cli/verify/command.py index 51a8f6e1..a9b98049 100644 --- a/src/agents_shipgate/cli/verify/command.py +++ b/src/agents_shipgate/cli/verify/command.py @@ -230,6 +230,8 @@ def verify( "(legacy v1 style, available for one minor release cycle)." ), ), + openshell_proof_config: Path | None = typer.Option(None, "--openshell-proof-config", help="Operator-owned external native-prover trust configuration; opt-in execution."), + openshell_proof_required: bool = typer.Option(False, "--openshell-proof-required", help="Require current native containment, including when no trust configuration is supplied."), verbose: bool = typer.Option(False, "--verbose", help="Show debug details."), ) -> None: """Run the canonical ongoing-PR verifier around the existing scan engine.""" @@ -246,6 +248,13 @@ def verify( try: configure_logging(verbose=verbose) stdout_format = _resolve_verify_format(format_, json_output=json_output, preview=preview) + if openshell_proof_config is not None: + from agents_shipgate.core.openshell_native import external_path + + try: + external_path(str(openshell_proof_config), workspace.resolve()) + except ValueError as exc: + raise ConfigError(str(exc)) from exc if ci_mode and ci_mode not in {"advisory", "strict"}: raise ConfigError("--ci-mode must be advisory or strict") for label, value in (("--base", base), ("--head", head)): @@ -258,6 +267,8 @@ def verify( ) parsed_fail_on = _parse_fail_on(fail_on) parsed_pr_comment_style = _parse_pr_comment_style(pr_comment_style) + if preview and (openshell_proof_config is not None or openshell_proof_required): + raise ConfigError("Native execution cannot be combined with --preview") if preview and authorization is not None: raise ConfigError("--authorization cannot be combined with --preview") except ConfigError as exc: @@ -316,6 +327,8 @@ def verify( baseline_mode=baseline_mode, diff_from=diff_from, authorization=authorization, + openshell_proof_config=openshell_proof_config, + openshell_proof_required=openshell_proof_required, policy_packs=policy_packs, plugins_enabled=False if no_plugins else None, strict_plugins=strict_plugins, diff --git a/src/agents_shipgate/cli/verify/orchestrator.py b/src/agents_shipgate/cli/verify/orchestrator.py index f8417d75..4e0427b2 100644 --- a/src/agents_shipgate/cli/verify/orchestrator.py +++ b/src/agents_shipgate/cli/verify/orchestrator.py @@ -339,7 +339,16 @@ def run_verify( pr_comment_style: str = "capability-review", auto_base: bool = False, authorization: Path | None = None, + openshell_proof_config: Path | None = None, + openshell_proof_required: bool = False, ) -> tuple[VerifierArtifact, ReadinessReport | None, int]: + if openshell_proof_config is not None: + from agents_shipgate.core.openshell_native import external_path + + try: + external_path(str(openshell_proof_config), workspace.resolve()) + except ValueError as exc: + raise ConfigError(str(exc)) from exc git_root = ensure_git_workspace(workspace.resolve()) config_path, config_relative = _resolve_config_under_workspace( git_root, @@ -391,6 +400,7 @@ def run_verify( out_dir=out_dir, inputs=[ ("config", config_path), + *([("OpenShell trust config", openshell_proof_config)] if openshell_proof_config is not None else []), *([("baseline", baseline_path)] if baseline_path is not None else []), *[("policy pack", path) for path in (policy_pack_paths or [])], *( @@ -400,6 +410,13 @@ def run_verify( ), ], ) + if openshell_proof_config is not None: + from agents_shipgate.core.openshell_native import reject_native_output_overlap + + try: + reject_native_output_overlap(openshell_proof_config, git_root, out_dir) + except ValueError as exc: + raise ConfigError(str(exc)) from exc # Before anything is written, and for every route this run can take: # `--head`, the worktree, and the manifest-free comparison below all leave # this directory out of what they read. @@ -451,6 +468,11 @@ def run_verify( no_heuristics=no_heuristics, authorization=authorization, ) + if openshell_proof_config is not None: + openshell_proof_config = Path(os.path.abspath(openshell_proof_config)) + rerun_options.extend(["--openshell-proof-config", shlex.quote(str(openshell_proof_config))]) + if openshell_proof_required: + rerun_options.append("--openshell-proof-required") if not config_path.is_file(): from .host_comparison import compare_host_refs, host_comparison_failure @@ -458,7 +480,7 @@ def run_verify( # A manifest-free host comparison is advisory evidence, not a synthetic # application policy. Explicit application inputs retain their failure. host_comparison = None - if ci_mode != "strict" and not any((baseline, policy_packs, diff_from, authorization, fail_on)): + if ci_mode != "strict" and not any((baseline, policy_packs, diff_from, authorization, fail_on, openshell_proof_config, openshell_proof_required)): try: host_comparison = compare_host_refs( workspace=git_root, base=base, head=head if archive_head else None, @@ -981,6 +1003,7 @@ def run_verify( capability_lock_diff: CapabilityLockDiffV1 | None = None capability_attestation_inputs: CapabilityDeltaAttestationInputs | None = None head_human_context: HumanArtifactContext | None = None + native_observation = None def capture_capability_lock(lock: CapabilityLockFileV1) -> None: nonlocal head_capability_lock @@ -1219,6 +1242,12 @@ def run_base_report_dir() -> Path: else None ) try: + if openshell_proof_config is not None or openshell_proof_required: + from agents_shipgate.core.openshell_native import observe_native + + native_observation = observe_native(config_path=openshell_proof_config, + required=openshell_proof_required, workspace=git_root, input_root=head_input_root) + base_notes.append(f"OpenShell native containment: {native_observation.status} ({native_observation.reason_code}); modeled inputs only.") report, head_exit_code = run_scan( config_path=head_config_path, output_dir=out_dir, @@ -1266,6 +1295,7 @@ def run_base_report_dir() -> Path: in _BASE_COMPARISON_FAILURES, enabled_plugin_hooks=enabled_plugin_hooks, enabled_plugin_hook_issues=tuple(enabled_plugin_hook_issues), + openshell_native=native_observation, ), capability_lock_callback=capture_capability_lock, human_context_callback=capture_human_context, @@ -1432,6 +1462,7 @@ def run_base_report_dir() -> Path: diff_from_path=base_report, authorization_path=authorization, verification_options={ + **({"openshell_native": native_observation.model_dump(mode="json")} if native_observation is not None else {}), "archive_head": archive_head, "baseline_mode": baseline_mode, "strict_plugins": strict_plugins, @@ -5043,6 +5074,14 @@ def _finalize(snapshot: StaticInputSnapshot | None) -> None: json.dumps(plan.model_dump(mode="json"), indent=2, sort_keys=True), encoding="utf-8", ) + if "openshell_native" in plan.inputs.options: + native_path = verifier_path.with_name("openshell-native.json") + from agents_shipgate.schemas.openshell_native import NativeEvidence + + native_evidence = NativeEvidence.model_validate({"observation": plan.inputs.options["openshell_native"], + "request_id": plan.request_id, "subject": plan.subject.model_dump(mode="json")}) + native_path.write_text(json.dumps(native_evidence.model_dump(mode="json"), indent=2, sort_keys=True), encoding="utf-8") + verifier.artifacts["openshell_native_json"] = _display_path(native_path, git_root) unit_result = build_unit_result( plan=plan, status=("succeeded" if verifier.execution in {"succeeded", "skipped"} else "failed"), diff --git a/src/agents_shipgate/core/openshell_native.py b/src/agents_shipgate/core/openshell_native.py new file mode 100644 index 00000000..3727eea7 --- /dev/null +++ b/src/agents_shipgate/core/openshell_native.py @@ -0,0 +1,429 @@ +"""Opt-in local execution; repository JSON can never substitute for a run.""" +from __future__ import annotations + +import hashlib +import json +import math +import os +import selectors +import signal +import stat +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +from pydantic import ValidationError + +from agents_shipgate.core.host_grants import ( + _openshell_public_value, + build_host_boundary_snapshot, +) +from agents_shipgate.core.openshell import load_document, parse_policy +from agents_shipgate.core.static_inputs import read_static_input_bytes +from agents_shipgate.schemas.openshell_native import ( + DOMAINS, + ExternalIdentity, + NativeEnvelope, + NativeObservation, + NativeTrustConfig, +) + +MAX_OUTPUT = 256 * 1024 +MAX_EXECUTABLE = 64 * 1024 * 1024 + + +def digest(data: bytes) -> str: + return "sha256:" + hashlib.sha256(data).hexdigest() + + +def external_path(value: str, workspace: Path) -> Path: + path = Path(value) + if not path.is_absolute() or any(char in value for char in "\x00\r\n") or _openshell_public_value(value) != value: + raise ValueError("trusted inputs require absolute external paths") + lexical = Path(os.path.abspath(path)) + # Reject aliases rather than silently changing the operator's selection. + if path != lexical or path.resolve() != lexical: + raise ValueError("trusted inputs cannot traverse symbolic links") + if lexical == workspace or workspace in lexical.parents: + raise ValueError("native trust inputs must be outside the candidate workspace") + return lexical + + +def _read_external_bytes(path: Path, limit: int) -> bytes: + # Exact descriptor traversal avoids scanning an unrelated /tmp or /home + # census. Every parent remains open and its named identity is rechecked. + if os.name != "posix" or os.open not in os.supports_dir_fd: + raise ValueError("trusted native input reads require POSIX descriptors") + descriptors = [] + observed = [] + try: + current = os.open(path.anchor, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + descriptors.append(current) + for part in path.parts[1:-1]: + child = os.open(part, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=current) + descriptors.append(child) + observed.append((current, part, os.fstat(child))) + current = child + file_fd = os.open(path.name, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=current) + descriptors.append(file_fd) + before = os.fstat(file_fd) + if (not stat.S_ISREG(before.st_mode) or before.st_nlink != 1 or before.st_size > limit + or before.st_uid not in {0, os.getuid()} or before.st_mode & 0o022): + raise ValueError("trusted input is not one bounded regular file") + chunks, size = [], 0 + while size <= limit: + chunk = os.read(file_fd, min(1024 * 1024, limit + 1 - size)) + if not chunk: + break + chunks.append(chunk) + size += len(chunk) + if size > limit: + raise ValueError("trusted input exceeds its read bound") + def identity(metadata): + return (metadata.st_dev, metadata.st_ino, metadata.st_mode) + for parent, part, metadata in observed: + if identity(os.stat(part, dir_fd=parent, follow_symlinks=False)) != identity(metadata): + raise ValueError("trusted parent changed while reading") + after = os.stat(path.name, dir_fd=current, follow_symlinks=False) + if (identity(before), before.st_size, before.st_mtime_ns, before.st_ctime_ns) != ( + identity(after), after.st_size, after.st_mtime_ns, after.st_ctime_ns): + raise ValueError("trusted input changed while reading") + return b"".join(chunks) + finally: + for descriptor in reversed(descriptors): + os.close(descriptor) + + +def capture_external(value: str, workspace: Path, identities: list, *, limit: int) -> bytes: + path = external_path(value, workspace) + try: + data = _read_external_bytes(path, limit) + except FileNotFoundError: + identities.append(ExternalIdentity(path=str(path), state="absent")) + raise + except (OSError, ValueError): + identities.append(ExternalIdentity(path=str(path), state="unconfirmable")) + raise + identities.append(ExternalIdentity(path=str(path), state="file", sha256=digest(data), size_bytes=len(data))) + return data + + +def validate_external_currency(observation, workspace: Path) -> None: + observed = NativeObservation.model_validate(observation) + if observed.status == "within_boundary": + if observed.provenance != "local_trusted_execution" or len(observed.external_inputs) != 3 or not observed.raw_result: + raise ValueError("successful proof lacks local execution identity") + if digest(observed.raw_result.encode()) != observed.raw_result_sha256: + raise ValueError("native raw result identity mismatch") + if validate_envelope(observed.raw_result.encode(), observed.actual_exit_code).result != "within_boundary": + raise ValueError("native proof observation contradicts its raw result") + for identity in observed.external_inputs: + path = external_path(identity.path, workspace) + if identity.state == "unconfirmable": + raise ValueError("native trust input identity is unconfirmable; rerun verification") + if identity.state == "absent": + # lstat includes dangling links and directories in the negative lookup. + try: + path.lstat() + except FileNotFoundError: + continue + raise ValueError("absent native trust input appeared; rerun verification") + identities = [] + data = capture_external(str(path), workspace, identities, limit=MAX_EXECUTABLE) + if digest(data) != identity.sha256 or len(data) != identity.size_bytes: + raise ValueError("native trust input changed; rerun verification") + + +def _unique_object(pairs): + result = {} + for key, value in pairs: + if key in result: + raise ValueError("duplicate JSON key") + result[key] = value + return result + + +def load_config(data: bytes) -> NativeTrustConfig: + value = _load_json(data) + if not isinstance(value, dict) or type(value.get("version")) is not int: + raise ValueError("invalid native trust configuration") + if _openshell_public_value(value) != value: + raise ValueError("native configuration requires redaction") + return NativeTrustConfig.model_validate(value) + + +def reject_native_output_overlap(config_path: Path | None, workspace: Path, out_dir: Path) -> None: + """Inspect trusted selections before any output artifact can overwrite them.""" + if config_path is None: + return + try: + config = load_config(capture_external(str(config_path), workspace, [], limit=64 * 1024)) + except (OSError, ValueError): + return # The evidence stage publishes its explicit invalid/absent route. + output = out_dir.resolve() + for selected in (config.executable.path, config.boundary.path): + candidate = Path(selected).resolve() + if candidate == output or output in candidate.parents: + raise ValueError("Verifier output overlaps a selected native trust input") + + +def validate_envelope(raw: bytes, exit_code: int) -> NativeEnvelope: + value = _load_json(raw) + if not isinstance(value, dict) or type(value.get("schema_version")) is not int: + raise ValueError("unknown native JSON schema") + envelope = NativeEnvelope.model_validate(value) + if _openshell_public_value(value) != value: + raise ValueError("native output requires redaction") + if envelope.inputs.model_dump() != {"candidate": "candidate.yaml", "boundary": "boundary.yaml"}: + raise ValueError("native output names different inputs") + expected = {"within_boundary": 0, "exceeds_boundary": 1, "unsupported": 3, + "inconclusive": 130 if envelope.reason_code == "cancelled" else 3, "error": 2} + if type(value.get("exit_code")) is not int or exit_code != envelope.exit_code or exit_code != expected[envelope.result]: + raise ValueError("native exit/result mismatch") + if envelope.result != "error" and ( + envelope.coverage is None or sorted(envelope.coverage.domains) != sorted(DOMAINS) + ): + raise ValueError("native output lacks required modeled domains") + if envelope.result == "within_boundary" and any( + item is not None for item in (envelope.counterexample, envelope.reason_code, envelope.reason) + ): + raise ValueError("contradictory successful proof") + if envelope.result == "exceeds_boundary" and ( + not envelope.counterexample or envelope.counterexample.get("domain") not in {"filesystem", "network", "landlock", "process"} + or envelope.reason_code is not None or envelope.reason is not None + ): + raise ValueError("invalid native counterexample") + if envelope.result in {"unsupported", "inconclusive", "error"} and ( + envelope.reason_code not in {"unsupported_policy_shape", "unresolved_workdir", "unresolved_binary_path", + "unresolved_filesystem_path", "solver_timeout", "solver_unknown", "resource_limit", "invalid_witness", + "cancelled", "invalid_input"} or not envelope.reason or envelope.counterexample is not None + ): + raise ValueError("native failure lacks a reason") + return envelope + + +def _limits(seconds: int) -> None: + import resource + + resource.setrlimit(resource.RLIMIT_CPU, (math.ceil(seconds) + 2, math.ceil(seconds) + 2)) + # Darwin rejects RLIMIT_AS/RLIMIT_DATA; wall/CPU/output/file/descriptor + # bounds still apply there. Linux additionally bounds virtual memory. + if sys.platform != "darwin": + resource.setrlimit(resource.RLIMIT_AS, (2 * 1024 ** 3, 2 * 1024 ** 3)) + resource.setrlimit(resource.RLIMIT_FSIZE, (1024 * 1024, 1024 * 1024)) + resource.setrlimit(resource.RLIMIT_NOFILE, (64, 64)) + + +def _execute(executable: bytes, candidate: bytes, boundary: bytes, seconds: int): + """Copy captured bytes into a private directory and bound the whole group.""" + if os.name != "posix": + return "unsupported", None, b"" + with tempfile.TemporaryDirectory(prefix="agents-shipgate-native-") as temp: + root = Path(temp) + for name, data in (("prover", executable), ("candidate.yaml", candidate), ("boundary.yaml", boundary)): + path = root / name + path.write_bytes(data) + path.chmod(0o500 if name == "prover" else 0o400) + command = [str(root / "prover"), "check", "candidate.yaml", "--boundary", "boundary.yaml", "--output", "json", "--timeout", f"{seconds}s"] + process = subprocess.Popen(command, cwd=root, shell=False, start_new_session=True, + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + env={"PATH": "/usr/bin:/bin", "HOME": temp, "LANG": "C.UTF-8"}, + preexec_fn=lambda: _limits(seconds)) + stdout = bytearray() + sizes = {process.stdout: 0, process.stderr: 0} + status = "finished" + deadline = time.monotonic() + seconds + 2 + selector = selectors.DefaultSelector() + for stream in sizes: + selector.register(stream, selectors.EVENT_READ) + try: + while selector.get_map(): + remaining = deadline - time.monotonic() + if remaining <= 0: + status = "timeout" + break + for key, _mask in selector.select(min(remaining, 0.1)): + chunk = os.read(key.fileobj.fileno(), 8192) + if not chunk: + selector.unregister(key.fileobj) + continue + sizes[key.fileobj] += len(chunk) + if sizes[key.fileobj] > MAX_OUTPUT: + status = "output_limit" + break + if key.fileobj is process.stdout: + stdout.extend(chunk) + if status != "finished": + break + if status == "finished": + try: + process.wait(timeout=max(0.001, deadline - time.monotonic())) + except subprocess.TimeoutExpired: + status = "timeout" + except KeyboardInterrupt: + status = "cancelled" + finally: + # Also kills a child that closed its pipes but left descendants. + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait() + selector.close() + for stream in sizes: + stream.close() + return status, process.returncode, bytes(stdout) + + +def observe_native(*, config_path: Path | None, required: bool, workspace: Path, input_root: Path) -> NativeObservation: + observation = NativeObservation(status="absent", reason_code="proof_not_supplied", required=required) + if config_path is None: + return observation + identities = observation.external_inputs + try: + config_data = capture_external(str(config_path), workspace, identities, limit=64 * 1024) + config = load_config(config_data) + observation.required = required or config.required + observation.candidate = {"selection": config.candidate.model_dump(mode="json")} + executable = capture_external(config.executable.path, workspace, identities, limit=MAX_EXECUTABLE) + boundary = capture_external(config.boundary.path, workspace, identities, limit=1024 * 1024) + if digest(executable) != config.executable.sha256 or digest(boundary) != config.boundary.sha256: + raise ValueError("pinned input identity mismatch") + boundary_policy = parse_policy(boundary.decode("utf-8")) + if _openshell_public_value(boundary_policy.model_dump(mode="json")) != boundary_policy.model_dump(mode="json"): + raise ValueError("boundary requires redaction") + snapshot = build_host_boundary_snapshot(input_root) + registration = config.candidate.registration + matching = [grant for grant in snapshot.inventory["grants"] if grant.get("kind") == "openshell_policy" + and grant["facts"]["registration"] == registration and ( + (config.candidate.path and grant["source"] == config.candidate.path) + or (config.candidate.composition and grant["facts"].get("composition", {}).get("name") == config.candidate.composition))] + if len(matching) != 1 or any(issue["host"] == "openshell" and issue["blocking"] for issue in snapshot.inventory["issues"]): + raise ValueError("candidate selection is incomplete or ambiguous") + grant, = matching + facts = grant["facts"] + if config.candidate.path: + artifact = next(item for item in snapshot.inventory["artifacts"] if item["host"] == "openshell" + and item["kind"] == "openshell_policy" and item["path"] == config.candidate.path) + target = (artifact.get("resolved_through") or [config.candidate.path])[-1] + candidate = read_static_input_bytes(input_root / target, max_bytes=1024 * 1024) + captured = snapshot.cache.openshell_input_reads[target] + if captured.get("limit") or digest(candidate) != "sha256:" + captured["sha256"] or len(candidate) != captured["size_bytes"]: + raise ValueError("candidate identity changed after extraction") + else: + candidate = _composed_candidate_bytes(facts, snapshot, input_root) + snapshot.cache.finish() + observation.candidate = {"selection": config.candidate.model_dump(mode="json"), "role": facts["role"], + "sha256": digest(candidate), "size_bytes": len(candidate), + "composition": facts.get("composition"), "runtime_freshness_verified": False} + observation.invocation = ["openshell-prover", "check", "candidate.yaml", "--boundary", "boundary.yaml", "--output", "json", "--timeout", f"{config.timeout_seconds}s"] + try: + status, exit_code, raw = _execute(executable, candidate, boundary, config.timeout_seconds) + except (OSError, subprocess.SubprocessError): + observation.status, observation.reason_code = "error", "native_execution_error" + return observation + observation.actual_exit_code = exit_code + if exit_code is not None: + observation.provenance = "local_trusted_execution" + if status != "finished": + observation.status = status if status in {"timeout", "cancelled", "unsupported"} else "error" + observation.reason_code = status + return observation + envelope = validate_envelope(raw, exit_code) + observation.status = "cancelled" if envelope.reason_code == "cancelled" else envelope.result + observation.reason_code = envelope.reason_code or envelope.result + observation.raw_result = raw.decode("utf-8") + observation.raw_result_sha256 = digest(raw) + # A mutation during the call cannot establish current evidence. + validate_external_currency(observation.model_dump(mode="json"), workspace) + return observation + except FileNotFoundError: + observation.status, observation.reason_code = "absent", "trusted_input_unavailable" + except (OSError, ValueError, ValidationError, UnicodeError, StopIteration): + observation.status, observation.reason_code = "invalid", "native_input_or_result_invalid" + return observation + + +def _composed_candidate_bytes(facts, snapshot, input_root: Path) -> bytes: + """Use selected authored shapes, retaining omission rather than DTO defaults. + + The native model rejects some explicit extension fields even when their + effective value is a runtime default. Contributor selection and generated + keys come from the static composer; its normalized policy must agree with + the reconstructed input before any execution. + """ + def document(path, kind): + artifact = next(item for item in snapshot.inventory["artifacts"] + if item["host"] == "openshell" and item["kind"] == kind and item["path"] == path) + target = (artifact.get("resolved_through") or [path])[-1] + raw = read_static_input_bytes(input_root / target, max_bytes=1024 * 1024) + captured = snapshot.cache.openshell_input_reads[target] + if captured.get("limit") or digest(raw) != "sha256:" + captured["sha256"] or len(raw) != captured["size_bytes"]: + raise ValueError("composition contributor identity changed after extraction") + return load_document(raw.decode("utf-8")) + + contributors = facts["composition"]["contributors"] + selected, = [row for row in contributors if row["selected"] and row["role"] != "profile"] + value = document(selected["path"], "openshell_policy") + for row in contributors: + if row["role"] != "profile" or not row["selected"]: + continue + profile = document(row["path"], "openshell_profile") + value.setdefault("network_policies", {})[row["rule_key"]] = { + "name": row["rule_key"], "endpoints": profile["endpoints"], + "binaries": [{"path": path} for path in profile["binaries"]], + } + candidate = json.dumps(value, sort_keys=True).encode() + if len(candidate) > 1024 * 1024 or parse_policy(candidate.decode()).model_dump(mode="json") != facts["policy"]: + raise ValueError("reconstructed composition differs from static policy identity") + return candidate + + +def _load_json(data): + def reject_constant(_value): + raise ValueError("native JSON contains a non-finite number") + + try: + value = json.loads(data, object_pairs_hook=_unique_object, parse_constant=reject_constant) + except RecursionError as exc: + raise ValueError("native JSON exceeds parser depth") from exc + pending, nodes = [(value, 0)], 0 + while pending: + item, depth = pending.pop() + nodes += 1 + if depth > 32 or nodes > 16384: + raise ValueError("native JSON exceeds structural bounds") + if isinstance(item, dict): + pending.extend((child, depth + 1) for child in item.values()) + elif isinstance(item, list): + pending.extend((child, depth + 1) for child in item) + return value + + +def native_findings(observation: NativeObservation, context): + from agents_shipgate.schemas.common import SourceReference + from agents_shipgate.schemas.report import EvidenceGap, EvidenceGapAction, Finding + + if observation.status == "within_boundary": + return [] + if observation.required and observation.status != "exceeds_boundary": + context.policy_evidence_gaps.append(EvidenceGap( + kind="invalid_evidence_provenance", subject="OpenShell required native containment", + source_type="openshell_native", policy_id="SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE", + why=f"Required modeled containment is unestablished: {observation.status} ({observation.reason_code}).", + next_action=EvidenceGapAction(kind="provide_policy_evidence", why="Only a trusted local execution establishes native containment.", + expects="An operator must repair the external trust configuration or unavailable execution, then rerun verify with the same proof options."))) + if observation.status == "absent" and not observation.required: + return [] + exceeds = observation.status == "exceeds_boundary" + return [Finding(check_id="SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED" if exceeds else "SHIP-VERIFY-OPENSHELL-PROOF-UNAVAILABLE", + title="OpenShell candidate exceeds the trusted boundary" if exceeds else "OpenShell native containment is unestablished", + severity="critical" if exceeds else "medium", category="verify", confidence="high", + provenance_kind="runtime_trace", agent_id=context.agent.id, + evidence={"status": observation.status, "reason_code": observation.reason_code, "required": observation.required}, + source=SourceReference(type="openshell_native", path=observation.candidate.get("selection", {}).get("registration")), + blocks_release=exceeds, + recommendation="Narrow the candidate and rerun trusted native proof; a successful proof does not approve other release obligations." if exceeds + else "Repair the external trust inputs or unsupported proof context and rerun verify; do not supply a repository-authored result.")] diff --git a/src/agents_shipgate/core/verification_input_currency.py b/src/agents_shipgate/core/verification_input_currency.py index a3b0302b..7ea8e882 100644 --- a/src/agents_shipgate/core/verification_input_currency.py +++ b/src/agents_shipgate/core/verification_input_currency.py @@ -190,6 +190,10 @@ def check(data: bytes, digest: str, size: int) -> None: if size > MAX_CURRENCY_INPUT_BYTES: raise ValueError("recorded input exceeds the currency read limit") check(reader.read_bytes(path, max_bytes=MAX_CURRENCY_INPUT_BYTES), digest, size) + if live_origins and "openshell_native" in plan.inputs.options: + from agents_shipgate.core.openshell_native import validate_external_currency + + validate_external_currency(plan.inputs.options["openshell_native"], root) validate_dependency_inputs(plan, root=root, snapshot=snapshot) validate_directory_inputs(plan, snapshot=snapshot) snapshot.finish() diff --git a/src/agents_shipgate/schemas/openshell_native.py b/src/agents_shipgate/schemas/openshell_native.py new file mode 100644 index 00000000..762e00fe --- /dev/null +++ b/src/agents_shipgate/schemas/openshell_native.py @@ -0,0 +1,107 @@ +"""Pinned native containment contract and operator-owned execution inputs.""" +from __future__ import annotations + +from typing import Literal + +from pydantic import BaseModel, ConfigDict, Field, model_validator + +from agents_shipgate.schemas.openshell import OpenShellPolicyReference +from agents_shipgate.schemas.verification_identity import VerificationSubject + +DOMAINS = ["filesystem", "network_l4", "network_rest", "process", "landlock"] + + +class NativeObject(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + +class PinnedFile(NativeObject): + path: str = Field(min_length=1, max_length=4096) + sha256: str = Field(pattern=r"^sha256:[a-f0-9]{64}$") + + +class CandidateSelection(NativeObject): + registration: str + path: str = "" + composition: str = "" + + @model_validator(mode="after") + def local_selection(self): + OpenShellPolicyReference(path=self.registration, role="authored") + if bool(self.path) == bool(self.composition): + raise ValueError("select exactly one document or local composition") + if self.path: + OpenShellPolicyReference(path=self.path, role="authored") + return self + + +class NativeTrustConfig(NativeObject): + version: Literal[1] + required: bool + runtime_version: Literal["0.1.2"] + prover_version: Literal["0.1.2"] + executable: PinnedFile + boundary: PinnedFile + candidate: CandidateSelection + required_domains: list[str] = Field(min_length=5, max_length=5) + timeout_seconds: int = Field(default=10, ge=1, le=30) + + @model_validator(mode="after") + def domain_contract(self): + if sorted(self.required_domains) != sorted(DOMAINS): + raise ValueError("all pinned containment domains are required") + return self + + +class Coverage(NativeObject): + domains: list[str] = Field(max_length=5) + + +class NativeInputs(NativeObject): + candidate: str + boundary: str + + +class NativeEnvelope(NativeObject): + schema_version: Literal[1] + prover_version: Literal["0.1.2"] + check: Literal["boundary"] + coverage: Coverage | None + result: Literal["within_boundary", "exceeds_boundary", "unsupported", "inconclusive", "error"] + exit_code: int + inputs: NativeInputs + counterexample: dict | None + reason_code: str | None + reason: str | None + + +class ExternalIdentity(NativeObject): + path: str + state: Literal["file", "absent", "unconfirmable"] + sha256: str | None = None + size_bytes: int = Field(default=0, ge=0, le=64 * 1024 * 1024) + + +class NativeObservation(NativeObject): + version: Literal[1] = 1 + provenance: Literal["not_executed", "local_trusted_execution"] = "not_executed" + status: Literal["absent", "within_boundary", "exceeds_boundary", "unsupported", "inconclusive", "error", "invalid", "timeout", "cancelled"] + reason_code: str + required: bool + external_inputs: list[ExternalIdentity] = Field(default_factory=list, max_length=3) + candidate: dict = Field(default_factory=dict) + invocation: list[str] = Field(default_factory=list) + required_domains: list[str] = Field(default_factory=lambda: list(DOMAINS)) + actual_exit_code: int | None = None + raw_result: str | None = Field(default=None, max_length=256 * 1024) + raw_result_sha256: str | None = None + modeled_containment_only: Literal[True] = True + deployment_enforcement_verified: Literal[False] = False + grants_merge_authority: Literal[False] = False + + +class NativeEvidence(NativeObject): + evidence_schema_version: Literal["1"] = "1" + observation: NativeObservation + request_id: str + subject: VerificationSubject diff --git a/src/agents_shipgate/schemas/verification.py b/src/agents_shipgate/schemas/verification.py index ef399b5b..0bc2d671 100644 --- a/src/agents_shipgate/schemas/verification.py +++ b/src/agents_shipgate/schemas/verification.py @@ -70,3 +70,6 @@ class VerificationContext(BaseModel): # serialization and never published. enabled_plugin_hooks: Any = Field(default=None, exclude=True, repr=False) enabled_plugin_hook_issues: tuple[Any, ...] = Field(default=(), exclude=True, repr=False) + + # Produced only by the opt-in trusted local execution, never by JSON import. + openshell_native: Any = Field(default=None, exclude=True, repr=False) diff --git a/tests/test_adapter_static_only.py b/tests/test_adapter_static_only.py index 9344e3be..d3dbd0d7 100644 --- a/tests/test_adapter_static_only.py +++ b/tests/test_adapter_static_only.py @@ -458,6 +458,36 @@ class AllowedException: "uses explicit repo-local override files, not dynamic imports." ), ), + AllowedException( + relative_path="core/openshell_native.py", + surface="import:subprocess", + line=11, + snippet="import subprocess", + rationale=( + "Explicit verify --openshell-proof-config execution uses one pinned " + "Popen boundary below. Default static commands never invoke it; " + "the operator must select external trust inputs and executable hashes." + ), + ), + AllowedException( + relative_path="core/openshell_native.py", + surface="attr_call:subprocess.Popen", + line=230, + snippet=( + "subprocess.Popen(command, cwd=root, shell=False, start_new_session=True, " + "stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, " + "env={'PATH': '/usr/bin:/bin', 'HOME': temp, 'LANG': 'C.UTF-8'}, " + "preexec_fn=lambda: _limits(seconds))" + ), + rationale=( + "Opt-in native containment copies captured, operator-pinned external " + "executable and boundary bytes into a private directory. It executes " + "fixed list argv without a shell or inherited credentials, bounds " + "the process group and output, and binds results to verifier currency. " + "No repository-discovered executable or imported result can substitute " + "for this call, and modeled containment grants no merge authority." + ), + ), ) diff --git a/tests/test_openshell_native.py b/tests/test_openshell_native.py new file mode 100644 index 00000000..97d63e7b --- /dev/null +++ b/tests/test_openshell_native.py @@ -0,0 +1,393 @@ +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest +from test_current_control import _live +from test_current_control import repo as repo # noqa: F401 +from test_openshell_composition import files as composed_files +from test_openshell_inputs import POLICY, REGISTRATION, register +from typer.testing import CliRunner + +from agents_shipgate.cli.main import app +from agents_shipgate.core.current_control import CurrentControlUnavailable, read_current_control +from agents_shipgate.core.openshell_native import ( + digest, + observe_native, + validate_envelope, + validate_external_currency, +) +from agents_shipgate.schemas.openshell_native import DOMAINS + + +def trusted(tmp_path, *, result="within_boundary", required=True, extra="", mutate=None): + trusted_root = tmp_path / "trusted" + trusted_root.mkdir(exist_ok=True) + executable = trusted_root / "prover" + envelope = {"schema_version": 1, "prover_version": "0.1.2", "check": "boundary", + "coverage": {"domains": DOMAINS}, "result": result, "exit_code": 0, + "inputs": {"candidate": "candidate.yaml", "boundary": "boundary.yaml"}, + "counterexample": None, "reason_code": None, "reason": None} + if result == "exceeds_boundary": + envelope.update(exit_code=1, counterexample={"domain": "filesystem", "access": "write", "path": "/outside"}) + elif result in {"unsupported", "inconclusive", "error"}: + envelope.update(exit_code=2 if result == "error" else 3, reason_code="invalid_input" if result == "error" else "unsupported_policy_shape", reason="Unsupported protocol fixture") + if result == "error": + envelope["coverage"] = None + if mutate: + mutate(envelope) + program = f"#!{sys.executable}\nimport json,sys,time\n{extra}\nprint({json.dumps(json.dumps(envelope))})\nsys.exit({envelope['exit_code']})\n" + executable.write_text(program) + executable.chmod(0o700) + boundary = trusted_root / "maximum.yaml" + boundary.write_text(POLICY) + config = {"version": 1, "required": required, "runtime_version": "0.1.2", "prover_version": "0.1.2", + "executable": {"path": str(executable), "sha256": digest(executable.read_bytes())}, + "boundary": {"path": str(boundary), "sha256": digest(boundary.read_bytes())}, + "candidate": {"registration": REGISTRATION, "path": "arbitrary.rules"}, + "required_domains": DOMAINS, "timeout_seconds": 1} + path = trusted_root / "trust.json" + path.write_text(json.dumps(config)) + return path, config + + +def observe(root, config_path, required=False): + return observe_native(config_path=config_path, required=required, workspace=root, input_root=root) + + +@pytest.mark.parametrize("outcome", ["within_boundary", "exceeds_boundary", "unsupported", "inconclusive", "error"]) +def test_native_outcomes_remain_distinct(tmp_path, outcome): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, _ = trusted(tmp_path, result=outcome) + observed = observe(root, path) + assert observed.status == outcome, observed + assert observed.raw_result and observed.raw_result_sha256 == digest(observed.raw_result.encode()) + assert observed.grants_merge_authority is False + assert len(observed.external_inputs) == 3 + + +@pytest.mark.parametrize("mutation", [ + lambda env: env.update(schema_version=2), + lambda env: env.update(prover_version="0.2.0"), + lambda env: env["coverage"].update(domains=["filesystem"]), + lambda env: env.update(exit_code=1), + lambda env: env["inputs"].update(candidate="other.yaml"), + lambda env: env.update(counterexample={"domain": "filesystem"}), +]) +def test_mismatched_native_envelope_cannot_pass(tmp_path, mutation): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, _ = trusted(tmp_path, mutate=mutation) + observed = observe(root, path) + assert observed.status == "invalid" and not observed.raw_result + + +@pytest.mark.parametrize("failure", ["timeout", "output", "malformed", "cancelled"]) +def test_execution_failures_are_not_passing_evidence(tmp_path, failure): + root = tmp_path / "repo" + root.mkdir() + register(root) + extra = {"timeout": "time.sleep(10)", "output": "print('x' * (300 * 1024))", + "malformed": "print('not JSON');sys.exit(0)", "cancelled": "sys.exit(130)"}[failure] + path, _ = trusted(tmp_path, extra=extra) + observed = observe(root, path) + assert observed.status in {"timeout", "error", "invalid"} + assert observed.actual_exit_code is not None + + +def test_repository_config_or_prover_boundary_cannot_self_authorize(tmp_path): + root = tmp_path / "repo" + root.mkdir() + register(root) + external, config = trusted(tmp_path) + local = root / "trust.json" + local.write_text(external.read_text()) + assert observe(root, local, required=True).status == "invalid" + for key in ("boundary", "executable"): + original = dict(config[key]) + inside = root / key + inside.write_bytes(Path(original["path"]).read_bytes()) + config[key]["path"] = str(inside) + external.write_text(json.dumps(config)) + assert observe(root, external).status == "invalid" + config[key] = original + + +@pytest.mark.parametrize("changed", ["config", "executable", "boundary"]) +def test_every_external_identity_invalidates_currency(tmp_path, changed): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, config = trusted(tmp_path) + observed = observe(root, path) + assert observed.status == "within_boundary" + target = path if changed == "config" else Path(config[changed]["path"]) + target.write_bytes(target.read_bytes() + b"\n") + with pytest.raises(ValueError): + validate_external_currency(observed.model_dump(mode="json"), root) + + +def test_forged_repository_result_cannot_be_imported_as_configuration(tmp_path): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, _ = trusted(tmp_path) + path.write_text(json.dumps({"version": 1, "result": "within_boundary", "candidate_sha256": digest(POLICY.encode())})) + assert observe(root, path, required=True).status == "invalid" + with pytest.raises(ValueError): + validate_envelope(b'{"schema_version":1,"schema_version":1}', 0) + + +def test_composed_candidate_binds_provenance_without_reading_credentials(tmp_path): + root = tmp_path / "repo" + root.mkdir() + for name, value in composed_files().items(): + target = root / name + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(value if isinstance(value, str) else json.dumps(value)) + path, config = trusted(tmp_path) + config["candidate"] = {"registration": REGISTRATION, "composition": "worker"} + path.write_text(json.dumps(config)) + observed = observe(root, path) + assert observed.status == "within_boundary", observed + assert observed.candidate["role"] == "composed" + assert any(row["role"] == "profile" for row in observed.candidate["composition"]["contributors"]) + + +@pytest.mark.parametrize("protocol", ["rest", "l4"]) +def test_composed_native_input_preserves_authored_omissions(tmp_path, protocol): + root = tmp_path / "repo" + root.mkdir() + contents = composed_files() + if protocol == "l4": + contents["profiles/github.profile"]["endpoints"] = [{"host": "api.provider.example", "port": 443}] + for name, value in contents.items(): + target = root / name + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(value if isinstance(value, str) else json.dumps(value)) + assertions = ( + "from pathlib import Path\n" + "value=json.loads(Path('candidate.yaml').read_text())\n" + "endpoint=value['network_policies']['_provider_work_github']['endpoints'][0]\n" + "assert 'persisted_queries' not in endpoint and 'graphql_max_body_bytes' not in endpoint\n" + "assert 'credential_signing' not in endpoint and 'mcp' not in endpoint\n" + ) + if protocol == "l4": + assertions += "assert 'enforcement' not in endpoint and 'protocol' not in endpoint\n" + path, config = trusted(tmp_path, extra=assertions) + config["candidate"] = {"registration": REGISTRATION, "composition": "worker"} + path.write_text(json.dumps(config)) + assert observe(root, path).status == "within_boundary" + + +def verify(root, *flags): + result = CliRunner().invoke(app, ["verify", "--workspace", str(root), "--base", "main", "--json", *flags]) + assert result.exit_code in {0, 10, 20}, result.output + return json.loads(result.output) + + +def test_required_absent_proof_routes_through_existing_insufficient_evidence(repo): + result = verify(repo, "--openshell-proof-required") + assert result["release_decision"]["decision"] == "insufficient_evidence" + assert result["control"]["permissions"]["report_complete"] is False + assert "--openshell-proof-required" in json.dumps(result["control"]) + + +def test_default_verifier_never_executes_a_prover(repo, monkeypatch): + monkeypatch.setattr("agents_shipgate.core.openshell_native._execute", lambda *args: pytest.fail("unexpected native execution")) + verify(repo) + plan = json.loads((repo / "agents-shipgate-reports/verification-plan.json").read_text()) + assert "openshell_native" not in plan["inputs"]["options"] + + +def test_invalid_selected_trust_configuration_cannot_complete_clean_verification(repo, tmp_path): + path, _ = trusted(tmp_path) + path.write_text('{"version":1,"required":true}') + result = verify(repo, "--openshell-proof-config", str(path)) + assert result["control"]["permissions"]["report_complete"] is False + assert result["release_decision"]["decision"] != "passed" + + +def test_optional_absent_proof_is_not_success_even_when_other_checks_complete(repo, tmp_path): + missing = tmp_path / "not-supplied.json" + result = verify(repo, "--openshell-proof-config", str(missing)) + evidence = json.loads((repo / "agents-shipgate-reports/openshell-native.json").read_text()) + assert evidence["observation"]["status"] == "absent" + assert evidence["observation"]["provenance"] == "not_executed" + assert result["release_decision"]["decision"] == "passed" + missing.write_text("{}") + with pytest.raises(CurrentControlUnavailable): + read_current_control(repo / "agents-shipgate-reports", live=lambda: _live(repo)) + + +def test_sensitive_trust_config_path_is_never_published_in_rerun_commands(repo, tmp_path): + secret = "sk-live-0123456789abcdef0123456789abcdef" + path = tmp_path / secret + path.write_text("{}") + result = CliRunner().invoke(app, ["verify", "--workspace", str(repo), "--base", "main", "--openshell-proof-config", str(path), "--json"]) + assert result.exit_code == 2 + assert secret not in result.output + for artifact in (repo / "agents-shipgate-reports").glob("*"): + if artifact.is_file(): + assert secret not in artifact.read_text() + + +@pytest.mark.parametrize("alias", ["dotdot", "symlink", "relative"]) +def test_cli_rejects_trust_path_aliases_before_writing_any_artifact(repo, tmp_path, alias): + path, _ = trusted(tmp_path) + if alias == "dotdot": + selected = path.parent / "unused" / ".." / path.name + elif alias == "symlink": + selected = path.parent / "alias.json" + selected.symlink_to(path) + else: + selected = Path("relative-trust.json") + result = CliRunner().invoke(app, ["verify", "--workspace", str(repo), "--base", "main", "--openshell-proof-config", str(selected), "--json"]) + assert result.exit_code == 2 + assert not (repo / "agents-shipgate-reports").exists() + + +@pytest.mark.parametrize("raw", ["[" * 1100 + "0" + "]" * 1100, "[" * 40 + "0" + "]" * 40, '{"version":NaN}']) +def test_malformed_native_json_is_a_bounded_invalid_observation(tmp_path, raw): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, _ = trusted(tmp_path) + path.write_text(raw) + assert observe(root, path).status == "invalid" + with pytest.raises(ValueError): + validate_envelope(raw.encode(), 0) + + +def test_native_receipt_binds_inputs_and_external_executable_currency(repo, tmp_path): + register(repo) + path, config = trusted(tmp_path) + result = verify(repo, "--openshell-proof-config", str(path)) + out = repo / "agents-shipgate-reports" + evidence = json.loads((out / "openshell-native.json").read_text()) + assert evidence["observation"]["status"] == "within_boundary" + assert evidence["subject"]["git"]["head_commit_sha"] + manifest = json.loads((out / "verification-artifacts.json").read_text()) + assert "openshell_native_json" in manifest["artifacts"] + read_current_control(out, live=lambda: _live(repo)) + Path(config["executable"]["path"]).write_text("changed executable") + with pytest.raises(CurrentControlUnavailable): + read_current_control(out, live=lambda: _live(repo)) + assert result["control"]["permissions"]["merge"] is False + + +def test_native_artifact_supports_unignored_output_reruns_and_is_cleared_by_default_verify(repo, tmp_path): + from agents_shipgate.cli.current_workspace import live_workspace + + register(repo) + path, _ = trusted(tmp_path) + out = repo / "proof-output" + for _ in range(2): + verify(repo, "--openshell-proof-config", str(path), "--out", str(out)) + assert json.loads((out / "openshell-native.json").read_text())["observation"]["status"] == "within_boundary" + read_current_control(out, live=lambda: live_workspace(repo, out)) + verify(repo, "--out", str(out)) + assert not (out / "openshell-native.json").exists() + assert "openshell_native_json" not in json.loads((out / "verification-artifacts.json").read_text())["artifacts"] + + +def test_native_counterexample_words_do_not_change_finding_identity(repo, tmp_path): + register(repo) + path, _ = trusted(tmp_path, result="exceeds_boundary") + first = verify(repo, "--openshell-proof-config", str(path)) + assert first["release_decision"]["decision"] == "blocked" + report = json.loads((repo / "agents-shipgate-reports/report.json").read_text()) + old, = [row for row in report["findings"] if row["check_id"] == "SHIP-VERIFY-OPENSHELL-BOUNDARY-EXCEEDED"] + path, _ = trusted(tmp_path, result="exceeds_boundary", mutate=lambda env: env["counterexample"].update(path="/different-wording")) + verify(repo, "--openshell-proof-config", str(path)) + report = json.loads((repo / "agents-shipgate-reports/report.json").read_text()) + new, = [row for row in report["findings"] if row["check_id"] == old["check_id"]] + assert old["fingerprint"] == new["fingerprint"] + + +def test_optional_missing_proof_is_absent_and_appearing_input_is_stale(tmp_path): + root = tmp_path / "repo" + root.mkdir() + missing = tmp_path / "missing.json" + observation = observe(root, missing) + assert observation.status == "absent" and observation.provenance == "not_executed" + validate_external_currency(observation.model_dump(mode="json"), root) + missing.write_text("{}") + with pytest.raises(ValueError): + validate_external_currency(observation.model_dump(mode="json"), root) + + +def test_valid_native_cancellation_is_distinct(tmp_path): + root = tmp_path / "repo" + root.mkdir() + register(root) + def cancel(envelope): + envelope.update(exit_code=130, reason_code="cancelled") + path, _ = trusted(tmp_path, result="inconclusive", mutate=cancel) + observed = observe(root, path) + assert observed.status == "cancelled" and observed.actual_exit_code == 130 + + +def test_pin_mismatch_never_executes_prover(tmp_path, monkeypatch): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, config = trusted(tmp_path) + Path(config["boundary"]["path"]).write_text(POLICY + "\n") + monkeypatch.setattr("agents_shipgate.core.openshell_native._execute", lambda *args: pytest.fail("untrusted execution")) + assert observe(root, path).status == "invalid" + + +def test_prover_gets_captured_inputs_and_clean_environment(tmp_path, monkeypatch): + root = tmp_path / "repo" + root.mkdir() + register(root) + monkeypatch.setenv("GITHUB_TOKEN", "fixture-secret") + monkeypatch.setenv("PYTHONPATH", str(root)) + path, _ = trusted(tmp_path, extra="import os\nassert 'GITHUB_TOKEN' not in os.environ\nassert 'PYTHONPATH' not in os.environ\nassert open('candidate.yaml').read() == open('boundary.yaml').read()") + assert observe(root, path).status == "within_boundary" + + +@pytest.mark.parametrize("alias", ["link", "hardlink", "writable"]) +def test_trust_files_cannot_be_aliases_or_group_writable(tmp_path, alias): + root = tmp_path / "repo" + root.mkdir() + register(root) + path, config = trusted(tmp_path) + boundary = Path(config["boundary"]["path"]) + if alias == "writable": + boundary.chmod(0o666) + else: + replacement = boundary.with_name("alias") + if alias == "link": + replacement.symlink_to(boundary) + else: + replacement.hardlink_to(boundary) + config["boundary"]["path"] = str(replacement) + path.write_text(json.dumps(config)) + assert observe(root, path).status == "invalid" + + +def test_output_directory_never_overwrites_native_trust_inputs(repo, tmp_path): + register(repo) + path, config = trusted(tmp_path) + boundary = Path(config["boundary"]["path"]) + original = boundary.read_bytes() + result = CliRunner().invoke(app, ["verify", "--workspace", str(repo), "--base", "main", + "--openshell-proof-config", str(path), "--out", str(boundary.parent)]) + assert result.exit_code == 2 and "overlap" in result.output.lower() + assert boundary.read_bytes() == original + + +def test_worker_cannot_import_native_proof_plan(repo, tmp_path): + register(repo) + path, _ = trusted(tmp_path) + verify(repo, "--openshell-proof-config", str(path)) + plan = repo / "agents-shipgate-reports/verification-plan.json" + result = CliRunner().invoke(app, ["verification", "worker", "--workspace", str(repo), "--plan", str(plan)]) + assert result.exit_code != 0 and "cannot be imported" in str(result.exception) From 21446d0e7fedd305045be74838915f287a844b19 Mon Sep 17 00:00:00 2001 From: Pengfei Hu Date: Sat, 3 Oct 2026 02:27:03 -0700 Subject: [PATCH 2/2] test(openshell): include native evidence in scan cleanup expectations --- tests/test_scan.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_scan.py b/tests/test_scan.py index 3fd10f2f..5bb61ef4 100644 --- a/tests/test_scan.py +++ b/tests/test_scan.py @@ -49,6 +49,7 @@ def test_scan_removes_stale_verifier_route_artifacts(tmp_path: Path) -> None: "verification-unit-result.json", "verification-artifacts.json", "verification-receipt.json", + "openshell-native.json", "human-authorization.json", # Identity-bearing too: it names the ``input_set_id`` of the run that # produced it, so a stale one beside a later run's receipt would offer