From 9fb89b7288df110fe78f7c8a47b04af0acb91e18 Mon Sep 17 00:00:00 2001 From: Jonathan Hefner Date: Tue, 29 Sep 2026 06:56:52 -0700 Subject: [PATCH 1/2] Make incremental report recording concurrency-safe Concurrent recordings can read the same report and overwrite one another even when every command succeeds. Serialize the local read/merge/publish transaction with an expiring filesystem lock, re-read the report under that lock, and keep HTTP health requests outside it. Publish a nonempty lock directory with a unique owner so stale cleanup cannot remove a successor. Expire abandoned owners after ten seconds, retry acquisition for up to fifteen seconds, and preserve the existing atomic report replacement and last-committed observation semantics. Add process-level regression coverage for concurrent records, same-key replacements, delayed health checks, overlapping starts, held locks, and recovery after a writer is killed. Agents can offload evidence as needed without coordinating individual writes. --- plugins/agent-plugins-conformance/README.md | 2 +- .../skills/run-conformance/scripts/report.mjs | 121 +++++- test/report-concurrency.test.mjs | 344 ++++++++++++++++++ 3 files changed, 446 insertions(+), 21 deletions(-) create mode 100644 test/report-concurrency.test.mjs diff --git a/plugins/agent-plugins-conformance/README.md b/plugins/agent-plugins-conformance/README.md index 7886ccd..28e389b 100644 --- a/plugins/agent-plugins-conformance/README.md +++ b/plugins/agent-plugins-conformance/README.md @@ -13,7 +13,7 @@ Results describe the observed run and covered cases; they are not whole-client c > Run the run-conformance skill from Agent Plugins Conformance and save the JSON report to /tmp/agent-plugins-conformance/report.json. -The [run-conformance skill](skills/run-conformance/SKILL.md) initializes the destination, then records observations as they are collected. The reporter creates missing parent directories and updates the JSON file after each recording. Starting a run replaces any previous report at the same path, so concurrently active runs need distinct destinations. +The [run-conformance skill](skills/run-conformance/SKILL.md) initializes the destination, then records observations as they are collected. The reporter creates missing parent directories and safely merges concurrent recordings into the JSON file. Starting a run replaces any previous report at the same path, so concurrently active runs need distinct destinations. When collection ends, the agent gives the report's absolute path and any collection or recording limitations. A saved report may be an incomplete snapshot if collection was interrupted; file existence alone does not establish completion. diff --git a/plugins/agent-plugins-conformance/skills/run-conformance/scripts/report.mjs b/plugins/agent-plugins-conformance/skills/run-conformance/scripts/report.mjs index 58a279a..face5ee 100644 --- a/plugins/agent-plugins-conformance/skills/run-conformance/scripts/report.mjs +++ b/plugins/agent-plugins-conformance/skills/run-conformance/scripts/report.mjs @@ -1,6 +1,7 @@ import { randomUUID } from 'node:crypto'; -import { mkdir, open, readFile, rename, rm } from 'node:fs/promises'; -import { dirname, isAbsolute, join } from 'node:path'; +import { mkdir, mkdtemp, open, readFile, readdir, realpath, rename, rm, rmdir, stat, unlink, utimes, writeFile } from 'node:fs/promises'; +import { basename, dirname, isAbsolute, join } from 'node:path'; +import { setTimeout } from 'node:timers/promises'; import { buildReport, observationKey, validateInput } from '../../../src/report.mjs'; import { validateSavedReport } from '../../../src/report-format.mjs'; import { checkHttpHealth } from '../../../src/http-health.mjs'; @@ -37,6 +38,81 @@ async function writeReport(outputPath, report) { } } +async function readReport(outputPath) { + let saved; + try { + saved = JSON.parse(await readFile(outputPath, 'utf8')); + } catch (error) { + if (error.code === 'ENOENT') throw new Error('Report does not exist; start collection first'); + throw error; + } + validateSavedReport(saved); + validateInput({ schemaVersion: saved.schemaVersion, observations: saved.observations }); + return saved; +} + +async function withReportLock(outputPath, update) { + const lockPath = `${outputPath}.lock`; + const staleAfter = 10_000; + const deadline = performance.now() + 15_000; + const candidate = await mkdtemp(`${lockPath}-`); + const owner = randomUUID(); + try { + await writeFile(join(candidate, owner), '', { flag: 'wx', mode: 0o600 }); + for (;;) { + const now = new Date(); + await utimes(join(candidate, owner), now, now); + try { + // Publish an already nonempty directory so cleanup cannot remove a new owner. + await rename(candidate, lockPath); + break; + } catch (error) { + if (!['EEXIST', 'ENOTEMPTY', 'EPERM', 'EACCES'].includes(error.code)) throw error; + let owners; + try { + owners = await readdir(lockPath); + } catch (readError) { + if (readError.code !== 'ENOENT') throw readError; + owners = []; + } + if (owners.length > 1 || (owners.length === 1 && !/^[\da-f]{8}(?:-[\da-f]{4}){3}-[\da-f]{12}$/.test(owners[0]))) { + throw new Error(`Unexpected report lock contents: ${lockPath}`); + } + // A stale observer can remove only the owner it actually inspected. + if (owners.length) { + const ownerPath = join(lockPath, owners[0]); + try { + if (Date.now() - (await stat(ownerPath)).mtimeMs > staleAfter) await unlink(ownerPath); + } catch (cleanupError) { + if (cleanupError.code !== 'ENOENT') throw cleanupError; + } + } + await removeEmptyLock(lockPath); + if (performance.now() >= deadline) { + throw new Error(`Timed out waiting for report lock: ${lockPath}. Retry the recording.`); + } + await setTimeout(25 + Math.random() * 50); + } + } + try { + await update(); + } finally { + await rm(join(lockPath, owner), { force: true }); + await removeEmptyLock(lockPath); + } + } finally { + await rm(candidate, { recursive: true, force: true }); + } +} + +async function removeEmptyLock(lockPath) { + try { + await rmdir(lockPath); + } catch (error) { + if (!['ENOENT', 'ENOTEMPTY', 'EEXIST'].includes(error.code)) throw error; + } +} + try { const [outputPath, ...extra] = process.argv.slice(2); if (!outputPath || !isAbsolute(outputPath) || extra.length) { @@ -46,30 +122,35 @@ try { for await (const chunk of process.stdin) chunks.push(chunk); const message = JSON.parse(Buffer.concat(chunks).toString('utf8')); validateMessage(message); - let report; + let incoming; if (message.action === 'start') { - report = buildReport({ schemaVersion: 1, observations: [] }); await mkdir(dirname(outputPath), { recursive: true }); } else { - let saved; - try { - saved = JSON.parse(await readFile(outputPath, 'utf8')); - } catch (error) { - if (error.code === 'ENOENT') throw new Error('Report does not exist; start collection first'); - throw error; - } - validateSavedReport(saved); // Validate all saved and incoming evidence before making a health request. - validateInput({ schemaVersion: saved.schemaVersion, observations: saved.observations }); - const incoming = message.observation; + await readReport(outputPath); + incoming = message.observation; validateInput({ schemaVersion: 1, observations: [incoming] }, { recording: true }); - const key = observationKey(incoming); - const observations = saved.observations.filter((existing) => observationKey(existing) !== key); - observations.push(incoming.kind === 'mcp-streamable-http' - ? { ...incoming, serverHealthCheck: await checkHttpHealth() } : incoming); - report = buildReport({ schemaVersion: saved.schemaVersion, observations }); + // Network latency must not hold up other recordings. + if (incoming.kind === 'mcp-streamable-http') { + incoming = { ...incoming, serverHealthCheck: await checkHttpHealth() }; + } } - await writeReport(outputPath, report); + // Parent aliases must use the same lock and publication destination. + const destination = join(await realpath(dirname(outputPath)), basename(outputPath)); + await withReportLock(destination, async () => { + let report; + if (message.action === 'start') { + report = buildReport({ schemaVersion: 1, observations: [] }); + } else { + // Re-read inside the lock; the preflight snapshot may already be obsolete. + const saved = await readReport(destination); + const key = observationKey(incoming); + const observations = saved.observations.filter((existing) => observationKey(existing) !== key); + observations.push(incoming); + report = buildReport({ schemaVersion: saved.schemaVersion, observations }); + } + await writeReport(destination, report); + }); console.log(message.action === 'start' ? 'Report started.' : 'Observation recorded.'); } catch (error) { console.error(`Report error: ${error.message}`); diff --git a/test/report-concurrency.test.mjs b/test/report-concurrency.test.mjs new file mode 100644 index 0000000..b5fe7ea --- /dev/null +++ b/test/report-concurrency.test.mjs @@ -0,0 +1,344 @@ +import assert from 'node:assert/strict'; +import { spawn } from 'node:child_process'; +import { randomUUID } from 'node:crypto'; +import { mkdir, mkdtemp, readFile, readdir, realpath, rename, rm, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { basename, dirname, join } from 'node:path'; +import { setTimeout as delay } from 'node:timers/promises'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import { isDeepStrictEqual } from 'node:util'; +import test from 'node:test'; +import { buildReport } from '../plugins/agent-plugins-conformance/src/report.mjs'; + +const reporter = fileURLToPath(new URL( + '../plugins/agent-plugins-conformance/skills/run-conformance/scripts/report.mjs', import.meta.url, +)); +const start = { action: 'start' }; +const skill = (name, marker = `APC_${name.toUpperCase()}_V1`) => ({ + kind: 'skill', skill: `conformance-${name}`, marker, +}); +const discoveryServers = [ + 'recovery-cwd-invalid-form', + 'recovery-cwd-escape', + 'recovery-cwd-data-escape', + 'recovery-cwd-symlink-escape', + 'recovery-command-symlink-posix', + 'recovery-command-symlink-windows', + 'recovery-unknown-field', + 'recovery-missing-type', + 'recovery-env-plugin-root', + 'recovery-env-plugin-data', + 'recovery-http-type', + 'recovery-http-non-loopback', + 'recovery-http-relative-url', + 'recovery-http-fragment', + 'recovery-http-userinfo', + 'recovery-http-duplicate-headers', + 'recovery-http-header-name', + 'recovery-http-header-value', + 'recovery-sse-non-loopback', + 'recovery-sse-relative-url', +]; +const discovery = (server) => ({ kind: 'mcp-discovery', server, advertised: false }); +const http = () => ({ + kind: 'mcp-streamable-http', server: 'http', evidence: { + type: 'request', version: 1, pathname: '/conformance/mcp', query: [['value', '$APC_HTTP_VALUE']], + headers: { 'x-apc-fixture': '${PLUGIN_ROOT}|${PLUGIN_DATA}|fixture value with spaces' }, + }, +}); + +function spawnReporter(fixture, message, nodeArgs = []) { + const child = spawn(process.execPath, [...nodeArgs, reporter, fixture.outputPath], { + cwd: fixture.directory, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }); + let stdout = ''; + let stderr = ''; + child.stdout.setEncoding('utf8').on('data', (chunk) => { stdout += chunk; }); + child.stderr.setEncoding('utf8').on('data', (chunk) => { stderr += chunk; }); + const result = new Promise((resolve, reject) => { + child.on('error', reject); + child.on('close', (code, signal) => resolve({ code, signal, stdout, stderr })); + }); + const running = { child, result }; + fixture.processes.add(running); + result.then( + () => fixture.processes.delete(running), + () => fixture.processes.delete(running), + ); + child.stdin.end(JSON.stringify(message)); + return { child, result }; +} + +async function runReporter(fixture, message, nodeArgs = []) { + return spawnReporter(fixture, message, nodeArgs).result; +} + +function assertSuccess(result, action = 'record') { + assert.equal(result.code, 0, result.stderr); + assert.equal(result.signal, null); + assert.equal(result.stderr, ''); + assert.equal(result.stdout, action === 'start' ? 'Report started.\n' : 'Observation recorded.\n'); +} + +async function makeFixture(t, observations = []) { + const directory = await mkdtemp(join(tmpdir(), 'report-concurrency-')); + const outputPath = join(directory, 'nested directory', 'report with spaces.json'); + await mkdir(dirname(outputPath), { recursive: true }); + const fixture = { directory, outputPath, processes: new Set() }; + t.after(async () => { + for (const { child } of fixture.processes) child.kill('SIGKILL'); + await Promise.allSettled([...fixture.processes].map(({ result }) => result)); + await rm(directory, { recursive: true, force: true }); + }); + assertSuccess(await runReporter(fixture, start), 'start'); + for (const observation of observations) { + assertSuccess(await runReporter(fixture, { action: 'record', observation })); + } + return fixture; +} + +async function readReport(fixture) { + return JSON.parse(await readFile(fixture.outputPath, 'utf8')); +} + +async function assertUnlocked(fixture) { + await assert.rejects(readdir(`${fixture.outputPath}.lock`), { code: 'ENOENT' }); +} + +async function waitFor(check, description, timeout = 10_000) { + const deadline = performance.now() + timeout; + for (;;) { + if (await check()) return; + if (performance.now() >= deadline) throw new Error(`Timed out waiting for ${description}`); + await delay(10); + } +} + +async function makeFsBarrier(fixture, mode) { + const barrier = await mkdtemp(join(fixture.directory, 'barrier-')); + const release = join(barrier, 'release'); + const preload = join(barrier, 'preload.mjs'); + await writeFile(preload, ` + import fs from 'node:fs/promises'; + import { syncBuiltinESMExports } from 'node:module'; + import { setTimeout as delay } from 'node:timers/promises'; + + const target = ${JSON.stringify(fixture.outputPath)}; + const barrier = ${JSON.stringify(barrier)}; + const release = ${JSON.stringify(release)}; + const mode = ${JSON.stringify(mode)}; + let firstTargetRead = true; + let firstTemporaryOpen = true; + const originalOpen = fs.open.bind(fs); + const originalReadFile = fs.readFile.bind(fs); + + async function signalAndWait() { + await fs.writeFile(barrier + '/' + process.pid + '.ready', 'ready'); + for (;;) { + try { + await fs.access(release); + return; + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + await delay(10); + } + } + + fs.readFile = async function readFile(path, ...args) { + const value = await originalReadFile(path, ...args); + if (mode === 'read' && firstTargetRead && String(path) === target) { + firstTargetRead = false; + await signalAndWait(); + } + return value; + }; + + fs.open = async function open(path, ...args) { + const handle = await originalOpen(path, ...args); + const name = String(path); + if (mode === 'open' && firstTemporaryOpen && name.includes('.conformance-report-') && name.endsWith('.tmp')) { + firstTemporaryOpen = false; + await signalAndWait(); + } + return handle; + }; + + syncBuiltinESMExports(); + `); + return { + nodeArgs: ['--import', pathToFileURL(preload).href], + release: () => writeFile(release, 'release'), + waitForReady: (count) => waitFor(async () => ( + await readdir(barrier) + ).filter((name) => name.endsWith('.ready')).length >= count, `${count} reporter processes at the barrier`), + }; +} + +async function makeFetchBarrier(fixture) { + const barrier = await mkdtemp(join(fixture.directory, 'fetch-barrier-')); + const ready = join(barrier, 'ready'); + const release = join(barrier, 'release'); + const preload = join(barrier, 'preload.mjs'); + await writeFile(preload, ` + import * as fs from 'node:fs/promises'; + import { setTimeout as delay } from 'node:timers/promises'; + + globalThis.fetch = async () => { + await fs.writeFile(${JSON.stringify(ready)}, 'ready'); + for (;;) { + try { + await fs.access(${JSON.stringify(release)}); + break; + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + await delay(10); + } + return { + status: 200, + json: async () => ({ fixture: 'agent-plugins-conformance-http', version: 1 }), + }; + }; + `); + return { + nodeArgs: ['--import', pathToFileURL(preload).href], + release: () => writeFile(release, 'release'), + waitForReady: () => waitFor(async () => { + try { + await readFile(ready); + return true; + } catch (error) { + if (error.code === 'ENOENT') return false; + throw error; + } + }, 'HTTP reporter health request'), + }; +} + +test('concurrent processes preserve records derived from the same preflight snapshot', async (t) => { + const fixture = await makeFixture(t); + const barrier = await makeFsBarrier(fixture, 'read'); + const observations = discoveryServers.map(discovery); + const writers = observations.map((observation) => spawnReporter( + fixture, { action: 'record', observation }, barrier.nodeArgs, + )); + + await barrier.waitForReady(writers.length); + await barrier.release(); + for (const writer of writers) assertSuccess(await writer.result); + + assert.deepEqual(await readReport(fixture), buildReport({ schemaVersion: 1, observations })); + await assertUnlocked(fixture); +}); + +test('concurrent same-key records publish one complete replacement', async (t) => { + const fixture = await makeFixture(t); + const barrier = await makeFsBarrier(fixture, 'read'); + const candidates = [skill('alpha'), skill('alpha', 'incorrect marker')]; + const writers = candidates.map((observation) => spawnReporter( + fixture, { action: 'record', observation }, barrier.nodeArgs, + )); + + await barrier.waitForReady(writers.length); + await barrier.release(); + for (const writer of writers) assertSuccess(await writer.result); + + const report = await readReport(fixture); + assert.equal(report.observations.length, 1); + assert.ok(candidates.some((candidate) => isDeepStrictEqual(report.observations[0], candidate))); + assert.deepEqual(report, buildReport({ schemaVersion: 1, observations: report.observations })); + await assertUnlocked(fixture); +}); + +test('a record committed while HTTP health waits survives the delayed writer', async (t) => { + const oldObservation = skill('alpha'); + const concurrentObservation = skill('beta'); + const httpObservation = http(); + const fixture = await makeFixture(t, [oldObservation]); + const barrier = await makeFetchBarrier(fixture); + const delayedWriter = spawnReporter(fixture, { + action: 'record', observation: httpObservation, + }, barrier.nodeArgs); + + await barrier.waitForReady(); + assertSuccess(await runReporter(fixture, { + action: 'record', observation: concurrentObservation, + })); + await barrier.release(); + assertSuccess(await delayedWriter.result); + + assert.deepEqual(await readReport(fixture), buildReport({ + schemaVersion: 1, + observations: [oldObservation, concurrentObservation, { ...httpObservation, serverHealthCheck: 'passed' }], + })); + await assertUnlocked(fixture); +}); + +test('start cannot be undone by a record whose preflight read saw the old report', async (t) => { + const oldObservation = skill('alpha'); + const newObservation = skill('beta'); + const fixture = await makeFixture(t, [oldObservation]); + const barrier = await makeFsBarrier(fixture, 'read'); + const writer = spawnReporter(fixture, { action: 'record', observation: newObservation }, barrier.nodeArgs); + + await barrier.waitForReady(1); + assertSuccess(await runReporter(fixture, start), 'start'); + await barrier.release(); + assertSuccess(await writer.result); + + assert.deepEqual(await readReport(fixture), buildReport({ schemaVersion: 1, observations: [newObservation] })); + await assertUnlocked(fixture); +}); + +test('a writer waits for a held report lock and proceeds after release', async (t) => { + const oldObservation = skill('alpha'); + const newObservation = skill('beta'); + const fixture = await makeFixture(t, [oldObservation]); + const destination = join(await realpath(dirname(fixture.outputPath)), basename(fixture.outputPath)); + const lockPath = `${destination}.lock`; + await mkdir(lockPath); + const owner = randomUUID(); + await writeFile(join(lockPath, owner), ''); + const writer = spawnReporter(fixture, { action: 'record', observation: newObservation }); + + await delay(250); + assert.equal(writer.child.exitCode, null); + await rename(lockPath, `${lockPath}-released`); + assertSuccess(await writer.result); + + assert.deepEqual(await readReport(fixture), buildReport({ + schemaVersion: 1, observations: [oldObservation, newObservation], + })); + await assertUnlocked(fixture); +}); + +test('writers recover an expired lock after its owner is killed and preserve every record', { timeout: 30_000 }, async (t) => { + const oldObservation = skill('alpha'); + const fixture = await makeFixture(t, [oldObservation]); + const before = await readFile(fixture.outputPath, 'utf8'); + const barrier = await makeFsBarrier(fixture, 'open'); + const interrupted = spawnReporter(fixture, { + action: 'record', observation: skill('beta'), + }, barrier.nodeArgs); + + await barrier.waitForReady(1); + assert.equal(interrupted.child.kill('SIGKILL'), true); + const interruptedResult = await interrupted.result; + assert.notEqual(interruptedResult.code, 0); + assert.equal(await readFile(fixture.outputPath, 'utf8'), before); + assert.deepEqual(await readReport(fixture), buildReport({ schemaVersion: 1, observations: [oldObservation] })); + + const observations = discoveryServers.map(discovery); + const writers = observations.map((observation) => spawnReporter( + fixture, { action: 'record', observation }, + )); + for (const writer of writers) assertSuccess(await writer.result); + + assert.deepEqual(await readReport(fixture), buildReport({ + schemaVersion: 1, observations: [oldObservation, ...observations], + })); + await assertUnlocked(fixture); +}); From 190e5557575e7670f3f7b216dcd1fa526d2d9c7b Mon Sep 17 00:00:00 2001 From: Jonathan Hefner Date: Tue, 29 Sep 2026 06:57:09 -0700 Subject: [PATCH 2/2] Clarify native conformance evidence collection Organize the guide around the fixture inventory, client discovery, runtime outcomes, and completion gaps. Keep observation collection distinct from the deterministic evaluator, and preserve native payloads and diagnostics without treating parsing failures or catalog searches as transport results. Incremental recording lets agents offload evidence without retaining the whole run. Allow collection and recording in parallel, retain retry guidance, and require only successful initialization before recording. Native GPT-5.6 Luna trials exposed an omitted redirect probe and an incorrect no-gaps response. Move the general tool-collection instruction ahead of discovery rules and explicitly include unavailable Core transports in completion gaps. Two subsequent medium-effort Codex trials collected every available fixture, preserved the evidence exactly, and disclosed the unavailable SSE fixtures. --- .../skills/run-conformance/SKILL.md | 215 ++++++++++++------ 1 file changed, 150 insertions(+), 65 deletions(-) diff --git a/plugins/agent-plugins-conformance/skills/run-conformance/SKILL.md b/plugins/agent-plugins-conformance/skills/run-conformance/SKILL.md index 657af9e..75f06cc 100644 --- a/plugins/agent-plugins-conformance/skills/run-conformance/SKILL.md +++ b/plugins/agent-plugins-conformance/skills/run-conformance/SKILL.md @@ -5,27 +5,25 @@ description: Collect Agent Plugins conformance observations through the current # Run the conformance probes -Collect observations from the installed Agent Plugins Conformance fixtures. Use the reporting script beside this skill to maintain the JSON report as you collect evidence. The reporter assigns outcomes for the covered cases; your role is to collect and record observations faithfully. +Observe the installed Agent Plugins Conformance fixtures through the client's normal skill and MCP surfaces. The reporting script beside this skill is the durable evidence store and deterministic evaluator; it assigns `pass`, `fail`, and `not_verified`. Record what the client exposes or returns without interpreting those outcomes yourself. -## Start the report +Client catalogs, resources delivered through the client's normal loading mechanism, successful MCP results, and native client diagnostics are evidence. A skill body delivered by that mechanism remains client-loaded evidence however the client represents it in the conversation. Skill text copied into a request without that provenance is not. Independently found package files may establish fixture or control facts, but not client advertisement, loading, or execution. -1. Identify the absolute JSON report path supplied by the user. If none was supplied, ask for it before starting. -2. Locate `scripts/report.mjs` relative to this client-loaded skill and use its absolute path in commands. Node.js 22 or newer must be available as `node`, and you must have a command-execution tool. If either is unavailable, explain the limitation. -3. Invoke the script with the user's absolute report path as its only argument. Supply this JSON message through stdin: +## Initialize the report - ```json - {"action":"start"} - ``` +Use the user's exact absolute JSON report path. If none was supplied, ask for it. Locate `scripts/report.mjs` relative to this client-loaded skill and use its absolute path. This workflow requires a command-execution tool and Node.js 22 or newer as `node`; if either is unavailable, explain the limitation. - Start every new run this way and confirm that initialization succeeds before collecting observations. The script creates missing parent directories and replaces any existing report at that exact path with a fresh report. All checks initially have status `not_verified`. Concurrently active runs must use distinct report paths. +Initialize the report by invoking the reporter with the report path as its only argument and sending this single message through stdin: -Each invocation has this form: +```json +{"action":"start"} +``` ```text node /scripts/report.mjs ``` -Send one JSON message through the command tool's stdin facility. If the tool accepts only shell commands, use the shell's literal-input mechanism. For example: +If the command tool accepts only shell commands, use its literal-input facility so it does not expand the JSON. For example, in a POSIX shell: ```sh node '/absolute/path/to/run-conformance/scripts/report.mjs' '/absolute/path/to/report.json' <<'CONFORMANCE_INPUT' @@ -33,11 +31,11 @@ node '/absolute/path/to/run-conformance/scripts/report.mjs' '/absolute/path/to/r CONFORMANCE_INPUT ``` -Replace the example paths. Write the report only to the destination the user requested. +Wait for successful initialization before recording observations. `start` creates missing parent directories, replaces any report at that path, and initializes every check as `not_verified`. Use different paths for concurrent runs and write only to the requested destination. -## Record each observation +## Record observations -Run recording commands sequentially. Never submit recording commands together in a parallel tool-call batch; wait for each command to finish before issuing the next. Record each observation before attempting another component. After obtaining an observation, immediately send this message through stdin to the same reporting script and report path: +Save observations during collection so you do not need to retain all evidence until the end. Send one observation per `record` message to the same reporter and path. Collection and recording may proceed in parallel. ```json { @@ -46,85 +44,172 @@ Run recording commands sequentially. Never submit recording commands together in } ``` -Replace `observation` with the complete observation just obtained. The script retains the other evidence, replaces any previous observation of the same kind for that skill or server, evaluates the accumulated evidence, and updates the JSON report. A brief acknowledgment confirms each successful recording. +Replace `observation` with the complete object obtained below. When possible, serialize the captured observation as JSON rather than retyping or reconstructing its fields. The reporter safely merges concurrent recordings and reevaluates the report. For the same kind and skill or server, the last committed recording replaces the previous observation. -Collect all fixture skills and MCP tools through the client's normal mechanisms. The user or CI may start the optional HTTP server before client loading; do not start it or ask the user to start it during collection. +Without a successful acknowledgment, the observation is not confirmed saved. Correct an input or command problem using the evidence already obtained and retry that `record`, or disclose the unconfirmed item at completion. Do not invoke `start` again; it would discard prior evidence. -1. Find and load `conformance-alpha` and `conformance-beta` from Agent Plugins Conformance — Core, `conformance-recovery-valid` from Agent Plugins Conformance — Recovery, and `conformance-invalid-mcp-valid` from Agent Plugins Conformance — Invalid MCP, through the client's normal skill mechanism. A client-provided skill catalog followed by reading its advertised resource is a valid mechanism. Record the observation each loaded body supplies, one at a time. The example above shows the alpha observation. Do not infer their markers or claim discovery from files located independently of the client or skill bodies received in a prompt. -2. Inspect the client's skill catalog for Core's `conformance-nested` reference. Record whether the client advertises it as a separate skill; names may be client-namespaced. Set `advertised` to `true` if it is advertised, or `false` if the catalog excludes it. If the catalog is unavailable or known to be incomplete, leave this observation missing unless it shows the nested skill. Finding the reference on disk does not establish advertisement, and you do not need to activate it. Record the observation through the same reporter: +## Collection inventory - ```json - { - "action":"record", - "observation":{"kind":"skill-discovery","skill":"conformance-nested","advertised":false} - } - ``` +The names below are targets, not proof that the client exposes them. Client-visible names may be namespaced. Loaded skill bodies and MCP tool descriptions supply the canonical names and server IDs saved in observations. -3. Inspect the client's MCP tool catalog for the `observe` tool of each invalid Recovery server with a probe (`recovery-cwd-invalid-form`, `recovery-cwd-escape`, `recovery-cwd-data-escape`, `recovery-cwd-symlink-escape`, `recovery-command-symlink-posix`, `recovery-command-symlink-windows`, `recovery-unknown-field`, `recovery-missing-type`, `recovery-env-plugin-root`, `recovery-env-plugin-data`, `recovery-http-type`, `recovery-http-non-loopback`, `recovery-http-relative-url`, `recovery-http-fragment`, `recovery-http-userinfo`, `recovery-http-duplicate-headers`, `recovery-http-header-name`, `recovery-http-header-value`, `recovery-sse-non-loopback`, `recovery-sse-relative-url`, `recovery-sse-fragment`, `recovery-sse-userinfo`, `recovery-sse-duplicate-headers`, `recovery-sse-header-name`, and `recovery-sse-header-value`). Each description identifies the fixture server ID; names may be client-namespaced. Set `advertised` to `true` if the tool is advertised, even in a partial catalog, and record that before calling it. Set `advertised` to `false` only after a known attempt to load Recovery and a usable complete client inventory that excludes the tool, including any deferred tools. For `recovery-cwd-symlink-escape`, `recovery-command-symlink-posix`, and `recovery-command-symlink-windows`, a completed startup failure still permits `advertised: false` when that complete inventory shows no tools for the server; installation may have removed the escaping symlink. For the other servers, leave the observation missing if startup or tool discovery fails. For every server, leave it missing if the load scope or inventory completeness is unknown, unless the tool is advertised. A failed call or a guessed tool name does not establish absence. No rejection diagnostic is required. Record each server separately through the same reporter, using its fixture server ID: +**Core — Skill body** - ```json - { - "action":"record", - "observation":{"kind":"mcp-discovery","server":"recovery-cwd-escape","advertised":false} - } - ``` +- `conformance-alpha` +- `conformance-beta` -4. Find the `observe` tools on the Core fixture servers (`default`, `relative`, `root`, `data`, `command-token-posix`, `command-token-windows`, `http`, `http-redirect`, `sse`, `sse-header-precedence`, `sse-redirect`, and `sse-endpoint-origin`) and the Recovery fixture servers (`recovery-valid`, `recovery-cwd-invalid-form`, `recovery-cwd-escape`, `recovery-cwd-data-escape`, `recovery-cwd-symlink-escape`, `recovery-command-symlink-posix`, `recovery-command-symlink-windows`, `recovery-unknown-field`, `recovery-missing-type`, `recovery-env-plugin-root`, `recovery-env-plugin-data`, `recovery-http-type`, `recovery-http-non-loopback`, `recovery-http-relative-url`, `recovery-http-fragment`, `recovery-http-userinfo`, `recovery-http-duplicate-headers`, `recovery-http-header-name`, `recovery-http-header-value`, `recovery-sse-non-loopback`, `recovery-sse-relative-url`, `recovery-sse-fragment`, `recovery-sse-userinfo`, `recovery-sse-duplicate-headers`, `recovery-sse-header-name`, and `recovery-sse-header-value`) through the client's normal MCP mechanism. Tool names may be client-namespaced. The remote MCP tool descriptions identify their fixture server IDs; retain that association when calling each tool so that an error can be recorded for the same server. Call each available tool with `{}`. First inspect whether the client returned a successful MCP result or a native discovery/call error. For a native remote MCP error, follow step 5 without parsing it as observation JSON. For a native stdio error, leave that server’s runtime observation missing; retain any discovery observation already recorded. Only the platform-appropriate command-token server is expected to run; a startup error for the other is not a collection limitation. For a successful result, take the observation from `structuredContent` (some clients show `structured_content`), or parse the JSON in the tool's text content. Immediately record that complete object; exclude the MCP result wrapper containing `content` or `structuredContent`. Preserve the returned paths and values exactly. The `http-redirect` and `sse-redirect` fixtures redirect requests to another local origin, and `sse-endpoint-origin` advertises a message endpoint on that origin. Do not authorize forwarding configured headers to these destinations. -5. If a completed remote MCP discovery or call attempt instead produces a native error, record it as error evidence: +**Recovery — Skill body** -```json -{ - "action":"record", - "observation":{ - "kind":"mcp-streamable-http", - "server":"http-redirect", - "evidence":{ - "type":"error", - "message":"", - "classification":null - } - } -} -``` +- `conformance-recovery-valid` + +**Invalid MCP — Skill body** + +- `conformance-invalid-mcp-valid` + +**Core nested reference — Skill discovery** + +- `conformance-nested` + +**Core — MCP stdio** + +- `default` +- `relative` +- `root` +- `data` +- `command-token-posix` +- `command-token-windows` + +**Core — MCP Streamable HTTP** + +- `http` +- `http-redirect` + +**Core — MCP legacy HTTP+SSE** + +- `sse` +- `sse-header-precedence` +- `sse-redirect` +- `sse-endpoint-origin` + +**Recovery valid — MCP stdio** + +- `recovery-valid` + +**Recovery invalid — MCP stdio** -Use `kind: "mcp-sse"` for `sse`, `sse-header-precedence`, `sse-redirect`, `sse-endpoint-origin`, and the `recovery-sse-*` fixtures. Use `kind: "mcp-streamable-http"` for `http`, `http-redirect`, and the `recovery-http-*` fixtures. Use the canonical fixture server ID from step 4 whose observation you attempted to collect. +- `recovery-cwd-invalid-form` +- `recovery-cwd-escape` +- `recovery-cwd-data-escape` +- `recovery-cwd-symlink-escape` +- `recovery-command-symlink-posix` +- `recovery-command-symlink-windows` +- `recovery-unknown-field` +- `recovery-missing-type` +- `recovery-env-plugin-root` +- `recovery-env-plugin-data` -Establish that identity from the discovery operation’s scope or the called tool; the error need not name the server. If one completed discovery operation covers multiple remote MCP fixtures and fails for each, record the same error separately for each affected fixture. Do not assign an error to a fixture outside that operation’s scope. Copy the client’s complete original error diagnostic into `message` without summarizing or rewriting it. Do not substitute an error raised while parsing, transforming, or recording that diagnostic. +**Recovery invalid — MCP Streamable HTTP** -Leave `classification` as `null` except in these two cases: +- `recovery-http-type` +- `recovery-http-non-loopback` +- `recovery-http-relative-url` +- `recovery-http-fragment` +- `recovery-http-userinfo` +- `recovery-http-duplicate-headers` +- `recovery-http-header-name` +- `recovery-http-header-value` -- For `http-redirect` or `sse-redirect`, use `"redirect-refused"` when the native error unambiguously establishes refusal to follow the redirect, such as rejecting its HTTP 307 response as an unexpected server response. Merely mentioning 307 is insufficient. -- For `sse-endpoint-origin`, use `"endpoint-refused"` when the native diagnostic unambiguously rejects the endpoint event because its origin differs from the configured connection origin. +**Recovery invalid — MCP legacy HTTP+SSE** -Unsupported SSE, generic transport or connection failures, timeouts, unknown-tool errors, and unrelated URL errors do not establish either refusal; leave their classification `null`. +- `recovery-sse-non-loopback` +- `recovery-sse-relative-url` +- `recovery-sse-fragment` +- `recovery-sse-userinfo` +- `recovery-sse-duplicate-headers` +- `recovery-sse-header-name` +- `recovery-sse-header-value` -6. An invalid Recovery remote MCP fixture excluded under step 3 needs only its discovery observation. For other completed remote MCP discovery or call attempts that produce neither an observation nor a native error, record `evidence: null` for each remote MCP fixture whose observation you attempted to collect: +The optional HTTP server may already have been started by the user or CI. Do not start it or ask the user to start it during collection. `http-redirect` and `sse-redirect` redirect to another local origin, and `sse-endpoint-origin` advertises a message endpoint there. Do not authorize forwarding configured headers to those destinations. + +## Skills + +Find and load the four skill bodies in the inventory through the client's normal skill mechanism, and record the complete observation supplied by each body. A client-provided catalog followed by reading its advertised resource is valid. If a body is unavailable, leave its observation missing and continue. + +For `conformance-nested`, inspect the client skill catalog without activating it. Its mention here or in another skill body is not separate advertisement. Record `advertised: true` if the catalog advertises it, or `advertised: false` if the catalog excludes it. If the catalog is unavailable or known to be incomplete, leave this observation missing unless it shows the nested skill. + +Save the catalog observation as `{"kind":"skill-discovery","skill":"conformance-nested","advertised":}`. + +## MCP + +Call every available registered conformance tool with `{}` and record its runtime outcome, including tools exposed from invalid entries. + +The client inventory is cumulative across ordinary pages, searches, and deferred results: once a registered tool appears, it remains advertised. A later result never erases that positive evidence. Treat absence as evidence only after the accumulated inventory is known complete for the relevant loaded scope. + +The `mcp-discovery` observation kind is only for Recovery invalid servers. If a registered tool belongs to a Recovery invalid server, record `{"kind":"mcp-discovery","server":"","advertised":true}` even when the inventory is still partial. Discovery and runtime outcomes are separate observations. + +Catalog inspection alone is not a remote transport attempt. A completed remote discovery outcome is also evidence when no tool appears. Record its native error or completed empty result for every remote fixture within that operation's known scope. + +### Absent Recovery invalid tools + +After accumulating the client inventory, apply these ordered rules to every Recovery invalid server still unseen: + +1. If the relevant Recovery load scope or inventory completeness is unknown, including unresolved pagination or deferred tools, leave discovery missing. +2. If a known-complete inventory for the loaded Recovery scope excludes the tool and no startup or discovery operation for that server failed, record `advertised: false`. This applies to all invalid servers, including the three symlink servers. +3. A completed startup failure still permits `advertised: false` only for `recovery-cwd-symlink-escape`, `recovery-command-symlink-posix`, or `recovery-command-symlink-windows`, and only when the complete loaded-scope inventory excludes that tool. Installation may have removed the escaping symlink; the reporter combines this assertion with independent fixture and control evidence. +4. Otherwise leave discovery missing. A startup or discovery failure for an ordinary invalid server does not prove configuration rejection. + +A failure affects only fixtures in that operation's scope; unaffected absent fixtures may still satisfy the complete-inventory rule. A failed call, guessed name, rejection diagnostic, or fixture inspection does not establish catalog absence. Negative discovery requires no rejection diagnostic. A negative discovery assertion is a catalog observation, not an empty runtime result; do not synthesize runtime `null` for an excluded remote fixture. + +### Runtime outcomes + +Apply these outcomes to an actual tool call or native transport discovery attempt, not to a catalog listing or search. Inspect the original client result before parsing. Preserve returned values and path literals exactly. A parsing or recording error is not a native client diagnostic. + +| Outcome | Observation | +| --- | --- | +| Successful MCP result | Record the complete object from `structuredContent` or `structured_content`, or parse it from the result's JSON text. Exclude the MCP wrapper. | +| Native stdio discovery, startup, or call error | Leave runtime observation missing; retain any discovery observation. | +| Native remote discovery or call error | Record the exact typed error below. | +| Completed remote attempt with no observation or native error | Record `evidence: null` for each fixture actually attempted. | +| Skipped, interrupted, or unattributable attempt | Leave runtime observation missing. | + +Only explicit `evidence: null` asserts a completed remote attempt without a payload or diagnostic. A missing runtime observation can also follow a completed stdio error, valid exclusion, unavailability, or interruption. + +Save a completed empty Streamable HTTP attempt as `{"kind":"mcp-streamable-http","server":"","evidence":null}`; use `kind: "mcp-sse"` for HTTP+SSE. + +Use `kind: "mcp-streamable-http"` for Streamable HTTP errors and `kind: "mcp-sse"` for legacy HTTP+SSE errors: ```json { - "action":"record", - "observation":{"kind":"mcp-streamable-http","server":"http","evidence":null} + "kind":"mcp-streamable-http", + "server":"http-redirect", + "evidence":{ + "type":"error", + "message":"", + "classification":null + } } ``` -Use `kind: "mcp-sse"` and the attempted SSE fixture server ID for an unsuccessful SSE attempt. +Identify the fixture from the completed discovery operation's scope or called tool; the error need not name it. If one operation fails for several remote fixtures, record the same complete original diagnostic separately for each affected fixture, never outside that scope. Do not summarize it or replace it with a later parsing, transformation, or reporter error. + +Keep `classification` null unless the native diagnostic unambiguously proves one of these meanings: -For unavailable skill bodies or stdio runtime observations, or any skipped or interrupted attempt, leave that evidence missing and continue collecting the remaining components. Preserve any discovery observations already recorded. +- For `http-redirect` or `sse-redirect`, use `"redirect-refused"` only when it establishes refusal to follow the redirect. Rejecting HTTP 307 as an unexpected response qualifies; merely mentioning 307 does not. +- For `sse-endpoint-origin`, use `"endpoint-refused"` only when it rejects the endpoint event because its origin differs from the configured connection origin. -For every Streamable HTTP recording, the reporter checks the local fixture and adds `serverHealthCheck` to the saved observation. Do not supply this field yourself. Keep track of collection limitations for your final response and leave outcome decisions to the reporter. +Unsupported SSE, generic transport or connection failures, timeouts, unknown-tool errors, and unrelated URL errors remain unclassified. For every Streamable HTTP recording, the reporter adds `serverHealthCheck`; do not supply it. -A recording command failure means its observation was not successfully saved. Address an input or command error using the actual evidence, or identify the unsaved item in your final response. Retry an individual recording with `record`; invoking `start` again discards previously collected evidence. +Only the platform-appropriate Core command-token fixture is expected to run: `command-token-posix` on POSIX and `command-token-windows` on Windows. -## Finish collection +## Complete the run -When collection ends, give the absolute JSON report path and note any collection or recording limitations. Expected fixture exclusions established under the discovery rules above are not collection limitations. Claim successful recording only for commands that succeeded. If initialization failed, do not claim a report was created for this run. The saved report represents the latest version of each recorded observation; its existence alone does not establish that collection finished. +Return the absolute report path and name every inventory item whose observation you could not collect or confirm saved. This includes unavailable Core transports even when their support is optional. Disclose any discovery assertions unsupported by catalog evidence. Expected invalid-fixture exclusions, the inapplicable command-token fixture, conformance findings, and successfully recorded remote error or null evidence are not gaps. -During normal completion, return only the path and limitations. Do not read the accumulated report into context or generate its summary unless the user asks. Users and CI can consume the JSON directly and choose which results to require. Preserve the evaluator's `pass`, `fail`, and `not_verified` statuses when discussing results. These results describe submitted evidence for covered cases, not client certification, and the evaluator trusts your account of how skills and tools were exposed. +Normal completion contains only the path and gaps, without listing successes. Do not read the accumulated report or generate a summary unless the user asks. The report contains the latest submitted evidence; its existence alone does not establish that the workflow completed. -If the user asks you to summarize or explain the saved results, invoke the summarizer once: +If the user requests a summary or explanation, invoke the summarizer once: ```text node /scripts/summarize.mjs ``` -Use its table and explanations to answer the user's request. +Use its table and explanations, preserving `pass`, `fail`, and `not_verified`. These statuses describe submitted evidence for covered cases, not client certification, and depend on the provenance of the observations recorded.