From 6d1925fe9c184ee54b4a8f50c81c5352d846f25f Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Fri, 25 Sep 2026 16:55:12 -0700 Subject: [PATCH 1/2] test(reads): red tests for single-way path-offset resume --- src/plugins/read-file-guard-plugin.test.ts | 135 +++++++++++++++++++++ 1 file changed, 135 insertions(+) diff --git a/src/plugins/read-file-guard-plugin.test.ts b/src/plugins/read-file-guard-plugin.test.ts index 5a3e920fe..abcb561e6 100644 --- a/src/plugins/read-file-guard-plugin.test.ts +++ b/src/plugins/read-file-guard-plugin.test.ts @@ -656,3 +656,138 @@ describe("readFileGuardPlugin", () => { expect(String(replay.content)).not.toContain("Blob not found"); }); }); + +describe("CL-8980 single-way path+offset resume (RED)", () => { + const fallback = async (call: ToolCall): Promise => ({ + callId: call.id, + content: "FALLBACK", + }); + + function freshRunner(blobReader?: ReturnType) { + const plugin = readFileGuardPlugin( + dir, + blobReader !== undefined ? { blobReader } : {}, + ); + const middleware = defined(plugin.middleware)(fallback); + return (id: string, args: Record) => + middleware({ id, name: "read_file", arguments: args }, neverAbort()); + } + + function noticeOffset(content: string): number { + const match = /Use offset=(\d+) to continue/.exec(content); + expect(match).not.toBeNull(); + return Number((match as RegExpExecArray)[1]); + } + + test("a truncated read emits same-path + explicit offset and no handle", async () => { + await fixture( + "single-way.txt", + Array.from({ length: 10 }, (_, i) => `line-${i}`).join("\n"), + ); + const content = String( + (await freshRunner()("w1", { path: "single-way.txt", limit: 4 })).content, + ); + expect(content).toMatch(/Use offset=(\d+) to continue/); + expect(content).not.toMatch(/Use path="tool-output:\/\/\//); + expect(content).not.toContain("single-use"); + expect(content).not.toContain("continuation handle"); + }); + + test("following the notice verbatim yields the next window on a fresh instance; replay and re-read are identical", async () => { + const rows = Array.from({ length: 10 }, (_, i) => `row-${i}`); + await fixture("chain.txt", rows.join("\n")); + const first = await freshRunner()("c1", { path: "chain.txt", limit: 4 }); + expect(first.isError).toBeFalsy(); + const offset = noticeOffset(String(first.content)); + + // Session resume is a fresh plugin instance: no cursor map survives, so + // the verbatim same-path + offset follow must still yield the next window. + const run = freshRunner(); + const second = await run("c2", { path: "chain.txt", offset, limit: 4 }); + expect(second.isError).toBeFalsy(); + const secondContent = String(second.content); + expect(secondContent).toContain("row-4"); + expect(secondContent).not.toContain("row-3"); + expect(secondContent).toMatch(/Use offset=(\d+) to continue/); + + const replay = await run("c3", { path: "chain.txt", offset, limit: 4 }); + expect(replay.isError).toBeFalsy(); + expect(String(replay.content)).toBe(secondContent); + + const reread = await run("c4", { path: "chain.txt", offset: 0, limit: 4 }); + expect(reread.isError).toBeFalsy(); + expect(String(reread.content)).toBe(String(first.content)); + }); + + test("a dead file offset names the file and the valid range", async () => { + const absolutePath = await fixture( + "dead-offset.txt", + Array.from({ length: 10 }, (_, i) => `line-${i}`).join("\n"), + ); + const result = await freshRunner()("d1", { + path: "dead-offset.txt", + offset: 500, + limit: 4, + }); + expect(result.isError).toBe(true); + const content = String(result.content); + expect(content).toContain("beyond end of file"); + expect(content).toContain(absolutePath); + expect(content).toContain("10 lines"); + expect(content).not.toContain("Blob not found"); + }); + + test("a dead blob offset names the spill URI and the valid range", async () => { + const encoder = new TextEncoder(); + const body = Array.from({ length: 100 }, (_, i) => `brow-${i}`).join("\n"); + const blobReader = createBlobReader({ + async readBlob(key) { + if (key === "dead-blob") return encoder.encode(body); + throw new Error(`missing ${key}`); + }, + }); + const result = await freshRunner(blobReader)("d2", { + path: "tool-output:///dead-blob", + offset: 500, + limit: 4, + }); + expect(result.isError).toBe(true); + const content = String(result.content); + expect(content).toContain("beyond end of file"); + expect(content).toContain("tool-output:///dead-blob"); + expect(content).toContain("100 lines"); + }); + + test("a spilled blob pages forward on the same URI with rising offsets; replay is identical", async () => { + const encoder = new TextEncoder(); + const body = Array.from({ length: 100 }, (_, i) => `srow-${i}`).join("\n"); + const blobReader = createBlobReader({ + async readBlob(key) { + if (key === "chain-blob") return encoder.encode(body); + throw new Error(`missing ${key}`); + }, + }); + const run = freshRunner(blobReader); + const first = await run("e1", { + path: "tool-output:///chain-blob", + offset: 0, + limit: 10, + }); + expect(first.isError).toBeFalsy(); + const offset = noticeOffset(String(first.content)); + const second = await run("e2", { + path: "tool-output:///chain-blob", + offset, + limit: 10, + }); + expect(second.isError).toBeFalsy(); + expect(String(second.content)).toContain("srow-10"); + expect(String(second.content)).not.toContain("srow-9"); + const replay = await run("e3", { + path: "tool-output:///chain-blob", + offset, + limit: 10, + }); + expect(String(replay.content)).toBe(String(second.content)); + }); +}); From fe0f23c7274a783593362687b7673a438872f37d Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Fri, 25 Sep 2026 16:58:28 -0700 Subject: [PATCH 2/2] feat(reads): resume truncated reads by same path plus offset --- src/plugins/read-file-guard-plugin.test.ts | 121 ++++++------- src/plugins/read-file-guard-plugin.ts | 191 ++------------------- src/session/compaction-verify.test.ts | 119 +++++++++++++ src/session/runtime-assembly.ts | 2 +- src/subagent/thrash.test.ts | 15 ++ 5 files changed, 201 insertions(+), 247 deletions(-) diff --git a/src/plugins/read-file-guard-plugin.test.ts b/src/plugins/read-file-guard-plugin.test.ts index abcb561e6..a780bb79f 100644 --- a/src/plugins/read-file-guard-plugin.test.ts +++ b/src/plugins/read-file-guard-plugin.test.ts @@ -329,8 +329,8 @@ describe("readFileGuardPlugin", () => { expect(result.content).toContain(" 2\tl2"); expect(result.content).toContain(" 3\tl3"); expect(result.content).not.toContain(" 4\tl4"); - expect(result.content).toContain('Use path="tool-output:///'); - expect(result.content).not.toContain("Use offset="); + expect(result.content).toContain("Use offset="); + expect(result.content).not.toContain('Use path="tool-output:///'); }); test("rejects tool-output URIs when no blob reader is configured", async () => { @@ -368,7 +368,7 @@ describe("readFileGuardPlugin", () => { expect(result.content).not.toBe("FALLBACK"); }); - test("pages a giant one-line tool-output blob across byte windows and resumes via the minted cursor", async () => { + test("pages a giant one-line tool-output blob across byte windows on the same URI with rising offsets", async () => { const encoder = new TextEncoder(); const payload = `HEAD-${"x".repeat(READ_FILE_MAX_BYTES)}-TAIL`; const blobReader = createBlobReader({ @@ -396,13 +396,19 @@ describe("readFileGuardPlugin", () => { expect(Buffer.byteLength(firstContent, "utf8")).toBeLessThanOrEqual( READ_FILE_MAX_BYTES, ); - const match = /Use path="(tool-output:\/\/\/[^"]+)"/.exec(firstContent); + const match = /Use offset=(\d+) to continue/.exec(firstContent); expect(match).not.toBeNull(); - const nextPath = (match as RegExpExecArray)[1] as string; - expect(nextPath).toMatch(/^tool-output:\/\/\//); + const nextOffset = Number((match as RegExpExecArray)[1]); const second = await middleware( - { id: "g2", name: "read_file", arguments: { path: nextPath } }, + { + id: "g2", + name: "read_file", + arguments: { + path: "tool-output:///giant-line", + offset: nextOffset, + }, + }, neverAbort(), ); expect(second.isError).toBeFalsy(); @@ -490,7 +496,7 @@ describe("readFileGuardPlugin", () => { expect(result.content).toBe("FALLBACK"); }); - test("a truncated read never asks the model to re-read the same path (CL-6961)", async () => { + test("a truncated read names the same path with an explicit offset (CL-8980)", async () => { await fixture( "many-lines.txt", Array.from({ length: 10 }, (_, i) => `line-${i}`).join("\n"), @@ -505,19 +511,17 @@ describe("readFileGuardPlugin", () => { }, neverAbort(), ); - expect(result.content).not.toContain("Use offset="); - expect(String(result.content)).toContain('Use path="tool-output:///'); - // The literal source path never reappears as the thing to read next. - expect(String(result.content)).not.toContain("many-lines.txt"); + expect(String(result.content)).toMatch(/Use offset=(\d+) to continue/); + expect(String(result.content)).not.toContain('Use path="tool-output:///'); + expect(String(result.content)).not.toContain("single-use"); }); - test("following the minted cursor resumes and eventually reads a large file to completion without any repeat call on the original path (CL-6961)", async () => { + test("following same-path offsets reads a large file to completion; every hop re-issues the original path with a rising offset (CL-8980)", async () => { const lines = Array.from({ length: 9_000 }, (_, i) => `line-${i} payload`); await fixture("huge.txt", lines.join("\n")); const plugin = readFileGuardPlugin(dir, {}); const middleware = defined(plugin.middleware)(fallback); - const pathsRead: string[] = ["huge.txt"]; let result = await middleware( { id: "c1", name: "read_file", arguments: { path: "huge.txt" } }, neverAbort(), @@ -526,35 +530,32 @@ describe("readFileGuardPlugin", () => { let guard = 0; for (;;) { guard++; - expect(guard).toBeLessThan(50); // fails loudly instead of hanging on a broken cursor chain + expect(guard).toBeLessThan(50); // fails loudly instead of hanging on a broken offset chain const content = String(result.content); const numbered = content.split("\n\n")[0] ?? ""; seen += numbered.trimEnd().split("\n").length; - const match = /Use path="(tool-output:\/\/\/[^"]+)"/.exec(content); + const match = /Use offset=(\d+) to continue/.exec(content); if (match === undefined || match === null) break; - const nextPath = match[1] as string; - expect(pathsRead).not.toContain(nextPath); // every hop targets a fresh, distinct path - pathsRead.push(nextPath); + const offset = Number(match[1] as string); result = await middleware( { - id: `c${pathsRead.length}`, + id: `c${guard + 1}`, name: "read_file", - arguments: { path: nextPath }, + arguments: { path: "huge.txt", offset }, }, neverAbort(), ); + expect(result.isError).toBeFalsy(); } expect(seen).toBe(lines.length); - expect(pathsRead.length).toBeGreaterThan(1); // it actually paginated - // Never told to re-issue a call against the literal original path. - expect(pathsRead.filter((p) => p === "huge.txt").length).toBe(1); + expect(guard).toBeGreaterThan(1); // it actually paginated }); - test("a stale (already-consumed) cursor names the original path and offset instead of a dead end", async () => { - const absolutePath = await fixture( + test("reusing a continuation offset after first use still yields the window — reads never expire", async () => { + await fixture( "stale.txt", Array.from({ length: 10 }, (_, i) => `line-${i}`).join("\n"), ); @@ -568,31 +569,29 @@ describe("readFileGuardPlugin", () => { }, neverAbort(), ); - const match = /Use path="(tool-output:\/\/\/[^"]+)"/.exec( - String(first.content), - ); + const match = /Use offset=(\d+) to continue/.exec(String(first.content)); expect(match).not.toBeNull(); - const cursorPath = (match as RegExpExecArray)[1] as string; + const offset = Number((match as RegExpExecArray)[1] as string); - await middleware( - { id: "s2", name: "read_file", arguments: { path: cursorPath } }, + const second = await middleware( + { id: "s2", name: "read_file", arguments: { path: "stale.txt", offset } }, neverAbort(), ); - // Second use of the same, already-consumed cursor: distinct from a - // generic missing-blob error, this must name a followable next step — - // the original source and the offset to resume from — rather than - // leaving the model to re-read the whole file from scratch. + expect(second.isError).toBeFalsy(); + expect(String(second.content)).toContain("line-4"); + // Second use of the same offset: reads are idempotent, so the replay is + // byte-identical instead of a spent-handle error. const replay = await middleware( - { id: "s3", name: "read_file", arguments: { path: cursorPath } }, + { id: "s3", name: "read_file", arguments: { path: "stale.txt", offset } }, neverAbort(), ); - expect(replay.isError).toBe(true); - expect(String(replay.content)).toContain("already used"); - expect(String(replay.content)).toContain(absolutePath); - expect(String(replay.content)).toMatch(/offset=4\b/); + expect(replay.isError).toBeFalsy(); + expect(String(replay.content)).toBe(String(second.content)); + expect(String(replay.content)).not.toContain("already used"); + expect(String(replay.content)).not.toContain("single-use"); }); - test("an unknown tool-output URI against a real blobReader gets the production 'blob not found' error, not a stale-cursor message", async () => { + test("an unknown tool-output URI against a real blobReader surfaces the blob store error", async () => { const blobReader = { async read(uri: string): Promise { throw new Error(`Blob not found for key: ${uri}`); @@ -608,16 +607,14 @@ describe("readFileGuardPlugin", () => { ); expect(result.isError).toBe(true); expect(String(result.content)).toContain("Blob not found for key"); - // Never a cursor's own wording, since this ID was never one of ours. + // No handle machinery remains: there is no spent/cursor wording anywhere. expect(String(result.content)).not.toContain("already used"); + expect(String(result.content)).not.toContain("single-use"); }); - test("a stale cursor short-circuits before reaching a real blobReader's production 'blob not found' error", async () => { - const encoder = new TextEncoder(); - const body = Array.from({ length: 8_000 }, (_, i) => `row-${i}`).join("\n"); + test("a replayed unknown tool-output URI surfaces the same blob error twice — no spent-handle state", async () => { const blobReader = { async read(uri: string): Promise { - if (uri === "tool-output:///spill-1") return encoder.encode(body); throw new Error(`Blob not found for key: ${uri}`); }, }; @@ -628,36 +625,26 @@ describe("readFileGuardPlugin", () => { { id: "b1", name: "read_file", - arguments: { path: "tool-output:///spill-1", limit: 5 }, + arguments: { path: "tool-output:///gone", limit: 5 }, }, neverAbort(), ); - const match = /Use path="(tool-output:\/\/\/[^"]+)"/.exec( - String(first.content), - ); - expect(match).not.toBeNull(); - const cursorPath = (match as RegExpExecArray)[1] as string; - - await middleware( - { id: "b2", name: "read_file", arguments: { path: cursorPath } }, - neverAbort(), - ); - // Replaying the consumed cursor must not fall through to blobReader.read() - // (which would throw the opaque "Blob not found" error naming only the - // random cursor UUID) -- it must short-circuit to the actionable message - // naming the real spill URI and the offset to resume from. + expect(first.isError).toBe(true); + expect(String(first.content)).toContain("Blob not found for key"); const replay = await middleware( - { id: "b3", name: "read_file", arguments: { path: cursorPath } }, + { + id: "b2", + name: "read_file", + arguments: { path: "tool-output:///gone", limit: 5 }, + }, neverAbort(), ); expect(replay.isError).toBe(true); - expect(String(replay.content)).toContain("already used"); - expect(String(replay.content)).toContain("tool-output:///spill-1"); - expect(String(replay.content)).not.toContain("Blob not found"); + expect(String(replay.content)).toBe(String(first.content)); }); }); -describe("CL-8980 single-way path+offset resume (RED)", () => { +describe("CL-8980 single-way path+offset resume", () => { const fallback = async (call: ToolCall): Promise => ({ callId: call.id, content: "FALLBACK", diff --git a/src/plugins/read-file-guard-plugin.ts b/src/plugins/read-file-guard-plugin.ts index e5f556d83..1f27dfe7a 100644 --- a/src/plugins/read-file-guard-plugin.ts +++ b/src/plugins/read-file-guard-plugin.ts @@ -1,4 +1,3 @@ -import { randomUUID } from "node:crypto"; import { createReadStream } from "node:fs"; import { stat } from "node:fs/promises"; import { resolve } from "node:path"; @@ -9,7 +8,6 @@ import type { BlobReader } from "@intx/types/runtime"; import { canonicalToolOutputUri, isToolOutputLike, - TOOL_OUTPUT_URI_PREFIX, } from "../util/tool-output-uri.js"; import { formatReadFileTimeoutMessage } from "./tool-time-budget.js"; @@ -46,92 +44,14 @@ export interface ReadFileGuardPluginOptions { blobReader?: BlobReader; } -// A truncated read used to tell the model "Use offset=N to continue" against -// the identical path -- exactly the same-path pagination fan-out CL-6961 -// measured (97% of 4+-reads-per-path clusters were legitimate chunked reads -// of one large file, penalized by detectors that only see "same path, many -// calls"). Each truncated result instead mints a single-use tool-output:// -// cursor pointing at the exact resumption point (source + next offset) and -// tells the model to pass THAT as `path`. Every follow-up read therefore -// targets a distinct path, so pagination no longer looks like a same-path -// loop, and the cursor is a real, resolvable handle -- not the "see the blob" -// promise result-truncation-plugin.ts's comment forbids, since nothing here -// claims discarded bytes are retrievable; it just remembers where to resume -// a fresh bounded read. -type ReadCursor = - | { kind: "file"; absolutePath: string; offset: number; consumed: boolean } - | { kind: "blob"; uri: string; offset: number; consumed: boolean }; - -// A cursor is single-use, but the record survives consumption (bounded by -// MAX_CURSOR_HISTORY below) so a stale replay -- consumed already, or a -// second process/turn racing the first -- can be told exactly where to -// resume instead of hitting an opaque "blob not found" dead end that names -// neither the file nor an offset and leaves re-reading from scratch (the -// original path, no offset) as the model's only move. -const MAX_CURSOR_HISTORY = 200; - -const CONTINUE_OFFSET_RE = /Use offset=(\d+) to continue\.\]$/; - -function pruneCursorHistory(cursors: Map): void { - while (cursors.size > MAX_CURSOR_HISTORY) { - const oldest = cursors.keys().next().value; - if (oldest === undefined) break; - cursors.delete(oldest); - } -} - -function mintCursor( - content: string, - cursors: Map, - source: - | { kind: "file"; absolutePath: string } - | { kind: "blob"; uri: string }, -): string { - const match = CONTINUE_OFFSET_RE.exec(content); - if (match === null) return content; - const offset = Number(match[1]); - const cursorId = randomUUID(); - cursors.set( - cursorId, - source.kind === "file" - ? { - kind: "file", - absolutePath: source.absolutePath, - offset, - consumed: false, - } - : { kind: "blob", uri: source.uri, offset, consumed: false }, - ); - pruneCursorHistory(cursors); - return content.replace( - CONTINUE_OFFSET_RE, - `Use path="${TOOL_OUTPUT_URI_PREFIX}///${cursorId}" (same tool, no offset needed) to continue reading the remainder — a fresh, working handle, not the original path.]`, - ); -} - -// Bound the source shown in a stale-cursor message: an adversarial or -// pathological path must not blow past a reasonable notice size. -const STALE_CURSOR_SOURCE_MAX = 300; - -function displaySource(source: string): string { - return source.length > STALE_CURSOR_SOURCE_MAX - ? `${source.slice(0, STALE_CURSOR_SOURCE_MAX)}…` - : source; -} - -/** - * Message for a cursor that is known but already used (or is being replayed - * from a stale/compacted turn). Distinct from "blob not found": it names the - * original source and the exact offset to resume from, so recovery is a - * single new call rather than a re-read from scratch of the whole file. - */ -function staleCursorMessage(cursor: ReadCursor): string { - const source = cursor.kind === "file" ? cursor.absolutePath : cursor.uri; - return ( - `this read_file continuation handle was already used (each cursor is single-use). ` + - `Resume with read_file, path="${displaySource(source)}", offset=${cursor.offset}.` - ); -} +// A truncated read tells the model to continue with the same path and the +// explicit next offset from the notice ("Use offset=N to continue"). There is +// no continuation handle: every read is a stateless, idempotent ranged read, +// so following a notice verbatim works on first use, on replay, and on a fresh +// plugin instance after compaction or session resume — and re-reading any +// earlier window behaves identically. Chunked same-path reads carry rising +// offsets, so detectors that key on the full call (including arguments) see +// one ranged read per window, not a same-path loop. function numArg(value: unknown): number | undefined { return typeof value === "number" && Number.isFinite(value) @@ -160,7 +80,7 @@ function mapFilesystemStreamError( * skipping `offset` lines (zero-based). Never splits the full decoded text in one pass. * When `wrapLongLines` is set, overlong lines are split into successive numbered * windows instead of being truncated and dropped — so a giant JSON line can be - * paged through with the same offset/cursor protocol as a multi-line file. + * paged through with the same offset protocol as a multi-line file. */ function readStreamBounded( stream: Readable, @@ -289,7 +209,7 @@ function readStreamBounded( } if (endReached) { done({ - content: `[offset ${offset} is beyond end of file (${lineNo} lines)]`, + content: `[offset ${offset} is beyond end of file ${displayPath} (${lineNo} lines); valid offsets 0-${lineNo - 1}]`, isError: true, }); } else { @@ -456,11 +376,6 @@ export function readFileGuardPlugin( options: ReadFileGuardPluginOptions = {}, ): ToolPlugin { const { blobReader } = options; - // Single-use resumption pointers minted by mintCursor(); scoped to this - // plugin instance (one per session/agent, per buildCorePosixToolPlugins), so - // it never outlives the session and never crosses sessions. - const cursors = new Map(); - const cursorUriPrefix = `${TOOL_OUTPUT_URI_PREFIX}///`; return { middleware: (next) => async (call, signal) => { if (call.name !== "read_file") return next(call, signal); @@ -482,76 +397,6 @@ export function readFileGuardPlugin( return next(call, signal); } - const cursorId = uri.startsWith(cursorUriPrefix) - ? uri.slice(cursorUriPrefix.length) - : ""; - const cursor = cursorId.length > 0 ? cursors.get(cursorId) : undefined; - if (cursor !== undefined && cursor.consumed) { - // Known cursor, already used -- distinct from a genuine missing - // blob: name the original source and offset so recovery is one - // targeted call, not a from-scratch re-read of the whole file. - return { - callId: call.id, - content: staleCursorMessage(cursor), - isError: true, - }; - } - if (cursor !== undefined) { - // A cursor is authoritative on position: the model passes only the - // handle (and optionally a limit), never an offset back into it. - cursor.consumed = true; - try { - signal.throwIfAborted(); - if (cursor.kind === "file") { - const res = await readFileBounded( - cursor.absolutePath, - cursor.offset, - limit, - signal, - ); - return res.isError - ? { callId: call.id, content: res.content, isError: true } - : { - callId: call.id, - content: mintCursor(res.content, cursors, { - kind: "file", - absolutePath: cursor.absolutePath, - }), - }; - } - if (blobReader === undefined) { - return { - callId: call.id, - content: `cannot read ${rawPath}: no blob reader is configured for tool-output spills`, - isError: true, - }; - } - const bytes = await blobReader.read(cursor.uri); - const res = await readBytesBounded( - bytes, - cursor.offset, - blobLimit, - signal, - cursor.uri, - ); - return res.isError - ? { callId: call.id, content: res.content, isError: true } - : { - callId: call.id, - content: mintCursor(res.content, cursors, { - kind: "blob", - uri: cursor.uri, - }), - }; - } catch (err) { - return { - callId: call.id, - content: err instanceof Error ? err.message : String(err), - isError: true, - }; - } - } - if (blobReader === undefined) { return { callId: call.id, @@ -571,13 +416,7 @@ export function readFileGuardPlugin( ); return res.isError ? { callId: call.id, content: res.content, isError: true } - : { - callId: call.id, - content: mintCursor(res.content, cursors, { - kind: "blob", - uri, - }), - }; + : { callId: call.id, content: res.content }; } catch (err) { return { callId: call.id, @@ -600,13 +439,7 @@ export function readFileGuardPlugin( const res = await readFileBounded(absolutePath, offset, limit, signal); return res.isError ? { callId: call.id, content: res.content, isError: true } - : { - callId: call.id, - content: mintCursor(res.content, cursors, { - kind: "file", - absolutePath, - }), - }; + : { callId: call.id, content: res.content }; } catch (err) { return { callId: call.id, diff --git a/src/session/compaction-verify.test.ts b/src/session/compaction-verify.test.ts index 454f9a5f2..b0784a373 100644 --- a/src/session/compaction-verify.test.ts +++ b/src/session/compaction-verify.test.ts @@ -821,3 +821,122 @@ describe("condenseTurns keep-set", () => { expect(condensed).toContain("Goal (first user message)"); }); }); + +describe("CL-8980 compaction preserves the path+offset resume recipe", () => { + function readCallTurn(id: string, offset: number): ConversationTurn { + return { + role: "assistant", + content: [ + { + type: "tool_call", + id, + name: "read_file", + arguments: { path: "var/log/big.log", offset, limit: 2 }, + }, + ], + timestamp: Date.now(), + }; + } + + function readResultTurn( + callId: string, + body: string, + notice: string, + ): ConversationTurn { + return { + role: "user", + content: [ + { + type: "tool_result", + callId, + content: [{ type: "text", text: `${body}\n\n${notice}` }], + }, + ], + timestamp: Date.now(), + }; + } + + const NOTICE_OFF_2 = + "[Showing lines 1-2; stopped at the 2-line limit. Use offset=2 to continue.]"; + const NOTICE_OFF_4 = + "[Showing lines 3-4; stopped at the 2-line limit. Use offset=4 to continue.]"; + + // Deliberately omits notice text: the kept result bodies — not the summary + // — are what must carry the resume recipe. + const summarize = async () => + "Reading var/log/big.log in windows. Next: keep reading."; + + function allResultText(turns: ConversationTurn[]): string { + return turns + .flatMap((t) => + t.content.flatMap((b) => { + if (b.type === "text") return [b.text]; + if (b.type === "tool_result") + return b.content + .filter((c) => c.type === "text") + .map((c) => c.text); + return []; + }), + ) + .join("\n"); + } + + test("distinct windows keep their bodies and notices; nothing hollows across windows", async () => { + const compactor = createPruningCompactor({ + keepRecentTurns: 8, + summaryMaxChars: 4000, + summarize, + }); + const turns: ConversationTurn[] = [ + textTurn("user", "Read var/log/big.log in full"), + textTurn("assistant", "Reading the log in full."), + readCallTurn("c1", 0), + readResultTurn("c1", "w1-row-a\nw1-row-b", NOTICE_OFF_2), + readCallTurn("c2", 2), + readResultTurn("c2", "w2-row-a\nw2-row-b", NOTICE_OFF_4), + readCallTurn("c3", 4), + readResultTurn("c3", "w3-row-a\nw3-row-b", "end of file"), + textTurn("user", "recent ask"), + textTurn("assistant", "recent reply"), + ]; + const result = await compactor.apply(turns, mockStrategyCtx); + expect(result.record.reason).toMatch(/compacted/); + const text = allResultText(result.output); + expect(text).toContain("w2-row-a"); + expect(text).toContain("w3-row-a"); + expect(text).toContain("Use offset=2 to continue"); + expect(text).toContain("Use offset=4 to continue"); + expect(text).not.toContain("omitted from context"); + expect(text).not.toContain("continuation handle"); + expect(text).not.toContain("already used"); + }); + + test("a verbatim replay stubs the older duplicate and keeps the newest whole with its notice", async () => { + const compactor = createPruningCompactor({ + keepRecentTurns: 6, + summaryMaxChars: 4000, + summarize, + }); + const turns: ConversationTurn[] = [ + textTurn("user", "Read var/log/big.log in full"), + textTurn("assistant", "Reading the log in full."), + readCallTurn("c1", 0), + readResultTurn("c1", "w1-row-a\nw1-row-b", NOTICE_OFF_2), + readCallTurn("c2", 2), + readResultTurn("c2", "w2-row-a\nw2-row-b", NOTICE_OFF_4), + readCallTurn("c3", 2), + readResultTurn("c3", "w2-row-a\nw2-row-b", NOTICE_OFF_4), + textTurn("user", "recent ask"), + textTurn("assistant", "recent reply"), + ]; + const result = await compactor.apply(turns, mockStrategyCtx); + expect(result.record.reason).toMatch(/compacted/); + expect(result.record.decisions).toMatchObject({ supersededReadCount: 1 }); + const text = allResultText(result.output); + expect(text).toContain("Use offset=4 to continue"); + expect(text).toContain("w2-row-a"); + expect(text).toContain("omitted from context"); + expect(text).not.toContain("continuation handle"); + expect(text).not.toContain("already used"); + }); +}); diff --git a/src/session/runtime-assembly.ts b/src/session/runtime-assembly.ts index 21ff6c61d..04cdddded 100644 --- a/src/session/runtime-assembly.ts +++ b/src/session/runtime-assembly.ts @@ -588,7 +588,7 @@ export function buildShellBackgroundMessage(exit: { } if (exit.spillUri !== undefined) { lines.push( - `Full output was spilled to ${exit.spillUri} (readable via read_file).`, + `Full output was spilled to ${exit.spillUri} — use read_file with that URI (offset/limit supported) to see the rest.`, ); } return { diff --git a/src/subagent/thrash.test.ts b/src/subagent/thrash.test.ts index f400b2046..d6eb97c24 100644 --- a/src/subagent/thrash.test.ts +++ b/src/subagent/thrash.test.ts @@ -119,6 +119,21 @@ describe("thrash pure module", () => { expect(state.readCounts.get("big.ts::500:500")).toBe(1); }); + test("CL-8980: a same-path offset chain is one ranged read per window, not a same-path loop", () => { + const chain = Array.from({ length: 12 }, (_, i) => + read("src/big.ts", { offset: i * 50, limit: 50 }), + ); + const state = applyAll(chain); + // Rising offsets key every window distinctly: no single key accumulates + // the chain, so an offset chain never reads as "same path, many calls". + expect(state.readCounts.size).toBe(12); + for (let i = 0; i < 12; i++) { + expect(state.readCounts.get(`src/big.ts::${i * 50}:50`)).toBe(1); + } + // Salvage still collapses the chain to the one path for the parent report. + expect(salvagePathsFromThrash(state, 40)).toEqual(["src/big.ts"]); + }); + test("greps count toward read evidence keyed by pattern and path", () => { const state = applyAll([grep("needle"), grep("needle")]); expect(state.readCounts.get("grep::needle::src")).toBe(2);