Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 14 additions & 5 deletions e2e/responses.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -239,12 +239,20 @@ describe("openai-responses adapter through runInference", () => {
expect(error.data.error.message).toContain("backend exploded");
});

test("response.incomplete surfaces its reason as a protocol_mismatch inference.error", async () => {
// A truncated stream is a terminal turn with the partial content the
// backend managed to emit, not a protocol violation: the deltas already
// delivered stay, the turn ends, and the billed usage is reported.
test("response.incomplete ends the turn with its partial text and usage", async () => {
harness = setupHarness({ adapters: registry });
const stream = harness.scenario.createStream();
harness.scenario.whenRequestMatches(() => true, stream);
stream.enqueueAll(
[
sse({
type: "response.output_text.delta",
item_id: "msg_1",
delta: "partial",
}),
sse({
type: "response.incomplete",
response: {
Expand All @@ -257,10 +265,11 @@ describe("openai-responses adapter through runInference", () => {
{ startAt: 1 },
);
const events = await collect(harness, [userTurn("hi")]);
expect(events.some((e) => e.type === "inference.done")).toBe(false);
const error = errorEvent(events);
expect(error.data.error.category).toBe("protocol_mismatch");
expect(error.data.error.message).toContain("max_output_tokens");
expect(events.some((e) => e.type === "inference.error")).toBe(false);
const done = doneEvent(events);
const texts = done.data.turn.content.filter((b) => b.type === "text");
expect(texts.map((b) => b.text)).toEqual(["partial"]);
expect(done.data.usage).toMatchObject({ input: 10, output: 100 });
});

test("a prior signed reasoning turn replays as a reasoning item ahead of its function_call", async () => {
Expand Down
35 changes: 20 additions & 15 deletions src/protocol/iterator.ts
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,10 @@ const FailedEvent = type({
"response?": { "error?": { "message?": "string" } },
});
const IncompleteEvent = type({
response: { "incomplete_details?": { "reason?": "string" } },
response: {
"incomplete_details?": { "reason?": "string" },
"usage?": ResponsesUsage,
},
});
const ErrorEvent = type({ "message?": "string" });

Expand Down Expand Up @@ -462,9 +465,11 @@ export function parseResponse(
const message = validated.response?.error?.message ?? "response failed";
throw new ProtocolMismatchError(`${provider}: ${message}`, parsed);
}
// Throws like parseJSONResponse does for status "incomplete". The
// harness drops the events of a batch that throws, so this event's usage
// cannot also be reported.
// A truncated turn is still a turn: the backend stopped (often on
// max_output_tokens) after emitting usable deltas, and those deltas were
// already delivered as events by earlier envelopes. Terminal handling
// lives in isResponsesStreamTerminal; the usage is what the backend
// bills for the partial turn, so it is reported like a completed one.
case "response.incomplete": {
const validated = IncompleteEvent(parsed);
if (validated instanceof type.errors) {
Expand All @@ -474,11 +479,14 @@ export function parseResponse(
parsed,
);
}
throw protocolMismatch(
provider,
`response status is "incomplete": ${validated.response.incomplete_details?.reason ?? "no reason given"}`,
parsed,
);
if (validated.response.usage !== undefined) {
events.push({
type: "inference.usage",
seq,
data: { usage: toInferenceUsage(validated.response.usage), source },
});
}
return events;
}
case "error": {
const validated = ErrorEvent(parsed);
Expand Down Expand Up @@ -614,12 +622,9 @@ export function parseJSONResponse(
parsed,
);
}
if (response.status === "incomplete") {
throw new ProtocolMismatchError(
`${provider} parseJSONResponse: response status is "incomplete": ${response.incomplete_details?.reason ?? "no reason given"}`,
parsed,
);
}
// A truncated body still carries the partial turn the backend managed
// to produce; throwing on it would discard usable content, so status
// "incomplete" decodes like a completed one.

const seq = 0;
const events: InferenceEvent[] = [];
Expand Down
48 changes: 42 additions & 6 deletions src/responses.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -357,6 +357,32 @@ describe("Responses parser — event schema validation", () => {
).toEqual([]);
});

// A truncated turn is still a turn: the backend stopped (often on
// max_output_tokens) after emitting usable deltas. Surfacing that as a
// protocol mismatch drops the partial content the stream already
// delivered, so the envelope reports its usage like response.completed
// does — terminal handling lives in isStreamTerminal.
test("response.incomplete reports usage instead of throwing", () => {
const adapter = createOpenAIResponsesAdapter(source, {});
adapter.buildRequest(turns, "model", {});
const delta = JSON.stringify({
type: "response.output_text.delta",
item_id: "item_1",
delta: "partial",
});
expect(adapter.parseResponse(delta)).toHaveLength(1);
const incomplete = JSON.stringify({
type: "response.incomplete",
response: {
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
usage: { input_tokens: 10, output_tokens: 100 },
},
});
const events = adapter.parseResponse(incomplete);
expect(events.map((e) => e.type)).toEqual(["inference.usage"]);
});

// A function_call_arguments.delta is only routable to the tool_call.start
// the harness already saw; one for an item_id that never arrived via
// output_item.added is an orphan fragment, not a fresh block to
Expand Down Expand Up @@ -391,17 +417,27 @@ describe("Responses parser — non-streaming failure states", () => {
);
});

test("throws on status:incomplete", () => {
// A truncated body still carries the partial turn the backend managed
// to produce; throwing on it would discard usable content, so it decodes
// like a completed one, usage included.
test("decodes partial output and usage on status:incomplete", () => {
const adapter = createOpenAIResponsesAdapter(source, {});
const incomplete = JSON.stringify({
output: [],
output: [
{
type: "message",
content: [{ type: "output_text", text: "partial" }],
},
],
status: "incomplete",
incomplete_details: { reason: "max_output_tokens" },
usage: null,
usage: { input_tokens: 10, output_tokens: 100 },
});
expect(() => adapter.parseJSONResponse(incomplete)).toThrow(
ProtocolMismatchError,
);
const events = adapter.parseJSONResponse(incomplete);
expect(
events.filter((e) => e.type === "inference.text.delta"),
).toHaveLength(1);
expect(events.some((e) => e.type === "inference.usage")).toBe(true);
});

test("throws on a response missing the required output field", () => {
Expand Down
Loading