diff --git a/apps/memos-local-plugin/core/llm/providers/anthropic.ts b/apps/memos-local-plugin/core/llm/providers/anthropic.ts
index 6c510ef75..03b8c6523 100644
--- a/apps/memos-local-plugin/core/llm/providers/anthropic.ts
+++ b/apps/memos-local-plugin/core/llm/providers/anthropic.ts
@@ -7,6 +7,7 @@
import { ERROR_CODES, MemosError } from "../../../agent-contract/errors.js";
import { decodeSse, httpPostJson, httpPostStream } from "../fetcher.js";
+import { sanitizeCompletionText } from "../sanitize.js";
import type {
LlmMessage,
LlmProvider,
@@ -74,10 +75,12 @@ export class AnthropicLlmProvider implements LlmProvider {
log,
});
- const text = (json.content ?? [])
- .filter((b) => b.type === "text" && typeof b.text === "string")
- .map((b) => b.text ?? "")
- .join("");
+ const text = sanitizeCompletionText(
+ (json.content ?? [])
+ .filter((b) => b.type === "text" && typeof b.text === "string")
+ .map((b) => b.text ?? "")
+ .join(""),
+ );
return {
text,
finishReason: mapFinish(json.stop_reason),
diff --git a/apps/memos-local-plugin/core/llm/providers/bedrock.ts b/apps/memos-local-plugin/core/llm/providers/bedrock.ts
index d956b7111..657836333 100644
--- a/apps/memos-local-plugin/core/llm/providers/bedrock.ts
+++ b/apps/memos-local-plugin/core/llm/providers/bedrock.ts
@@ -13,6 +13,7 @@
import { ERROR_CODES, MemosError } from "../../../agent-contract/errors.js";
import { httpPostJson } from "../fetcher.js";
+import { sanitizeCompletionText } from "../sanitize.js";
import type {
LlmMessage,
LlmProvider,
@@ -92,7 +93,7 @@ export class BedrockLlmProvider implements LlmProvider {
});
const blocks = json.output?.message?.content ?? [];
- const text = blocks.map((b) => b.text ?? "").join("");
+ const text = sanitizeCompletionText(blocks.map((b) => b.text ?? "").join(""));
return {
text,
finishReason: mapFinish(json.stopReason),
diff --git a/apps/memos-local-plugin/core/llm/providers/gemini.ts b/apps/memos-local-plugin/core/llm/providers/gemini.ts
index 6bfa323eb..129cfdb3f 100644
--- a/apps/memos-local-plugin/core/llm/providers/gemini.ts
+++ b/apps/memos-local-plugin/core/llm/providers/gemini.ts
@@ -7,6 +7,7 @@
import { ERROR_CODES, MemosError } from "../../../agent-contract/errors.js";
import { decodeSse, httpPostJson, httpPostStream } from "../fetcher.js";
+import { sanitizeCompletionText } from "../sanitize.js";
import type {
LlmMessage,
LlmProvider,
@@ -66,7 +67,9 @@ export class GeminiLlmProvider implements LlmProvider {
});
const cand = json.candidates?.[0];
- const text = cand?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
+ const text = sanitizeCompletionText(
+ cand?.content?.parts?.map((p) => p.text ?? "").join("") ?? "",
+ );
return {
text,
finishReason: mapFinish(cand?.finishReason),
@@ -109,7 +112,20 @@ export class GeminiLlmProvider implements LlmProvider {
log,
});
+ // Buffer the SSE deltas until the stream terminates, then sanitize the
+ // full accumulated text before yielding it (issue #2336 follow-up).
+ //
+ // Rationale: `...` blocks and DeepSeek-style special
+ // tokens can be split across chunk boundaries, so a naive per-chunk
+ // `sanitizeCompletionText` would miss any artifact that straddles two
+ // SSE events. Callers accumulate `chunk.delta` (see
+ // `core/llm/client.ts::stream`), so we can safely emit one sanitized
+ // delta at the tail: `chunks.map(c => c.delta).join("")` still yields
+ // the sanitized full text, matching the contract exercised by the
+ // provider tests.
let done = false;
+ let accumulated = "";
+ let pendingUsage: GemResp["usageMetadata"] | undefined;
for await (const payload of decodeSse(resp.body!)) {
let evt: GemResp;
try {
@@ -120,25 +136,43 @@ export class GeminiLlmProvider implements LlmProvider {
const cand = evt.candidates?.[0];
const delta = cand?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
const finish = cand?.finishReason;
- if (delta.length > 0) yield { delta, done: false };
+ if (delta.length > 0) accumulated += delta;
+ if (evt.usageMetadata) pendingUsage = evt.usageMetadata;
if (finish) {
done = true;
+ const sanitized = sanitizeCompletionText(accumulated);
+ if (sanitized.length > 0) yield { delta: sanitized, done: false };
+ const usage = evt.usageMetadata ?? pendingUsage;
yield {
delta: "",
done: true,
finishReason: mapFinish(finish),
- usage: evt.usageMetadata
+ usage: usage
? {
- promptTokens: evt.usageMetadata.promptTokenCount,
- completionTokens: evt.usageMetadata.candidatesTokenCount,
- totalTokens: evt.usageMetadata.totalTokenCount,
+ promptTokens: usage.promptTokenCount,
+ completionTokens: usage.candidatesTokenCount,
+ totalTokens: usage.totalTokenCount,
}
: undefined,
};
return;
}
}
- if (!done) yield { delta: "", done: true };
+ if (!done) {
+ const sanitized = sanitizeCompletionText(accumulated);
+ if (sanitized.length > 0) yield { delta: sanitized, done: false };
+ yield {
+ delta: "",
+ done: true,
+ usage: pendingUsage
+ ? {
+ promptTokens: pendingUsage.promptTokenCount,
+ completionTokens: pendingUsage.candidatesTokenCount,
+ totalTokens: pendingUsage.totalTokenCount,
+ }
+ : undefined,
+ };
+ }
}
}
diff --git a/apps/memos-local-plugin/core/llm/providers/openai.ts b/apps/memos-local-plugin/core/llm/providers/openai.ts
index 562521756..d29554ce4 100644
--- a/apps/memos-local-plugin/core/llm/providers/openai.ts
+++ b/apps/memos-local-plugin/core/llm/providers/openai.ts
@@ -8,6 +8,7 @@
import { ERROR_CODES, MemosError } from "../../../agent-contract/errors.js";
import { applyOpenRouterProviderRouting } from "../../openrouter.js";
import { decodeSse, httpPostJson, httpPostStream } from "../fetcher.js";
+import { sanitizeCompletionText } from "../sanitize.js";
import type {
LlmMessage,
LlmProvider,
@@ -100,7 +101,7 @@ export class OpenAiLlmProvider implements LlmProvider {
});
const choice = json.choices?.[0];
- const text = choice?.message?.content ?? "";
+ const text = sanitizeCompletionText(choice?.message?.content ?? "");
return {
text,
finishReason: mapFinish(choice?.finish_reason),
@@ -166,12 +167,37 @@ export class OpenAiLlmProvider implements LlmProvider {
log,
});
+ // Buffer the SSE deltas until the stream terminates, then sanitize the
+ // full accumulated text before yielding it (issue #2336 follow-up).
+ //
+ // Rationale: `...` blocks and DeepSeek special tokens
+ // (`<|end▁of▁sentence|>`, etc.) can be split across chunk boundaries,
+ // so a naive per-chunk `sanitizeCompletionText` would miss any artifact
+ // that straddles two SSE events. Callers accumulate `chunk.delta` (see
+ // `core/llm/client.ts::stream`), so we can safely emit one sanitized
+ // delta at the tail: `chunks.map(c => c.delta).join("")` still yields
+ // the sanitized full text, matching the contract exercised by the
+ // provider tests.
let emittedDone = false;
+ let accumulated = "";
+ let pendingUsage: OaResp["usage"] | undefined;
for await (const payload of decodeSse(resp.body!)) {
if (payload === "[DONE]") {
if (!emittedDone) {
emittedDone = true;
- yield { delta: "", done: true };
+ const sanitized = sanitizeCompletionText(accumulated);
+ if (sanitized.length > 0) yield { delta: sanitized, done: false };
+ yield {
+ delta: "",
+ done: true,
+ usage: pendingUsage
+ ? {
+ promptTokens: pendingUsage.prompt_tokens,
+ completionTokens: pendingUsage.completion_tokens,
+ totalTokens: pendingUsage.total_tokens,
+ }
+ : undefined,
+ };
}
return;
}
@@ -186,26 +212,44 @@ export class OpenAiLlmProvider implements LlmProvider {
const delta = choice?.delta?.content ?? "";
const finish = choice?.finish_reason;
if (delta.length > 0) {
- yield { delta, done: false };
+ accumulated += delta;
}
+ if (parsed.usage) pendingUsage = parsed.usage;
if (finish) {
emittedDone = true;
+ const sanitized = sanitizeCompletionText(accumulated);
+ if (sanitized.length > 0) yield { delta: sanitized, done: false };
+ const usage = parsed.usage ?? pendingUsage;
yield {
delta: "",
done: true,
finishReason: mapFinish(finish),
- usage: parsed.usage
+ usage: usage
? {
- promptTokens: parsed.usage.prompt_tokens,
- completionTokens: parsed.usage.completion_tokens,
- totalTokens: parsed.usage.total_tokens,
+ promptTokens: usage.prompt_tokens,
+ completionTokens: usage.completion_tokens,
+ totalTokens: usage.total_tokens,
}
: undefined,
};
return;
}
}
- if (!emittedDone) yield { delta: "", done: true };
+ if (!emittedDone) {
+ const sanitized = sanitizeCompletionText(accumulated);
+ if (sanitized.length > 0) yield { delta: sanitized, done: false };
+ yield {
+ delta: "",
+ done: true,
+ usage: pendingUsage
+ ? {
+ promptTokens: pendingUsage.prompt_tokens,
+ completionTokens: pendingUsage.completion_tokens,
+ totalTokens: pendingUsage.total_tokens,
+ }
+ : undefined,
+ };
+ }
}
}
diff --git a/apps/memos-local-plugin/core/llm/sanitize.ts b/apps/memos-local-plugin/core/llm/sanitize.ts
new file mode 100644
index 000000000..b4eb252c3
--- /dev/null
+++ b/apps/memos-local-plugin/core/llm/sanitize.ts
@@ -0,0 +1,73 @@
+/**
+ * Strips gateway artifacts from LLM completion text before it is returned
+ * from a provider — see issue #2336.
+ *
+ * Thinking-enabled DeepSeek-family models served through gateways that keep
+ * the model's `...` reasoning block inside
+ * `choice.message.content` (rather than surfacing it via a separate
+ * `reasoning_content` field) leak that block into `ProviderCompletion.text`.
+ * Ops that persist this text — most notably `feedback.classify`, which
+ * writes into `FeedbackRow.rationale` — then pollute retrieval-visible
+ * memory with strings like `"\n<|end▁of▁sentence|>\n..."`, and
+ * that pollution flows back into the model's own context on future turns.
+ *
+ * This module centralizes the sanitizer so every provider strips the same
+ * set of artifacts. Fixing it here is O(1) call sites for O(N) ops.
+ */
+
+/**
+ * Matched `...` block. Non-greedy so multiple blocks are
+ * stripped independently; multi-line so newlines inside the block are
+ * consumed. Case-insensitive because some gateways upcase tag names.
+ */
+const THINK_BLOCK_RE = /<\s*think\b[^>]*>[\s\S]*?<\s*\/\s*think\s*>/gi;
+
+/**
+ * Orphan closing `` fragment. The reporter's evidence in #2336
+ * starts with one — a gateway that truncates thinking-block output can
+ * drop the opener while leaving the closer intact.
+ */
+const ORPHAN_CLOSE_THINK_RE = /<\s*\/\s*think\s*>/gi;
+
+/**
+ * Orphan opening `` fragment plus everything after it. Symmetric
+ * defense against the reverse failure mode: gateway keeps the opener but
+ * cuts before the closer, leaving reasoning trailing off the response.
+ * We drop everything from the opener onward because the model has already
+ * committed to reasoning-mode output — retaining it would still be
+ * unactionable thinking text.
+ */
+const ORPHAN_OPEN_THINK_RE = /<\s*think\b[^>]*>[\s\S]*$/i;
+
+/**
+ * DeepSeek-family special tokens. Uses the Chinese full-width `|`
+ * (U+FF5C) that the model actually emits — NOT the ASCII `|`. Some
+ * gateways strip these before sending, others don't; strip them here
+ * unconditionally so the invariant is provider-agnostic.
+ */
+const DEEPSEEK_SPECIAL_TOKENS_RE =
+ /<|(?:begin|end)▁(?:of▁sentence|of▁session)|>/g;
+
+/**
+ * Collapse runs of ≥3 blank lines that the strip may have introduced.
+ * Two consecutive newlines (a paragraph break) are still allowed.
+ */
+const EXCESS_BLANK_LINES_RE = /\n{3,}/g;
+
+/**
+ * Remove `` blocks and DeepSeek gateway tokens from LLM completion
+ * text. Idempotent and safe on empty input.
+ *
+ * Order matters: strip matched blocks first, THEN orphan fragments — the
+ * orphan patterns are broad and would eat surrounding content if a
+ * well-formed block hadn't been removed first.
+ */
+export function sanitizeCompletionText(text: string): string {
+ if (!text) return text;
+ let out = text.replace(THINK_BLOCK_RE, "");
+ out = out.replace(ORPHAN_OPEN_THINK_RE, "");
+ out = out.replace(ORPHAN_CLOSE_THINK_RE, "");
+ out = out.replace(DEEPSEEK_SPECIAL_TOKENS_RE, "");
+ out = out.replace(EXCESS_BLANK_LINES_RE, "\n\n");
+ return out.trim();
+}
diff --git a/apps/memos-local-plugin/tests/unit/llm/providers.test.ts b/apps/memos-local-plugin/tests/unit/llm/providers.test.ts
index 49e14658b..ccc0dc1ca 100644
--- a/apps/memos-local-plugin/tests/unit/llm/providers.test.ts
+++ b/apps/memos-local-plugin/tests/unit/llm/providers.test.ts
@@ -14,6 +14,7 @@ import type {
LlmProviderCtx,
LlmProviderLogger,
LlmMessage,
+ LlmStreamChunk,
ProviderCallInput,
} from "../../../core/llm/types.js";
@@ -101,6 +102,50 @@ describe("llm/providers", () => {
await expect(p.complete(msgs, call(), ctxFor(cfg({ apiKey: "" })))).rejects.toBeInstanceOf(MemosError);
});
+ // Regression pin for issue #2336: thinking-enabled DeepSeek-family models
+ // on gateways that keep the reasoning block inside `choice.message.content`
+ // used to leak `...` and DeepSeek session tokens into
+ // `ProviderCompletion.text`, which then ended up persisted in
+ // `FeedbackRow.rationale`. The provider must sanitize before returning.
+ it("strips blocks and DeepSeek gateway tokens from returned text (issue #2336)", async () => {
+ captureFetch({
+ choices: [
+ {
+ message: {
+ content:
+ "let me reason about polarity\n<|end▁of▁sentence|>\n" +
+ '{"polarity":"negative","rationale":"user says wrong"}',
+ },
+ finish_reason: "stop",
+ },
+ ],
+ });
+ const p = new OpenAiLlmProvider();
+ const res = await p.complete(msgs, call(), ctxFor(cfg()));
+ expect(res.text).toBe(
+ '{"polarity":"negative","rationale":"user says wrong"}',
+ );
+ expect(res.text).not.toContain("");
+ expect(res.text).not.toContain("");
+ expect(res.text).not.toContain("<|end▁of▁sentence|>");
+ });
+
+ it("strips an orphan fragment as observed in the #2336 evidence", async () => {
+ captureFetch({
+ choices: [
+ {
+ message: {
+ content:
+ "\n<|end▁of▁sentence|>\n<|end▁of▁session|>\n\n---\n\n[Writing Rule] final.",
+ },
+ },
+ ],
+ });
+ const p = new OpenAiLlmProvider();
+ const res = await p.complete(msgs, call(), ctxFor(cfg()));
+ expect(res.text).toBe("---\n\n[Writing Rule] final.");
+ });
+
it("forwards config.reasoning into an OpenRouter request body", async () => {
const cap = captureFetch({ choices: [{ message: { content: "{}" } }] });
const p = new OpenAiLlmProvider();
@@ -294,6 +339,92 @@ describe("llm/providers", () => {
});
expect(body.reasoning).toEqual({ enabled: true, max_tokens: 4_000 });
});
+
+ // Regression pin for issue #2336 follow-up: the streaming path used to
+ // pass raw `` and DeepSeek gateway tokens through per-chunk,
+ // leaking them to any consumer that accumulates `chunk.delta`. The
+ // sanitizer only ran on the non-streaming `complete()` path. This test
+ // splits a `...` block across two SSE events (proving
+ // per-chunk sanitization would miss it) plus DeepSeek session tokens,
+ // and asserts the joined delta stream matches the sanitizer output.
+ it("sanitizes blocks that span SSE chunk boundaries during streaming", async () => {
+ const sseBody = [
+ // Opener + first half of the reasoning block.
+ `data: ${JSON.stringify({
+ choices: [{ delta: { content: "let me " } }],
+ })}\n\n`,
+ // Closing tag + first special token in a second event — a naive
+ // per-chunk sanitizer would miss the block because the opener and
+ // closer arrive in different SSE frames.
+ `data: ${JSON.stringify({
+ choices: [{ delta: { content: "reason\n<|end▁of▁sentence|>\n" } }],
+ })}\n\n`,
+ // The actual payload the caller cares about.
+ `data: ${JSON.stringify({
+ choices: [{ delta: { content: '{"polarity":"negative"}' } }],
+ })}\n\n`,
+ // Finish frame.
+ `data: ${JSON.stringify({
+ choices: [{ delta: {}, finish_reason: "stop" }],
+ usage: { prompt_tokens: 5, completion_tokens: 12, total_tokens: 17 },
+ })}\n\n`,
+ `data: [DONE]\n\n`,
+ ].join("");
+ vi.stubGlobal(
+ "fetch",
+ vi.fn(async () =>
+ new Response(sseBody, {
+ status: 200,
+ headers: { "content-type": "text/event-stream" },
+ }),
+ ),
+ );
+ const p = new OpenAiLlmProvider();
+ const chunks: LlmStreamChunk[] = [];
+ for await (const c of p.stream(msgs, call(), ctxFor(cfg()))) chunks.push(c);
+ const joined = chunks.map((c) => c.delta).join("");
+ expect(joined).toBe('{"polarity":"negative"}');
+ expect(joined).not.toContain("");
+ expect(joined).not.toContain("");
+ expect(joined).not.toContain("<|end▁of▁sentence|>");
+ // The done chunk still carries finish reason + usage.
+ const last = chunks[chunks.length - 1];
+ expect(last?.done).toBe(true);
+ expect(last?.finishReason).toBe("stop");
+ expect(last?.usage?.totalTokens).toBe(17);
+ });
+
+ it("sanitizes an orphan streamed as the first SSE chunk", async () => {
+ const sseBody = [
+ `data: ${JSON.stringify({
+ choices: [
+ {
+ delta: {
+ content:
+ "\n<|end▁of▁sentence|>\n<|end▁of▁session|>\n\n---\n\n[Writing Rule] final.",
+ },
+ },
+ ],
+ })}\n\n`,
+ `data: ${JSON.stringify({
+ choices: [{ delta: {}, finish_reason: "stop" }],
+ })}\n\n`,
+ `data: [DONE]\n\n`,
+ ].join("");
+ vi.stubGlobal(
+ "fetch",
+ vi.fn(async () =>
+ new Response(sseBody, {
+ status: 200,
+ headers: { "content-type": "text/event-stream" },
+ }),
+ ),
+ );
+ const p = new OpenAiLlmProvider();
+ const chunks: LlmStreamChunk[] = [];
+ for await (const c of p.stream(msgs, call(), ctxFor(cfg()))) chunks.push(c);
+ expect(chunks.map((c) => c.delta).join("")).toBe("---\n\n[Writing Rule] final.");
+ });
});
// ─── anthropic ─────────────────────────────────────────────────────────────
@@ -363,6 +494,52 @@ describe("llm/providers", () => {
// The fetch mock only keeps the last call — but since we fired one,
// the value is that one.
});
+
+ // Regression pin for issue #2336 follow-up: same defect as the OpenAI
+ // streaming path — per-chunk deltas leaked `` blocks and DeepSeek
+ // gateway tokens because the sanitizer only ran on `complete()`.
+ it("sanitizes blocks that span SSE chunk boundaries during streaming", async () => {
+ const sseBody = [
+ `data: ${JSON.stringify({
+ candidates: [{ content: { parts: [{ text: "reason " }] } }],
+ })}\n\n`,
+ `data: ${JSON.stringify({
+ candidates: [
+ { content: { parts: [{ text: "goes here\n<|end▁of▁sentence|>\n" }] } },
+ ],
+ })}\n\n`,
+ `data: ${JSON.stringify({
+ candidates: [{ content: { parts: [{ text: '{"polarity":"positive"}' }] } }],
+ })}\n\n`,
+ `data: ${JSON.stringify({
+ candidates: [{ finishReason: "STOP" }],
+ usageMetadata: { promptTokenCount: 3, candidatesTokenCount: 5, totalTokenCount: 8 },
+ })}\n\n`,
+ ].join("");
+ vi.stubGlobal(
+ "fetch",
+ vi.fn(async () =>
+ new Response(sseBody, {
+ status: 200,
+ headers: { "content-type": "text/event-stream" },
+ }),
+ ),
+ );
+ const p = new GeminiLlmProvider();
+ const chunks: LlmStreamChunk[] = [];
+ for await (const c of p.stream(msgs, call(), ctxFor(cfg({ provider: "gemini" })))) {
+ chunks.push(c);
+ }
+ const joined = chunks.map((c) => c.delta).join("");
+ expect(joined).toBe('{"polarity":"positive"}');
+ expect(joined).not.toContain("");
+ expect(joined).not.toContain("");
+ expect(joined).not.toContain("<|end▁of▁sentence|>");
+ const last = chunks[chunks.length - 1];
+ expect(last?.done).toBe(true);
+ expect(last?.finishReason).toBe("stop");
+ expect(last?.usage?.totalTokens).toBe(8);
+ });
});
// ─── bedrock ───────────────────────────────────────────────────────────────
diff --git a/apps/memos-local-plugin/tests/unit/llm/sanitize.test.ts b/apps/memos-local-plugin/tests/unit/llm/sanitize.test.ts
new file mode 100644
index 000000000..9e02f8cef
--- /dev/null
+++ b/apps/memos-local-plugin/tests/unit/llm/sanitize.test.ts
@@ -0,0 +1,148 @@
+/**
+ * Regression tests for the LLM completion sanitizer (issue #2336).
+ *
+ * The bug: the openai_compatible provider was returning
+ * `choice.message.content` verbatim, so thinking-enabled DeepSeek-family
+ * models on gateways that keep the `` block inside `content`
+ * (rather than moving it to `reasoning_content`) leaked that block —
+ * plus DeepSeek gateway tokens like `<|end▁of▁sentence|>` — into
+ * persisted feedback rationales and other memory artifacts.
+ *
+ * The fix is a single provider-side transform; these tests pin the
+ * exact shape of that transform.
+ */
+
+import { describe, expect, it } from "vitest";
+
+import { sanitizeCompletionText } from "../../../core/llm/sanitize.js";
+
+describe("llm/sanitize", () => {
+ describe("passthrough", () => {
+ it("returns empty input untouched", () => {
+ expect(sanitizeCompletionText("")).toBe("");
+ });
+
+ it("returns clean text untouched apart from trim", () => {
+ expect(sanitizeCompletionText("hello world")).toBe("hello world");
+ expect(sanitizeCompletionText(" hello world ")).toBe("hello world");
+ });
+
+ it("does not touch JSON payloads that contain no think tags", () => {
+ const json = '{"polarity":"negative","rationale":"user says wrong"}';
+ expect(sanitizeCompletionText(json)).toBe(json);
+ });
+ });
+
+ describe("matched blocks", () => {
+ it("removes a leading ... block", () => {
+ const raw = "let me reason\nfinal answer";
+ expect(sanitizeCompletionText(raw)).toBe("final answer");
+ });
+
+ it("removes a multi-line block including newlines", () => {
+ const raw = [
+ "",
+ "step 1: read the request",
+ "step 2: decide feedback polarity",
+ "",
+ "",
+ '{"polarity":"negative"}',
+ ].join("\n");
+ expect(sanitizeCompletionText(raw)).toBe('{"polarity":"negative"}');
+ });
+
+ it("removes multiple blocks independently (non-greedy)", () => {
+ const raw = "akeep1bkeep2";
+ expect(sanitizeCompletionText(raw)).toBe("keep1keep2");
+ });
+
+ it("handles upper-case tag names", () => {
+ const raw = "reason\nkeep";
+ expect(sanitizeCompletionText(raw)).toBe("keep");
+ });
+
+ it("handles whitespace inside tag delimiters", () => {
+ const raw = "< think >r think >final";
+ expect(sanitizeCompletionText(raw)).toBe("final");
+ });
+ });
+
+ describe("orphan closing ", () => {
+ it("strips the orphan closing tag exactly as reported in #2336", () => {
+ // This is the leading fragment the reporter observed in a stored
+ // feedback rationale row. It is a gateway artifact — the opening
+ // `` got truncated before reaching us; only the closer +
+ // DeepSeek session tokens survive.
+ const raw =
+ "\n<|end▁of▁sentence|>\n<|end▁of▁session|>\n\n---\n\n[Writing Rule] first sentence.";
+ expect(sanitizeCompletionText(raw)).toBe(
+ "---\n\n[Writing Rule] first sentence.",
+ );
+ });
+
+ it("strips an orphan closer without opener but keeps the actual answer", () => {
+ const raw = "\nthe answer is 42";
+ expect(sanitizeCompletionText(raw)).toBe("the answer is 42");
+ });
+ });
+
+ describe("orphan opening ", () => {
+ it("drops the opener and everything after it (truncated reasoning)", () => {
+ // Reverse failure: gateway kept `` open but cut the response
+ // before the closer. Anything after the opener is trailing
+ // reasoning text — unactionable and not the final answer.
+ const raw = "final answer here\nstill thinking about";
+ expect(sanitizeCompletionText(raw)).toBe("final answer here");
+ });
+ });
+
+ describe("DeepSeek special tokens", () => {
+ it("strips <|begin▁of▁sentence|> using Chinese full-width bars", () => {
+ const raw = "<|begin▁of▁sentence|>hello";
+ expect(sanitizeCompletionText(raw)).toBe("hello");
+ });
+
+ it("strips <|end▁of▁sentence|> and <|end▁of▁session|>", () => {
+ const raw = "the answer<|end▁of▁sentence|><|end▁of▁session|>";
+ expect(sanitizeCompletionText(raw)).toBe("the answer");
+ });
+
+ it("does NOT strip an ASCII-pipe lookalike (safety guard)", () => {
+ // If a user's own message legitimately contained "<|end|>" (ASCII
+ // pipes) it should survive — only the full-width DeepSeek token
+ // is a gateway artifact.
+ const raw = "user wrote <|end|> in their prompt";
+ expect(sanitizeCompletionText(raw)).toBe(
+ "user wrote <|end|> in their prompt",
+ );
+ });
+ });
+
+ describe("interactions", () => {
+ it("collapses ≥3 blank lines the strip introduced back down to a paragraph break", () => {
+ const raw = "before\n\n\n\nx\n\n\n\nafter";
+ expect(sanitizeCompletionText(raw)).toBe("before\n\n\n\nafter".replace(/\n{3,}/g, "\n\n"));
+ });
+
+ it("handles the full reported artifact soup end-to-end", () => {
+ const raw = [
+ "",
+ "user is unhappy — flag negative",
+ "",
+ "<|end▁of▁sentence|>",
+ '{"polarity":"negative","rationale":"user says wrong"}',
+ ].join("\n");
+ expect(sanitizeCompletionText(raw)).toBe(
+ '{"polarity":"negative","rationale":"user says wrong"}',
+ );
+ });
+
+ it("is idempotent", () => {
+ const raw =
+ "rkeep<|end▁of▁sentence|>tail";
+ const once = sanitizeCompletionText(raw);
+ const twice = sanitizeCompletionText(once);
+ expect(twice).toBe(once);
+ });
+ });
+});