From 0c7c3cf950db537c18c5d809316d1fca166884fd Mon Sep 17 00:00:00 2001 From: Gorkem Date: Tue, 4 Aug 2026 18:44:28 +0300 Subject: [PATCH 1/6] fold: manual echo guard family onto current tips (from fix/manual-echo-guard, PR 963 delta) --- dist/index.js | 10 +- dist/src/manual-echo-guard.js | 113 +++++++++++++++++++++ dist/src/smart-extractor.js | 18 +++- dist/src/tools.js | 3 + index.ts | 10 ++ package.json | 2 +- scripts/ci-test-manifest.mjs | 1 + src/manual-echo-guard.ts | 116 +++++++++++++++++++++ src/smart-extractor.ts | 23 ++++- src/tools.ts | 10 ++ test/manual-echo-guard.test.mjs | 173 ++++++++++++++++++++++++++++++++ 11 files changed, 475 insertions(+), 4 deletions(-) create mode 100644 dist/src/manual-echo-guard.js create mode 100644 src/manual-echo-guard.ts create mode 100644 test/manual-echo-guard.test.mjs diff --git a/dist/index.js b/dist/index.js index bf421e55..6c3e47d1 100644 --- a/dist/index.js +++ b/dist/index.js @@ -26,6 +26,7 @@ import { createRetriever, normalizeRetrievalConfig, } from "./src/retriever.js"; import { createScopeManager, resolveScopeFilter, isSystemBypassId, parseAgentIdFromSessionKey } from "./src/scopes.js"; import { createMigrator } from "./src/migrate.js"; import { registerAllMemoryTools } from "./src/tools.js"; +import { ManualEchoLedger } from "./src/manual-echo-guard.js"; import { appendSelfImprovementEntry, ensureSelfImprovementLearningFiles } from "./src/self-improvement-files.js"; import { shouldSkipRetrieval } from "./src/adaptive-retrieval.js"; import { parseClawteamScopes, applyClawteamScopes } from "./src/clawteam-scope.js"; @@ -1989,6 +1990,10 @@ function _initPluginState(api) { // enabled. admissionControl.enabled remains a supported configuration on // its own. let smartExtractor = null; + // Echo guard: shared between the manual store/update tools (record side) + // and the smart extractor (drop side); lives here so the tools keep + // recording even when smart extraction is disabled. + const manualEchoLedger = new ManualEchoLedger(); let admissionController = null; let admissionControllerReflectionLane = null; if (config.smartExtraction !== false || config.admissionControl?.enabled === true) { @@ -2056,6 +2061,7 @@ function _initPluginState(api) { noiseBank.init(embedder).catch((err) => api.logger.debug(`memory-lancedb-pro: noise bank init: ${String(err)}`)); smartExtractor = new SmartExtractor(store, embedder, llmClient, { user: "User", + manualEchoLedger, captureAssistantEligible: config.captureAssistant === true, extractMinMessages: config.extractMinMessages ?? 4, extractMaxChars: config.extractMaxChars ?? 8000, @@ -2126,6 +2132,7 @@ function _initPluginState(api) { scopeManager, migrator, smartExtractor, + manualEchoLedger, mdMirror, extractionRateLimiter, reflectionErrorStateBySession, @@ -2254,7 +2261,7 @@ const memoryLanceDBProPlugin = { _registeredApisMap.delete(api); // dual-track rollback: Map un-claim throw err; } - const { config, resolvedDbPath, vectorDim, store, embedder, retriever, canonicalCorpusIndexer, dreamingEngine, dreamingScheduler, scopeManager, migrator, smartExtractor, mdMirror, decayEngine, tierManager, extractionRateLimiter, reflectionErrorStateBySession, reflectionDerivedBySession, reflectionDerivedSuppressionBySession, reflectionByAgentCache, reflectionByAgentCacheGeneration, recallHistory, turnCounter, autoCaptureSeenTextCount, autoCapturePendingIngressTexts, autoCaptureCountedPendingCount, autoCaptureRecentTurns, autoCaptureDeferredFlushTurns, autoCaptureSessionIdToKey, autoCaptureInFlightRuns, captureAdmissionController, captureAdmissionAudit, captureReflectionAdmissionController, admissionRejectionAuditWriter, } = singleton; + const { config, resolvedDbPath, vectorDim, store, embedder, retriever, canonicalCorpusIndexer, dreamingEngine, dreamingScheduler, scopeManager, migrator, smartExtractor, manualEchoLedger, mdMirror, decayEngine, tierManager, extractionRateLimiter, reflectionErrorStateBySession, reflectionDerivedBySession, reflectionDerivedSuppressionBySession, reflectionByAgentCache, reflectionByAgentCacheGeneration, recallHistory, turnCounter, autoCaptureSeenTextCount, autoCapturePendingIngressTexts, autoCaptureCountedPendingCount, autoCaptureRecentTurns, autoCaptureDeferredFlushTurns, autoCaptureSessionIdToKey, autoCaptureInFlightRuns, captureAdmissionController, captureAdmissionAudit, captureReflectionAdmissionController, admissionRejectionAuditWriter, } = singleton; const learnAutoCaptureSessionAlias = (sessionId, sessionKey) => { if (typeof sessionId !== "string" || !sessionId || typeof sessionKey !== "string" || !sessionKey @@ -2614,6 +2621,7 @@ const memoryLanceDBProPlugin = { workspaceBoundary: config.workspaceBoundary, selfImprovementMaxEntries: config.selfImprovement?.maxEntries, manualStoreSupersede: config.manualStoreSupersede === true, + manualEchoLedger, // Mirrors the CLI context wiring below: keep in-process reflection caches // consistent after a live memory_forget delete too, not just CLI delete/delete-bulk. onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js new file mode 100644 index 00000000..8606f231 --- /dev/null +++ b/dist/src/manual-echo-guard.js @@ -0,0 +1,113 @@ +/** + * Manual-store echo guard (class): when the user dictates a memory, + * the same sentence reaches BOTH the manual store lane (memory_store / + * memory_update, always-priority, verbatim) and auto-capture extraction, + * which mints near-twin candidates the dedup layer cannot reliably collide — + * the manual row may be seconds old (fresh-row vector visibility) or land in + * a different category. The guard remembers recent manual texts per agent + * and drops near-identical extraction candidates BEFORE the admission judge: + * deterministic, string-only, no LLM calls, no vector search. + * + * Scoped per agent (not per session): the store tool and the auto-capture + * hook derive their session keys differently, but both resolve the same + * agent id, and an echo of ANY recent manual text of the same agent is a + * correct drop regardless of session boundaries. The ring bounds staleness. + */ +export const MANUAL_ECHO_JACCARD_THRESHOLD = 0.75; +export const MANUAL_ECHO_SUBSET_THRESHOLD = 0.9; +export const MANUAL_ECHO_RING_SIZE = 8; +const MAX_TRACKED_AGENTS = 128; +const MIN_CONTAINMENT_TOKENS = 3; +const DEFAULT_AGENT_BUCKET = "main"; +export function normalizeEchoText(text) { + return text + .toLowerCase() + .replace(/[^\p{L}\p{N}\s]+/gu, " ") + .replace(/\s+/g, " ") + .trim(); +} +function tokenSet(normalized) { + return new Set(normalized.split(" ").filter((t) => t.length > 0)); +} +function jaccard(a, b) { + if (a.size === 0 || b.size === 0) + return 0; + let intersection = 0; + for (const token of a) { + if (b.has(token)) + intersection++; + } + return intersection / (a.size + b.size - intersection); +} +function subsetRatio(inner, outer) { + if (inner.size === 0) + return 0; + let contained = 0; + for (const token of inner) { + if (outer.has(token)) + contained++; + } + return contained / inner.size; +} +export function isNearIdenticalEcho(candidateText, manualText) { + const candidate = normalizeEchoText(candidateText); + const manual = normalizeEchoText(manualText); + if (candidate.length === 0 || manual.length === 0) + return false; + if (candidate === manual) + return true; + const manualTokens = tokenSet(manual); + // Very short manual texts over-match as substrings ("blue mug" is inside + // any sentence mentioning it); those only count as echoes when exact. + if (manualTokens.size < MIN_CONTAINMENT_TOKENS) + return false; + if (candidate.includes(manual) || manual.includes(candidate)) + return true; + // Token-subset containment: the canonical echo shape is the extractor + // sentence-wrapping the manual fact ("favorite teacup: the red one" -> + // "User's favorite teacup is the red one"), where glue words break + // character containment and dilute Jaccard. One side's tokens (nearly) + // all present in the other = echo. + const candidateTokens = tokenSet(candidate); + if (subsetRatio(manualTokens, candidateTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD || + (candidateTokens.size >= MIN_CONTAINMENT_TOKENS && + subsetRatio(candidateTokens, manualTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD)) { + return true; + } + return jaccard(candidateTokens, manualTokens) >= MANUAL_ECHO_JACCARD_THRESHOLD; +} +export class ManualEchoLedger { + byAgent = new Map(); + record(agentId, text) { + if (typeof text !== "string" || text.trim().length === 0) + return; + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key) ?? []; + ring.push(text); + while (ring.length > MANUAL_ECHO_RING_SIZE) + ring.shift(); + this.byAgent.delete(key); + this.byAgent.set(key, ring); + while (this.byAgent.size > MAX_TRACKED_AGENTS) { + const oldest = this.byAgent.keys().next().value; + if (oldest === undefined) + break; + this.byAgent.delete(oldest); + } + } + /** Returns the matched manual text, or null when the candidate is no echo. */ + match(agentId, candidateText) { + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key); + if (!ring || ring.length === 0) + return null; + for (let i = ring.length - 1; i >= 0; i--) { + if (isNearIdenticalEcho(candidateText, ring[i])) + return ring[i]; + } + return null; + } + clear(agentId) { + this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); + } +} diff --git a/dist/src/smart-extractor.js b/dist/src/smart-extractor.js index b442bbb3..548be421 100644 --- a/dist/src/smart-extractor.js +++ b/dist/src/smart-extractor.js @@ -455,7 +455,23 @@ export class SmartExtractor { } // Step 1: LLM extraction const extraction = await this.extractCandidates(conversationText, policyMode, options.conversationTurns, options.protectedPrefixTurns); - const candidates = extraction.candidates; + let candidates = extraction.candidates; + // Echo guard: candidates near-identical to a recent manual + // memory_store/memory_update text are echoes of a row that already + // exists verbatim -- drop them before any judge/dedup/merge spend. + const echoLedger = this.config.manualEchoLedger; + if (echoLedger && candidates.length > 0) { + const kept = []; + for (const candidate of candidates) { + if (echoLedger.match(agentId, candidate.content)) { + this.log(`memory-pro: smart-extractor: manual-echo guard dropped candidate (near-identical to a recent manual store) category=${candidate.category} abstract=${JSON.stringify(candidate.abstract.slice(0, 120))}`); + } + else { + kept.push(candidate); + } + } + candidates = kept; + } if (candidates.length === 0) { this.log("memory-pro: smart-extractor: no memories extracted"); if (extraction.status === "empty_input") { diff --git a/dist/src/tools.js b/dist/src/tools.js index e5bdf040..846b91c0 100644 --- a/dist/src/tools.js +++ b/dist/src/tools.js @@ -1318,6 +1318,7 @@ export function registerMemoryStoreTool(api, context) { // invalidation instead of silently reporting it as superseded. console.warn(`memory-pro: failed to invalidate superseded record ${failure.id.slice(0, 8)}: ${failure.reason}`); } + context.manualEchoLedger?.record(agentId, text); // Dual-write to Markdown mirror if enabled if (context.mdMirror) { await context.mdMirror({ text, category: storageCategory, scope: targetScope, timestamp: newEntry.timestamp }, { source: "memory_store", agentId }); @@ -1397,6 +1398,7 @@ export function registerMemoryStoreTool(api, context) { valid_until: validUntil, })), }); + context.manualEchoLedger?.record(agentId, text); // Dual-write to Markdown mirror if enabled if (context.mdMirror) { await context.mdMirror({ text, category: storageCategory, scope: targetScope, timestamp: entry.timestamp }, { source: "memory_store", agentId }); @@ -1780,6 +1782,7 @@ export function registerMemoryUpdateTool(api, context) { details: { error: "not_found", id: resolvedId }, }; } + context.manualEchoLedger?.record(agentId, updated.text); return { content: [ { diff --git a/index.ts b/index.ts index d044133b..ee733640 100644 --- a/index.ts +++ b/index.ts @@ -40,6 +40,7 @@ import { import { createScopeManager, resolveScopeFilter, isSystemBypassId, parseAgentIdFromSessionKey } from "./src/scopes.js"; import { createMigrator } from "./src/migrate.js"; import { registerAllMemoryTools } from "./src/tools.js"; +import { ManualEchoLedger } from "./src/manual-echo-guard.js"; import { appendSelfImprovementEntry, ensureSelfImprovementLearningFiles } from "./src/self-improvement-files.js"; import type { MdMirrorWriter } from "./src/tools.js"; import { shouldSkipRetrieval } from "./src/adaptive-retrieval.js"; @@ -2491,6 +2492,7 @@ interface PluginSingletonState { scopeManager: ReturnType; migrator: ReturnType; smartExtractor: SmartExtractor | null; + manualEchoLedger: ManualEchoLedger; mdMirror: MdMirrorWriter | null; extractionRateLimiter: ReturnType; // Session Maps — persist across scope refreshes instead of being recreated @@ -2698,6 +2700,10 @@ function _initPluginState(api: OpenClawPluginApi): PluginSingletonState { // enabled. admissionControl.enabled remains a supported configuration on // its own. let smartExtractor: SmartExtractor | null = null; + // Echo guard: shared between the manual store/update tools (record side) + // and the smart extractor (drop side); lives here so the tools keep + // recording even when smart extraction is disabled. + const manualEchoLedger = new ManualEchoLedger(); let admissionController: AdmissionController | null = null; let admissionControllerReflectionLane: AdmissionController | null = null; if (config.smartExtraction !== false || config.admissionControl?.enabled === true) { @@ -2784,6 +2790,7 @@ function _initPluginState(api: OpenClawPluginApi): PluginSingletonState { smartExtractor = new SmartExtractor(store, embedder, llmClient, { user: "User", + manualEchoLedger, captureAssistantEligible: config.captureAssistant === true, extractMinMessages: config.extractMinMessages ?? 4, extractMaxChars: config.extractMaxChars ?? 8000, @@ -2861,6 +2868,7 @@ function _initPluginState(api: OpenClawPluginApi): PluginSingletonState { scopeManager, migrator, smartExtractor, + manualEchoLedger, mdMirror, extractionRateLimiter, reflectionErrorStateBySession, @@ -3019,6 +3027,7 @@ const memoryLanceDBProPlugin = { scopeManager, migrator, smartExtractor, + manualEchoLedger, mdMirror, decayEngine, tierManager, @@ -3493,6 +3502,7 @@ const memoryLanceDBProPlugin = { workspaceBoundary: config.workspaceBoundary, selfImprovementMaxEntries: config.selfImprovement?.maxEntries, manualStoreSupersede: config.manualStoreSupersede === true, + manualEchoLedger, // Mirrors the CLI context wiring below: keep in-process reflection caches // consistent after a live memory_forget delete too, not just CLI delete/delete-bulk. onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), diff --git a/package.json b/package.json index b041e23b..c406c8c9 100644 --- a/package.json +++ b/package.json @@ -33,7 +33,7 @@ "skills/**/*.md" ], "scripts": { - "test": "node test/embedder-error-hints.test.mjs && node --test test/embedder-max-input-chars.test.mjs && node test/cjk-recursion-regression.test.mjs && node test/extraction-prompt-structural-noise.test.mjs && node test/i18n-memory-triggers.test.mjs && node test/migrate-legacy-schema.test.mjs && node --test test/config-session-strategy-migration.test.mjs && node --test test/scope-access-undefined.test.mjs && node --test test/reflection-bypass-hook.test.mjs && node --test test/reflection-unattributed-session-read.test.mjs && node --test test/smart-extractor-scope-filter.test.mjs && node --test test/store-empty-scope-filter.test.mjs && node --test test/recall-text-cleanup.test.mjs && node test/update-consistency-lancedb.test.mjs && node --test test/strip-envelope-metadata.test.mjs && node test/cli-smoke.mjs && node test/functional-e2e.mjs && node --test test/per-agent-auto-recall.test.mjs && node test/retriever-rerank-regression.mjs && node test/smart-memory-lifecycle.mjs && node test/smart-extractor-branches.mjs && node --test test/smart-extractor-noise-gating.test.mjs && node test/memory-capability-runtime.test.mjs && node --test test/startup-health-diagnostics.test.mjs && node test/corpus-indexer.test.mjs && node --test test/regex-fallback-bulk-store.test.mjs && node test/plugin-manifest-regression.mjs && node --test test/dreaming-engine.test.mjs && node --test test/session-summary-before-reset.test.mjs && node --test test/sync-plugin-version.test.mjs && node test/smart-metadata-v2.mjs && node test/vector-search-cosine.test.mjs && node test/context-support-e2e.mjs && node test/temporal-facts.test.mjs && node test/memory-update-supersede.test.mjs && node test/memory-update-metadata-refresh.test.mjs && node test/memory-upgrader-diagnostics.test.mjs && node --test test/llm-api-key-client.test.mjs && node --test test/llm-oauth-client.test.mjs && node --test test/cli-oauth-login.test.mjs && node --test test/workflow-fork-guards.test.mjs && node --test test/clawteam-scope.test.mjs && node --test test/cross-process-lock.test.mjs && node --test test/preference-slots.test.mjs && node test/is-latest-auto-supersede.test.mjs && node --test test/temporal-awareness.test.mjs && node --test test/command-reflection-guard.test.mjs && node --test test/tier1-counters.test.mjs && node --test test/startup-check-timeout.test.mjs && node --test test/memory-subsession-prompt-hooks.test.mjs && node --test test/read-consistency-interval.test.mjs && node --test test/reflection-distiller-hook-skip.test.mjs && node --test test/register-scope-dedup.test.mjs && node --test test/raw-run-distiller-hooks.test.mjs && node --test test/autocapture-watermark-reset.test.mjs && node --test test/autocapture-internal-session-guard.test.mjs && node --test test/memory-categories-storage-map.test.mjs && node --test test/delete-invalidate-reflection-caches.test.mjs && node --test test/reflection-mapped-rows-admission.test.mjs && node --test test/smart-metadata-source-classification.test.mjs && node --test test/reflection-embed-transient-retry.test.mjs && node --test test/scope-owner-leak-hardening.test.mjs && node --test test/isOwnedByAgent.test.mjs && node --test test/typed-array-vector-fetch.test.mjs && node --test test/extraction-grounding-register.test.mjs && node test/grounding-rejudge.test.mjs && node --test test/reverse-map-legacy-category.test.mjs && node --test test/reflection-mapped-category-stamping.test.mjs && node --test test/memory-upgrader-category-normalization.test.mjs && node --test test/autocapture-fallback-gating.test.mjs && node --test test/prompt-architecture.test.mjs && node test/extraction-category-rubric.test.mjs && node --test test/admission-control-batch-utility.test.mjs && node --test test/smart-extractor-batch-admission.test.mjs && node --test test/admission-control-prompt-shape.test.mjs && node --test test/smart-extractor-merge-accounting.test.mjs && node --test test/admission-utility-veto.test.mjs && node --test test/cli-subcommand-attachment.test.mjs && node --test test/admission-lane-model-affinity.test.mjs && node --test test/admission-model-resolution.test.mjs && node --test test/admission-controller-standalone.test.mjs && node --test test/smart-extractor-admission-controller-injection.test.mjs && node --test test/admission-without-smart-extraction.test.mjs && node --test test/llm-thinklevel.test.mjs && node --test test/memory-id-prefix-resolution.test.mjs && node --test test/manual-store-supersede.test.mjs && node --test test/extraction-transcript-speaker-tags.test.mjs && node --test test/session-compressor.test.mjs && node --test test/reflection-derived-cache-invalidation.test.mjs && node --test test/reflection-tagged-input.test.mjs && node --test test/reflection-mapped-uniform-pipeline.test.mjs && node --test test/llm-host-transport.test.mjs && node --test test/llm-host-transport-composition.test.mjs && node --test test/admission-control-host-transport.test.mjs && node --test test/llm-transport-credential-hygiene.test.mjs && node --test test/memory-consolidate.test.mjs && node --test test/memory-consolidate-cost-gate.test.mjs && node --test test/memory-consolidate-two-phase-apply.test.mjs && node --test test/memory-consolidate-admission-independence.test.mjs && node --test test/memory-consolidate-polish.test.mjs && node --test test/consolidate-cli-settled-persistence.test.mjs && node --test test/consolidate-cli-apply-exit-status.test.mjs && node --test test/invalidated-rows-visibility.test.mjs && node --test test/store-excludeinactive-default.test.mjs", + "test": "node test/embedder-error-hints.test.mjs && node --test test/embedder-max-input-chars.test.mjs && node test/cjk-recursion-regression.test.mjs && node test/extraction-prompt-structural-noise.test.mjs && node test/i18n-memory-triggers.test.mjs && node test/migrate-legacy-schema.test.mjs && node --test test/config-session-strategy-migration.test.mjs && node --test test/scope-access-undefined.test.mjs && node --test test/reflection-bypass-hook.test.mjs && node --test test/reflection-unattributed-session-read.test.mjs && node --test test/smart-extractor-scope-filter.test.mjs && node --test test/store-empty-scope-filter.test.mjs && node --test test/recall-text-cleanup.test.mjs && node test/update-consistency-lancedb.test.mjs && node --test test/strip-envelope-metadata.test.mjs && node test/cli-smoke.mjs && node test/functional-e2e.mjs && node --test test/per-agent-auto-recall.test.mjs && node test/retriever-rerank-regression.mjs && node test/smart-memory-lifecycle.mjs && node test/smart-extractor-branches.mjs && node --test test/smart-extractor-noise-gating.test.mjs && node test/memory-capability-runtime.test.mjs && node --test test/startup-health-diagnostics.test.mjs && node test/corpus-indexer.test.mjs && node --test test/regex-fallback-bulk-store.test.mjs && node test/plugin-manifest-regression.mjs && node --test test/dreaming-engine.test.mjs && node --test test/session-summary-before-reset.test.mjs && node --test test/sync-plugin-version.test.mjs && node test/smart-metadata-v2.mjs && node test/vector-search-cosine.test.mjs && node test/context-support-e2e.mjs && node test/temporal-facts.test.mjs && node test/memory-update-supersede.test.mjs && node test/memory-update-metadata-refresh.test.mjs && node test/memory-upgrader-diagnostics.test.mjs && node --test test/llm-api-key-client.test.mjs && node --test test/llm-oauth-client.test.mjs && node --test test/cli-oauth-login.test.mjs && node --test test/workflow-fork-guards.test.mjs && node --test test/clawteam-scope.test.mjs && node --test test/cross-process-lock.test.mjs && node --test test/preference-slots.test.mjs && node test/is-latest-auto-supersede.test.mjs && node --test test/temporal-awareness.test.mjs && node --test test/command-reflection-guard.test.mjs && node --test test/tier1-counters.test.mjs && node --test test/startup-check-timeout.test.mjs && node --test test/memory-subsession-prompt-hooks.test.mjs && node --test test/read-consistency-interval.test.mjs && node --test test/reflection-distiller-hook-skip.test.mjs && node --test test/register-scope-dedup.test.mjs && node --test test/raw-run-distiller-hooks.test.mjs && node --test test/autocapture-watermark-reset.test.mjs && node --test test/autocapture-internal-session-guard.test.mjs && node --test test/memory-categories-storage-map.test.mjs && node --test test/delete-invalidate-reflection-caches.test.mjs && node --test test/reflection-mapped-rows-admission.test.mjs && node --test test/smart-metadata-source-classification.test.mjs && node --test test/reflection-embed-transient-retry.test.mjs && node --test test/scope-owner-leak-hardening.test.mjs && node --test test/isOwnedByAgent.test.mjs && node --test test/typed-array-vector-fetch.test.mjs && node --test test/extraction-grounding-register.test.mjs && node test/grounding-rejudge.test.mjs && node --test test/reverse-map-legacy-category.test.mjs && node --test test/reflection-mapped-category-stamping.test.mjs && node --test test/memory-upgrader-category-normalization.test.mjs && node --test test/autocapture-fallback-gating.test.mjs && node --test test/prompt-architecture.test.mjs && node test/extraction-category-rubric.test.mjs && node --test test/admission-control-batch-utility.test.mjs && node --test test/smart-extractor-batch-admission.test.mjs && node --test test/admission-control-prompt-shape.test.mjs && node --test test/smart-extractor-merge-accounting.test.mjs && node --test test/admission-utility-veto.test.mjs && node --test test/cli-subcommand-attachment.test.mjs && node --test test/admission-lane-model-affinity.test.mjs && node --test test/admission-model-resolution.test.mjs && node --test test/admission-controller-standalone.test.mjs && node --test test/smart-extractor-admission-controller-injection.test.mjs && node --test test/admission-without-smart-extraction.test.mjs && node --test test/llm-thinklevel.test.mjs && node --test test/memory-id-prefix-resolution.test.mjs && node --test test/manual-store-supersede.test.mjs && node --test test/extraction-transcript-speaker-tags.test.mjs && node --test test/session-compressor.test.mjs && node --test test/reflection-derived-cache-invalidation.test.mjs && node --test test/reflection-tagged-input.test.mjs && node --test test/reflection-mapped-uniform-pipeline.test.mjs && node --test test/llm-host-transport.test.mjs && node --test test/llm-host-transport-composition.test.mjs && node --test test/admission-control-host-transport.test.mjs && node --test test/llm-transport-credential-hygiene.test.mjs && node --test test/memory-consolidate.test.mjs && node --test test/memory-consolidate-cost-gate.test.mjs && node --test test/memory-consolidate-two-phase-apply.test.mjs && node --test test/memory-consolidate-admission-independence.test.mjs && node --test test/memory-consolidate-polish.test.mjs && node --test test/consolidate-cli-settled-persistence.test.mjs && node --test test/consolidate-cli-apply-exit-status.test.mjs && node --test test/invalidated-rows-visibility.test.mjs && node --test test/store-excludeinactive-default.test.mjs && node --test test/manual-echo-guard.test.mjs", "test:cli-smoke": "node scripts/run-ci-tests.mjs --group cli-smoke", "test:core-regression": "node scripts/run-ci-tests.mjs --group core-regression", "test:storage-and-schema": "node scripts/run-ci-tests.mjs --group storage-and-schema", diff --git a/scripts/ci-test-manifest.mjs b/scripts/ci-test-manifest.mjs index d1c99619..e58e038d 100644 --- a/scripts/ci-test-manifest.mjs +++ b/scripts/ci-test-manifest.mjs @@ -158,6 +158,7 @@ export const CI_TEST_MANIFEST = [ { group: "core-regression", runner: "node", file: "test/consolidate-cli-apply-exit-status.test.mjs", args: ["--test"] }, { group: "core-regression", runner: "node", file: "test/invalidated-rows-visibility.test.mjs", args: ["--test"] }, { group: "core-regression", runner: "node", file: "test/store-excludeinactive-default.test.mjs", args: ["--test"] }, + { group: "core-regression", runner: "node", file: "test/manual-echo-guard.test.mjs", args: ["--test"] }, ]; export function getEntriesForGroup(group) { diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts new file mode 100644 index 00000000..a8640a37 --- /dev/null +++ b/src/manual-echo-guard.ts @@ -0,0 +1,116 @@ +/** + * Manual-store echo guard (class): when the user dictates a memory, + * the same sentence reaches BOTH the manual store lane (memory_store / + * memory_update, always-priority, verbatim) and auto-capture extraction, + * which mints near-twin candidates the dedup layer cannot reliably collide — + * the manual row may be seconds old (fresh-row vector visibility) or land in + * a different category. The guard remembers recent manual texts per agent + * and drops near-identical extraction candidates BEFORE the admission judge: + * deterministic, string-only, no LLM calls, no vector search. + * + * Scoped per agent (not per session): the store tool and the auto-capture + * hook derive their session keys differently, but both resolve the same + * agent id, and an echo of ANY recent manual text of the same agent is a + * correct drop regardless of session boundaries. The ring bounds staleness. + */ + +export const MANUAL_ECHO_JACCARD_THRESHOLD = 0.75; +export const MANUAL_ECHO_SUBSET_THRESHOLD = 0.9; +export const MANUAL_ECHO_RING_SIZE = 8; +const MAX_TRACKED_AGENTS = 128; +const MIN_CONTAINMENT_TOKENS = 3; +const DEFAULT_AGENT_BUCKET = "main"; + +export function normalizeEchoText(text: string): string { + return text + .toLowerCase() + .replace(/[^\p{L}\p{N}\s]+/gu, " ") + .replace(/\s+/g, " ") + .trim(); +} + +function tokenSet(normalized: string): Set { + return new Set(normalized.split(" ").filter((t) => t.length > 0)); +} + +function jaccard(a: Set, b: Set): number { + if (a.size === 0 || b.size === 0) return 0; + let intersection = 0; + for (const token of a) { + if (b.has(token)) intersection++; + } + return intersection / (a.size + b.size - intersection); +} + +function subsetRatio(inner: Set, outer: Set): number { + if (inner.size === 0) return 0; + let contained = 0; + for (const token of inner) { + if (outer.has(token)) contained++; + } + return contained / inner.size; +} + +export function isNearIdenticalEcho(candidateText: string, manualText: string): boolean { + const candidate = normalizeEchoText(candidateText); + const manual = normalizeEchoText(manualText); + if (candidate.length === 0 || manual.length === 0) return false; + if (candidate === manual) return true; + + const manualTokens = tokenSet(manual); + // Very short manual texts over-match as substrings ("blue mug" is inside + // any sentence mentioning it); those only count as echoes when exact. + if (manualTokens.size < MIN_CONTAINMENT_TOKENS) return false; + + if (candidate.includes(manual) || manual.includes(candidate)) return true; + + // Token-subset containment: the canonical echo shape is the extractor + // sentence-wrapping the manual fact ("favorite teacup: the red one" -> + // "User's favorite teacup is the red one"), where glue words break + // character containment and dilute Jaccard. One side's tokens (nearly) + // all present in the other = echo. + const candidateTokens = tokenSet(candidate); + if ( + subsetRatio(manualTokens, candidateTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD || + (candidateTokens.size >= MIN_CONTAINMENT_TOKENS && + subsetRatio(candidateTokens, manualTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD) + ) { + return true; + } + + return jaccard(candidateTokens, manualTokens) >= MANUAL_ECHO_JACCARD_THRESHOLD; +} + +export class ManualEchoLedger { + private readonly byAgent = new Map(); + + record(agentId: string | undefined, text: string): void { + if (typeof text !== "string" || text.trim().length === 0) return; + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key) ?? []; + ring.push(text); + while (ring.length > MANUAL_ECHO_RING_SIZE) ring.shift(); + this.byAgent.delete(key); + this.byAgent.set(key, ring); + while (this.byAgent.size > MAX_TRACKED_AGENTS) { + const oldest = this.byAgent.keys().next().value; + if (oldest === undefined) break; + this.byAgent.delete(oldest); + } + } + + /** Returns the matched manual text, or null when the candidate is no echo. */ + match(agentId: string | undefined, candidateText: string): string | null { + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key); + if (!ring || ring.length === 0) return null; + for (let i = ring.length - 1; i >= 0; i--) { + if (isNearIdenticalEcho(candidateText, ring[i])) return ring[i]; + } + return null; + } + + clear(agentId: string | undefined): void { + this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); + } +} diff --git a/src/smart-extractor.ts b/src/smart-extractor.ts index 27c5c204..1bb82700 100644 --- a/src/smart-extractor.ts +++ b/src/smart-extractor.ts @@ -17,6 +17,7 @@ import { buildBatchDedupPrompt, buildBatchMergePrompt, } from "./extraction-prompts.js"; +import type { ManualEchoLedger } from "./manual-echo-guard.js"; import { formatExistingMemoryEntry } from "./prompt-blocks.js"; import { AdmissionController, @@ -545,6 +546,8 @@ export interface SmartExtractorConfig { user?: string; /** Minimum conversation messages before extraction triggers. */ extractMinMessages?: number; + /** Echo guard: drops candidates near-identical to a recent manual memory_store/memory_update text, pre-judge. */ + manualEchoLedger?: ManualEchoLedger; /** Maximum characters of conversation text to process. */ extractMaxChars?: number; /** Per-call chunk bound for the batched dedup decider and merge writer (1-50, default 10). */ @@ -726,7 +729,25 @@ export class SmartExtractor { options.conversationTurns, options.protectedPrefixTurns, ); - const candidates = extraction.candidates; + let candidates = extraction.candidates; + + // Echo guard: candidates near-identical to a recent manual + // memory_store/memory_update text are echoes of a row that already + // exists verbatim -- drop them before any judge/dedup/merge spend. + const echoLedger = this.config.manualEchoLedger; + if (echoLedger && candidates.length > 0) { + const kept: CandidateMemory[] = []; + for (const candidate of candidates) { + if (echoLedger.match(agentId, candidate.content)) { + this.log( + `memory-pro: smart-extractor: manual-echo guard dropped candidate (near-identical to a recent manual store) category=${candidate.category} abstract=${JSON.stringify(candidate.abstract.slice(0, 120))}`, + ); + } else { + kept.push(candidate); + } + } + candidates = kept; + } if (candidates.length === 0) { this.log("memory-pro: smart-extractor: no memories extracted"); diff --git a/src/tools.ts b/src/tools.ts index 6ac327cd..1221eab8 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -45,6 +45,7 @@ import { type WorkspaceBoundaryConfig, } from "./workspace-boundary.js"; import { isSuppressed as isTier1Suppressed } from "./auto-recall-tier1.js"; +import type { ManualEchoLedger } from "./manual-echo-guard.js"; import { enqueueManualRecallMetadata } from "./manual-recall-metadata-queue.js"; // ============================================================================ @@ -78,6 +79,9 @@ interface ToolContext { // row supersedes it instead of being rejected as a duplicate; the manual // text always lands verbatim. manualStoreSupersede?: boolean; + // Echo guard: manual store/update texts are recorded here so + // auto-capture extraction can drop near-identical echo candidates. + manualEchoLedger?: ManualEchoLedger; // Mirrors MemoryCliContext's onMemoriesDeleted (cli.ts): lets the host invalidate // in-process reflection caches after a live delete, not just CLI delete/delete-bulk. onMemoriesDeleted?: (info: { scopeFilter?: string[] }) => void; @@ -1695,6 +1699,8 @@ export function registerMemoryStoreTool( ); } + context.manualEchoLedger?.record(agentId, text); + // Dual-write to Markdown mirror if enabled if (context.mdMirror) { await context.mdMirror( @@ -1786,6 +1792,8 @@ export function registerMemoryStoreTool( ), }); + context.manualEchoLedger?.record(agentId, text); + // Dual-write to Markdown mirror if enabled if (context.mdMirror) { await context.mdMirror( @@ -2249,6 +2257,8 @@ export function registerMemoryUpdateTool( }; } + context.manualEchoLedger?.record(agentId, updated.text); + return { content: [ { diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs new file mode 100644 index 00000000..b7e767f6 --- /dev/null +++ b/test/manual-echo-guard.test.mjs @@ -0,0 +1,173 @@ +/** + * Manual-store echo guard: deterministic pre-judge drop of extraction + * candidates that near-duplicate a recent manual memory_store/memory_update + * text. + * + * Mechanism (design ruling 2026-07-21): when the user dictates a memory + * ("remember this: ..."), the same sentence flows through BOTH the manual + * store lane and auto-capture extraction, minting near-twin rows the dedup + * layer cannot reliably collide (fresh-row vector visibility, category + * splits). The guard keeps an in-memory per-agent ring of recent manual + * texts and drops near-identical extraction candidates BEFORE the admission + * judge — string-only comparison, no LLM, no vector search. + * + * Match test: normalized containment (either direction, min token guard) or + * token-set Jaccard >= 0.75 on the candidate content. + * + * Fixtures are entirely synthetic; no real fleet data. + */ + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import jitiFactory from "jiti"; + +const jiti = jitiFactory(import.meta.url, { interopDefault: true }); +const { + ManualEchoLedger, + isNearIdenticalEcho, + normalizeEchoText, + MANUAL_ECHO_JACCARD_THRESHOLD, + MANUAL_ECHO_RING_SIZE, +} = jiti("../src/manual-echo-guard.ts"); + +describe("normalizeEchoText", () => { + it("lowercases, strips punctuation, collapses whitespace", () => { + assert.equal( + normalizeEchoText(" Favorite Teacup: the RED one! "), + "favorite teacup the red one", + ); + }); + + it("keeps unicode letters and digits", () => { + assert.equal(normalizeEchoText("Çay saati 15:30'da"), "çay saati 15 30 da"); + }); +}); + +describe("isNearIdenticalEcho", () => { + const manual = "the office plant needs watering every friday"; + + it("matches exact text", () => { + assert.equal(isNearIdenticalEcho(manual, manual), true); + }); + + it("matches when the candidate contains the manual text", () => { + assert.equal( + isNearIdenticalEcho( + `User stated that the office plant needs watering every Friday.`, + manual, + ), + true, + ); + }); + + it("matches when the manual text contains the candidate", () => { + assert.equal( + isNearIdenticalEcho("office plant needs watering", manual), + true, + ); + }); + + it("matches high token overlap above the Jaccard threshold", () => { + assert.equal( + isNearIdenticalEcho( + "office plant needs deep watering every friday morning", + "the office plant needs watering every friday morning", + ), + true, + ); + }); + + it("matches the sentence-wrapped echo shape via token-subset containment", () => { + assert.equal( + isNearIdenticalEcho( + "User's favorite teacup is the red one", + "favorite teacup: the red one", + ), + true, + ); + }); + + it("rejects unrelated candidates", () => { + assert.equal( + isNearIdenticalEcho("user's dog is named Biscuit", manual), + false, + ); + }); + + it("rejects low-overlap candidates sharing a few tokens", () => { + assert.equal( + isNearIdenticalEcho( + "user waters the garden on weekends with a hose", + manual, + ), + false, + ); + }); + + it("requires exact match for very short manual texts", () => { + assert.equal(isNearIdenticalEcho("blue mug", "blue mug"), true); + assert.equal( + isNearIdenticalEcho("user owns a blue mug from portugal", "blue mug"), + false, + ); + }); + + it("exposes the documented threshold", () => { + assert.equal(MANUAL_ECHO_JACCARD_THRESHOLD, 0.75); + }); +}); + +describe("ManualEchoLedger", () => { + it("records and matches per agent", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + assert.ok( + ledger.match("agent-one", "User's favorite teacup is the red one"), + ); + assert.equal( + ledger.match("agent-two", "User's favorite teacup is the red one"), + null, + ); + }); + + it("returns null when nothing recorded", () => { + const ledger = new ManualEchoLedger(); + assert.equal(ledger.match("agent-one", "anything at all"), null); + }); + + it("buckets undefined agent ids together", () => { + const ledger = new ManualEchoLedger(); + ledger.record(undefined, "kneeling chair height is 104cm"); + assert.ok(ledger.match(undefined, "the kneeling chair height is 104cm")); + }); + + it("caps the ring and evicts the oldest entry", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "the very first manual fact about topic zero"); + for (let i = 1; i <= MANUAL_ECHO_RING_SIZE; i++) { + ledger.record("agent-one", `distinct manual fact number ${i} about topic ${i}`); + } + assert.equal( + ledger.match("agent-one", "the very first manual fact about topic zero"), + null, + ); + assert.ok( + ledger.match("agent-one", `distinct manual fact number ${MANUAL_ECHO_RING_SIZE} about topic ${MANUAL_ECHO_RING_SIZE}`), + ); + }); + + it("ignores empty and whitespace-only records", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", " "); + assert.equal(ledger.match("agent-one", " "), null); + }); + + it("clear() empties one agent's ring only", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + ledger.record("agent-two", "favorite teacup: the red one"); + ledger.clear("agent-one"); + assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); + assert.ok(ledger.match("agent-two", "favorite teacup: the red one")); + }); +}); From 6a06eff1e38f211015c84f5f501f68bf1c6957a1 Mon Sep 17 00:00:00 2001 From: Gorkem Date: Tue, 1 Sep 2026 11:14:38 +0300 Subject: [PATCH 2/6] fix(echo-guard): conservative one-sided matching, TTL+consume ledger, settled echo batches, CJK fallback, forget invalidation Round-1 review: the guard must never trade duplicate rows for silent loss of legitimate updates. 1. Matching is now one-sided: a candidate is an echo only when it adds NOTHING beyond the manual text (exact, manual-contains-candidate, or full content-token containment after glue-word stripping). Negation and temporal markers on either side refuse the match, so corrections, changed values, qualified statements, and added facts always survive. The bidirectional Jaccard/subset fuzz is gone. 2. Ledger entries carry a 10-minute TTL and are consumed on match: one manual store suppresses at most one echo, so later identical statements are deliberate re-assertions and never dropped. 3. The successful temporal memory_update supersede path records the new text before its early return (handler-level regression included). 4. An echo-only batch counts its drops as skipped and reports settledOutcomes, so the auto-capture caller consumes the input instead of deferring a retry that re-runs the same extraction. 5. CJK fallback: whitespace-stripped containment with a marker-guarded wrapper budget, so wrapped CJK echoes drop while qualified or negated CJK statements survive. 6. memory_forget invalidates the deleted row's ledger entry on both the id and query paths. test/manual-echo-guard.test.mjs rewritten around the new contract (31 tests incl. full auto-capture path + handler-level supersede coverage). --- dist/src/manual-echo-guard.js | 206 ++++++++++++---- dist/src/smart-extractor.js | 10 + dist/src/tools.js | 14 ++ src/manual-echo-guard.ts | 209 ++++++++++++---- src/smart-extractor.ts | 10 + src/tools.ts | 14 ++ test/manual-echo-guard.test.mjs | 411 ++++++++++++++++++++++++++++++-- 7 files changed, 767 insertions(+), 107 deletions(-) diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js index 8606f231..1cedb39b 100644 --- a/dist/src/manual-echo-guard.js +++ b/dist/src/manual-echo-guard.js @@ -8,17 +8,64 @@ * and drops near-identical extraction candidates BEFORE the admission judge: * deterministic, string-only, no LLM calls, no vector search. * + * Matching is deliberately ONE-SIDED and conservative: a candidate is an + * echo only when it adds NOTHING substantive beyond the recorded manual + * text (exact match, the manual text containing the candidate, or every + * candidate content token already present in the manual text). A candidate + * carrying extra content — a negation ("no longer"), a changed value, a + * temporal qualifier ("until friday"), or additional facts — is new + * information and always survives; the worst case of the guard staying + * quiet is the pre-guard status quo (one duplicate row for dedup). + * + * Entries are short-lived and consumed: each recorded manual text expires + * after MANUAL_ECHO_TTL_MS and suppresses at most ONE candidate (the + * immediate re-extraction of the same turn). A later identical statement is + * a deliberate user re-assertion, not an echo. + * * Scoped per agent (not per session): the store tool and the auto-capture * hook derive their session keys differently, but both resolve the same * agent id, and an echo of ANY recent manual text of the same agent is a - * correct drop regardless of session boundaries. The ring bounds staleness. + * correct drop regardless of session boundaries. TTL + consumption bound + * staleness; the ring bounds size. */ -export const MANUAL_ECHO_JACCARD_THRESHOLD = 0.75; -export const MANUAL_ECHO_SUBSET_THRESHOLD = 0.9; export const MANUAL_ECHO_RING_SIZE = 8; +export const MANUAL_ECHO_TTL_MS = 10 * 60 * 1000; const MAX_TRACKED_AGENTS = 128; const MIN_CONTAINMENT_TOKENS = 3; +const MIN_CJK_CONTAINMENT_CHARS = 6; +const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; +/** + * Glue vocabulary the extractor wraps a dictated fact in ("User stated + * that ..."). Stripped before token containment so the canonical wrap echo + * still collapses; negation and temporal markers are deliberately NOT here. + */ +const ECHO_STOPWORDS = new Set([ + "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", + "that", "this", "these", "those", "to", "of", "in", "on", "at", "for", + "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", + "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", + "your", "user", "users", "stated", "said", "says", "saying", "mentioned", + "noted", "prefers", "prefer", "likes", "like", "wants", "want", "has", + "have", "had", "also", +]); +/** + * A marker on exactly one side of the pair means the two texts assert + * different things (a correction, a retraction, a bounded validity): never + * treat that as an echo. + */ +const NEGATION_AND_TEMPORAL_MARKERS = new Set([ + "no", "not", "never", "none", "stopped", "stop", "stops", "quit", + "former", "formerly", "anymore", "longer", "until", "till", "unless", + "except", "without", "before", "after", "used", +]); +/** Conservative CJK marker fragments (negation / bounded validity). */ +const CJK_MARKER_FRAGMENTS = [ + "不", "没", "别", "未", "无", "非", "勿", "直到", "之前", "以前", "除非", + "ない", "じゃない", "ではない", "まで", "もう", + "않", "안", "까지", "전에", +]; +const CJK_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u; export function normalizeEchoText(text) { return text .toLowerCase() @@ -26,28 +73,52 @@ export function normalizeEchoText(text) { .replace(/\s+/g, " ") .trim(); } -function tokenSet(normalized) { - return new Set(normalized.split(" ").filter((t) => t.length > 0)); +function tokenList(normalized) { + return normalized.split(" ").filter((t) => t.length > 0); } -function jaccard(a, b) { - if (a.size === 0 || b.size === 0) - return 0; - let intersection = 0; - for (const token of a) { - if (b.has(token)) - intersection++; +function contentTokens(normalized) { + const out = new Set(); + for (const token of tokenList(normalized)) { + if (token.length <= 1 && !CJK_RE.test(token)) + continue; + if (ECHO_STOPWORDS.has(token)) + continue; + out.add(token); } - return intersection / (a.size + b.size - intersection); + return out; } -function subsetRatio(inner, outer) { - if (inner.size === 0) - return 0; - let contained = 0; - for (const token of inner) { - if (outer.has(token)) - contained++; +function markerAsymmetry(aTokens, bTokens) { + const a = new Set(aTokens); + const b = new Set(bTokens); + for (const marker of NEGATION_AND_TEMPORAL_MARKERS) { + if (a.has(marker) !== b.has(marker)) + return true; } - return contained / inner.size; + return false; +} +function containsCjkMarker(fragment) { + return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); +} +function isCjkEcho(candidate, manual) { + const cand = candidate.replace(/\s+/g, ""); + const man = manual.replace(/\s+/g, ""); + if (cand.length === 0 || man.length === 0) + return false; + if (cand === man) + return true; + // Shortened echo: the candidate re-states a piece of the manual text and + // adds nothing. Length-gated so tiny fragments cannot over-match. + if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) + return true; + // Wrapped echo: the candidate is the manual text plus a small amount of + // glue ("用户说…"). Allowed only when the residual is short AND carries no + // negation/temporal marker, so a qualified or corrected statement + // ("…直到周五", "不再…") is never treated as an echo. + if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { + const residual = cand.replace(man, ""); + return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && !containsCjkMarker(residual); + } + return false; } export function isNearIdenticalEcho(candidateText, manualText) { const candidate = normalizeEchoText(candidateText); @@ -56,34 +127,46 @@ export function isNearIdenticalEcho(candidateText, manualText) { return false; if (candidate === manual) return true; - const manualTokens = tokenSet(manual); + if (CJK_RE.test(candidate) || CJK_RE.test(manual)) { + return isCjkEcho(candidate, manual); + } + const manualTokenList = tokenList(manual); + const candidateTokenList = tokenList(candidate); + // A correction, retraction, or bounded-validity statement is never an + // echo, whichever side carries the marker. + if (markerAsymmetry(manualTokenList, candidateTokenList)) + return false; // Very short manual texts over-match as substrings ("blue mug" is inside // any sentence mentioning it); those only count as echoes when exact. - if (manualTokens.size < MIN_CONTAINMENT_TOKENS) + const manualContent = contentTokens(manual); + if (manualContent.size < MIN_CONTAINMENT_TOKENS) return false; - if (candidate.includes(manual) || manual.includes(candidate)) - return true; - // Token-subset containment: the canonical echo shape is the extractor - // sentence-wrapping the manual fact ("favorite teacup: the red one" -> - // "User's favorite teacup is the red one"), where glue words break - // character containment and dilute Jaccard. One side's tokens (nearly) - // all present in the other = echo. - const candidateTokens = tokenSet(candidate); - if (subsetRatio(manualTokens, candidateTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD || - (candidateTokens.size >= MIN_CONTAINMENT_TOKENS && - subsetRatio(candidateTokens, manualTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD)) { + // Shortened echo: the manual text contains the whole candidate. + if (manual.includes(candidate)) return true; + // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite + // teacup: the red one" -> "User stated their favorite teacup is the red + // one"). After glue-word stripping, EVERY candidate content token must + // already be in the manual text — one-sided by design, so a candidate + // with any extra content token (changed value, qualifier, added fact) is + // new information and survives. + const candidateContent = contentTokens(candidate); + if (candidateContent.size === 0) + return false; + for (const token of candidateContent) { + if (!manualContent.has(token)) + return false; } - return jaccard(candidateTokens, manualTokens) >= MANUAL_ECHO_JACCARD_THRESHOLD; + return true; } export class ManualEchoLedger { byAgent = new Map(); - record(agentId, text) { + record(agentId, text, now = Date.now()) { if (typeof text !== "string" || text.trim().length === 0) return; const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; - const ring = this.byAgent.get(key) ?? []; - ring.push(text); + const ring = (this.byAgent.get(key) ?? []).filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); + ring.push({ text, at: now }); while (ring.length > MANUAL_ECHO_RING_SIZE) ring.shift(); this.byAgent.delete(key); @@ -95,18 +178,55 @@ export class ManualEchoLedger { this.byAgent.delete(oldest); } } - /** Returns the matched manual text, or null when the candidate is no echo. */ - match(agentId, candidateText) { + /** + * Returns the matched manual text, or null when the candidate is no echo. + * A hit CONSUMES the entry: each manual store suppresses at most one + * candidate, so a later identical statement (a deliberate re-assertion) + * is never silently dropped. + */ + match(agentId, candidateText, now = Date.now()) { const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; const ring = this.byAgent.get(key); if (!ring || ring.length === 0) return null; - for (let i = ring.length - 1; i >= 0; i--) { - if (isNearIdenticalEcho(candidateText, ring[i])) - return ring[i]; + const live = ring.filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); + if (live.length !== ring.length) { + if (live.length === 0) { + this.byAgent.delete(key); + return null; + } + this.byAgent.set(key, live); + } + for (let i = live.length - 1; i >= 0; i--) { + if (isNearIdenticalEcho(candidateText, live[i].text)) { + const [hit] = live.splice(i, 1); + if (live.length === 0) + this.byAgent.delete(key); + return hit.text; + } } return null; } + /** + * Drops entries matching a deleted memory's text, so a forgotten manual + * fact can never keep suppressing its own re-statement. + */ + invalidate(agentId, text) { + if (typeof text !== "string" || text.trim().length === 0) + return; + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key); + if (!ring || ring.length === 0) + return; + const target = normalizeEchoText(text); + const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); + if (kept.length === 0) { + this.byAgent.delete(key); + } + else if (kept.length !== ring.length) { + this.byAgent.set(key, kept); + } + } clear(agentId) { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } diff --git a/dist/src/smart-extractor.js b/dist/src/smart-extractor.js index 548be421..b913e014 100644 --- a/dist/src/smart-extractor.js +++ b/dist/src/smart-extractor.js @@ -460,10 +460,17 @@ export class SmartExtractor { // memory_store/memory_update text are echoes of a row that already // exists verbatim -- drop them before any judge/dedup/merge spend. const echoLedger = this.config.manualEchoLedger; + let echoDropped = 0; if (echoLedger && candidates.length > 0) { const kept = []; for (const candidate of candidates) { if (echoLedger.match(agentId, candidate.content)) { + // An echo drop is a SETTLED outcome (the fact already exists as + // the manual row), so it counts as skipped: an echo-only batch + // must consume its input instead of deferring for a retry that + // would re-run the same extraction. + echoDropped += 1; + stats.skipped += 1; this.log(`memory-pro: smart-extractor: manual-echo guard dropped candidate (near-identical to a recent manual store) category=${candidate.category} abstract=${JSON.stringify(candidate.abstract.slice(0, 120))}`); } else { @@ -474,6 +481,9 @@ export class SmartExtractor { } if (candidates.length === 0) { this.log("memory-pro: smart-extractor: no memories extracted"); + if (echoDropped > 0) { + stats.settledOutcomes = true; + } if (extraction.status === "empty_input") { // No LLM call was made, so the caller's rate limiter must not be charged. stats.skippedNoInput = true; diff --git a/dist/src/tools.js b/dist/src/tools.js index 846b91c0..cee386ce 100644 --- a/dist/src/tools.js +++ b/dist/src/tools.js @@ -1487,8 +1487,16 @@ export function registerMemoryForgetTool(api, context) { details: resolved.details ?? { error: "not_found", id: memoryId }, }; } + const forgottenRow = context.manualEchoLedger + ? await context.store.getById(resolved.id, scopeFilter).catch(() => null) + : null; const deleted = await context.store.delete(resolved.id, scopeFilter); if (deleted) { + // A forgotten fact must not keep suppressing its own + // re-statement through the echo ledger. + if (forgottenRow?.text) { + context.manualEchoLedger?.invalidate(agentId, forgottenRow.text); + } context.onMemoriesDeleted?.({ scopeFilter }); return { content: [ @@ -1526,6 +1534,7 @@ export function registerMemoryForgetTool(api, context) { if (results.length === 1 && results[0].score > 0.9) { const deleted = await context.store.delete(results[0].entry.id, scopeFilter); if (deleted) { + context.manualEchoLedger?.invalidate(agentId, results[0].entry.text); context.onMemoriesDeleted?.({ scopeFilter }); return { content: [ @@ -1711,6 +1720,11 @@ export function registerMemoryUpdateTool(api, context) { // New record is already the source of truth; log but don't fail console.warn(`memory-pro: failed to patch superseded record ${resolvedId.slice(0, 8)}: ${patchErr}`); } + // The superseding write succeeded: this text will echo through + // the same turn's auto-capture extraction exactly like a plain + // manual store, so it must be recorded on this early-return + // path too. + context.manualEchoLedger?.record(agentId, text); return { content: [ { diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts index a8640a37..f75d8f79 100644 --- a/src/manual-echo-guard.ts +++ b/src/manual-echo-guard.ts @@ -8,19 +8,70 @@ * and drops near-identical extraction candidates BEFORE the admission judge: * deterministic, string-only, no LLM calls, no vector search. * + * Matching is deliberately ONE-SIDED and conservative: a candidate is an + * echo only when it adds NOTHING substantive beyond the recorded manual + * text (exact match, the manual text containing the candidate, or every + * candidate content token already present in the manual text). A candidate + * carrying extra content — a negation ("no longer"), a changed value, a + * temporal qualifier ("until friday"), or additional facts — is new + * information and always survives; the worst case of the guard staying + * quiet is the pre-guard status quo (one duplicate row for dedup). + * + * Entries are short-lived and consumed: each recorded manual text expires + * after MANUAL_ECHO_TTL_MS and suppresses at most ONE candidate (the + * immediate re-extraction of the same turn). A later identical statement is + * a deliberate user re-assertion, not an echo. + * * Scoped per agent (not per session): the store tool and the auto-capture * hook derive their session keys differently, but both resolve the same * agent id, and an echo of ANY recent manual text of the same agent is a - * correct drop regardless of session boundaries. The ring bounds staleness. + * correct drop regardless of session boundaries. TTL + consumption bound + * staleness; the ring bounds size. */ -export const MANUAL_ECHO_JACCARD_THRESHOLD = 0.75; -export const MANUAL_ECHO_SUBSET_THRESHOLD = 0.9; export const MANUAL_ECHO_RING_SIZE = 8; +export const MANUAL_ECHO_TTL_MS = 10 * 60 * 1000; const MAX_TRACKED_AGENTS = 128; const MIN_CONTAINMENT_TOKENS = 3; +const MIN_CJK_CONTAINMENT_CHARS = 6; +const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; +/** + * Glue vocabulary the extractor wraps a dictated fact in ("User stated + * that ..."). Stripped before token containment so the canonical wrap echo + * still collapses; negation and temporal markers are deliberately NOT here. + */ +const ECHO_STOPWORDS = new Set([ + "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", + "that", "this", "these", "those", "to", "of", "in", "on", "at", "for", + "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", + "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", + "your", "user", "users", "stated", "said", "says", "saying", "mentioned", + "noted", "prefers", "prefer", "likes", "like", "wants", "want", "has", + "have", "had", "also", +]); + +/** + * A marker on exactly one side of the pair means the two texts assert + * different things (a correction, a retraction, a bounded validity): never + * treat that as an echo. + */ +const NEGATION_AND_TEMPORAL_MARKERS = new Set([ + "no", "not", "never", "none", "stopped", "stop", "stops", "quit", + "former", "formerly", "anymore", "longer", "until", "till", "unless", + "except", "without", "before", "after", "used", +]); + +/** Conservative CJK marker fragments (negation / bounded validity). */ +const CJK_MARKER_FRAGMENTS = [ + "不", "没", "别", "未", "无", "非", "勿", "直到", "之前", "以前", "除非", + "ない", "じゃない", "ではない", "まで", "もう", + "않", "안", "까지", "전에", +]; + +const CJK_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u; + export function normalizeEchoText(text: string): string { return text .toLowerCase() @@ -29,26 +80,50 @@ export function normalizeEchoText(text: string): string { .trim(); } -function tokenSet(normalized: string): Set { - return new Set(normalized.split(" ").filter((t) => t.length > 0)); +function tokenList(normalized: string): string[] { + return normalized.split(" ").filter((t) => t.length > 0); } -function jaccard(a: Set, b: Set): number { - if (a.size === 0 || b.size === 0) return 0; - let intersection = 0; - for (const token of a) { - if (b.has(token)) intersection++; +function contentTokens(normalized: string): Set { + const out = new Set(); + for (const token of tokenList(normalized)) { + if (token.length <= 1 && !CJK_RE.test(token)) continue; + if (ECHO_STOPWORDS.has(token)) continue; + out.add(token); } - return intersection / (a.size + b.size - intersection); + return out; +} + +function markerAsymmetry(aTokens: string[], bTokens: string[]): boolean { + const a = new Set(aTokens); + const b = new Set(bTokens); + for (const marker of NEGATION_AND_TEMPORAL_MARKERS) { + if (a.has(marker) !== b.has(marker)) return true; + } + return false; +} + +function containsCjkMarker(fragment: string): boolean { + return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); } -function subsetRatio(inner: Set, outer: Set): number { - if (inner.size === 0) return 0; - let contained = 0; - for (const token of inner) { - if (outer.has(token)) contained++; +function isCjkEcho(candidate: string, manual: string): boolean { + const cand = candidate.replace(/\s+/g, ""); + const man = manual.replace(/\s+/g, ""); + if (cand.length === 0 || man.length === 0) return false; + if (cand === man) return true; + // Shortened echo: the candidate re-states a piece of the manual text and + // adds nothing. Length-gated so tiny fragments cannot over-match. + if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) return true; + // Wrapped echo: the candidate is the manual text plus a small amount of + // glue ("用户说…"). Allowed only when the residual is short AND carries no + // negation/temporal marker, so a qualified or corrected statement + // ("…直到周五", "不再…") is never treated as an echo. + if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { + const residual = cand.replace(man, ""); + return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && !containsCjkMarker(residual); } - return contained / inner.size; + return false; } export function isNearIdenticalEcho(candidateText: string, manualText: string): boolean { @@ -57,38 +132,51 @@ export function isNearIdenticalEcho(candidateText: string, manualText: string): if (candidate.length === 0 || manual.length === 0) return false; if (candidate === manual) return true; - const manualTokens = tokenSet(manual); + if (CJK_RE.test(candidate) || CJK_RE.test(manual)) { + return isCjkEcho(candidate, manual); + } + + const manualTokenList = tokenList(manual); + const candidateTokenList = tokenList(candidate); + // A correction, retraction, or bounded-validity statement is never an + // echo, whichever side carries the marker. + if (markerAsymmetry(manualTokenList, candidateTokenList)) return false; + // Very short manual texts over-match as substrings ("blue mug" is inside // any sentence mentioning it); those only count as echoes when exact. - if (manualTokens.size < MIN_CONTAINMENT_TOKENS) return false; - - if (candidate.includes(manual) || manual.includes(candidate)) return true; - - // Token-subset containment: the canonical echo shape is the extractor - // sentence-wrapping the manual fact ("favorite teacup: the red one" -> - // "User's favorite teacup is the red one"), where glue words break - // character containment and dilute Jaccard. One side's tokens (nearly) - // all present in the other = echo. - const candidateTokens = tokenSet(candidate); - if ( - subsetRatio(manualTokens, candidateTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD || - (candidateTokens.size >= MIN_CONTAINMENT_TOKENS && - subsetRatio(candidateTokens, manualTokens) >= MANUAL_ECHO_SUBSET_THRESHOLD) - ) { - return true; + const manualContent = contentTokens(manual); + if (manualContent.size < MIN_CONTAINMENT_TOKENS) return false; + + // Shortened echo: the manual text contains the whole candidate. + if (manual.includes(candidate)) return true; + + // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite + // teacup: the red one" -> "User stated their favorite teacup is the red + // one"). After glue-word stripping, EVERY candidate content token must + // already be in the manual text — one-sided by design, so a candidate + // with any extra content token (changed value, qualifier, added fact) is + // new information and survives. + const candidateContent = contentTokens(candidate); + if (candidateContent.size === 0) return false; + for (const token of candidateContent) { + if (!manualContent.has(token)) return false; } + return true; +} - return jaccard(candidateTokens, manualTokens) >= MANUAL_ECHO_JACCARD_THRESHOLD; +interface ManualEchoEntry { + text: string; + at: number; } export class ManualEchoLedger { - private readonly byAgent = new Map(); + private readonly byAgent = new Map(); - record(agentId: string | undefined, text: string): void { + record(agentId: string | undefined, text: string, now: number = Date.now()): void { if (typeof text !== "string" || text.trim().length === 0) return; const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; - const ring = this.byAgent.get(key) ?? []; - ring.push(text); + const ring = (this.byAgent.get(key) ?? []).filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); + ring.push({ text, at: now }); while (ring.length > MANUAL_ECHO_RING_SIZE) ring.shift(); this.byAgent.delete(key); this.byAgent.set(key, ring); @@ -99,17 +187,52 @@ export class ManualEchoLedger { } } - /** Returns the matched manual text, or null when the candidate is no echo. */ - match(agentId: string | undefined, candidateText: string): string | null { + /** + * Returns the matched manual text, or null when the candidate is no echo. + * A hit CONSUMES the entry: each manual store suppresses at most one + * candidate, so a later identical statement (a deliberate re-assertion) + * is never silently dropped. + */ + match(agentId: string | undefined, candidateText: string, now: number = Date.now()): string | null { const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; const ring = this.byAgent.get(key); if (!ring || ring.length === 0) return null; - for (let i = ring.length - 1; i >= 0; i--) { - if (isNearIdenticalEcho(candidateText, ring[i])) return ring[i]; + const live = ring.filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); + if (live.length !== ring.length) { + if (live.length === 0) { + this.byAgent.delete(key); + return null; + } + this.byAgent.set(key, live); + } + for (let i = live.length - 1; i >= 0; i--) { + if (isNearIdenticalEcho(candidateText, live[i].text)) { + const [hit] = live.splice(i, 1); + if (live.length === 0) this.byAgent.delete(key); + return hit.text; + } } return null; } + /** + * Drops entries matching a deleted memory's text, so a forgotten manual + * fact can never keep suppressing its own re-statement. + */ + invalidate(agentId: string | undefined, text: string): void { + if (typeof text !== "string" || text.trim().length === 0) return; + const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; + const ring = this.byAgent.get(key); + if (!ring || ring.length === 0) return; + const target = normalizeEchoText(text); + const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); + if (kept.length === 0) { + this.byAgent.delete(key); + } else if (kept.length !== ring.length) { + this.byAgent.set(key, kept); + } + } + clear(agentId: string | undefined): void { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } diff --git a/src/smart-extractor.ts b/src/smart-extractor.ts index 1bb82700..9f9159ff 100644 --- a/src/smart-extractor.ts +++ b/src/smart-extractor.ts @@ -735,10 +735,17 @@ export class SmartExtractor { // memory_store/memory_update text are echoes of a row that already // exists verbatim -- drop them before any judge/dedup/merge spend. const echoLedger = this.config.manualEchoLedger; + let echoDropped = 0; if (echoLedger && candidates.length > 0) { const kept: CandidateMemory[] = []; for (const candidate of candidates) { if (echoLedger.match(agentId, candidate.content)) { + // An echo drop is a SETTLED outcome (the fact already exists as + // the manual row), so it counts as skipped: an echo-only batch + // must consume its input instead of deferring for a retry that + // would re-run the same extraction. + echoDropped += 1; + stats.skipped += 1; this.log( `memory-pro: smart-extractor: manual-echo guard dropped candidate (near-identical to a recent manual store) category=${candidate.category} abstract=${JSON.stringify(candidate.abstract.slice(0, 120))}`, ); @@ -751,6 +758,9 @@ export class SmartExtractor { if (candidates.length === 0) { this.log("memory-pro: smart-extractor: no memories extracted"); + if (echoDropped > 0) { + stats.settledOutcomes = true; + } if (extraction.status === "empty_input") { // No LLM call was made, so the caller's rate limiter must not be charged. stats.skippedNoInput = true; diff --git a/src/tools.ts b/src/tools.ts index 1221eab8..841dc2c0 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -1904,8 +1904,16 @@ export function registerMemoryForgetTool( details: resolved.details ?? { error: "not_found", id: memoryId }, }; } + const forgottenRow = context.manualEchoLedger + ? await context.store.getById(resolved.id, scopeFilter).catch(() => null) + : null; const deleted = await context.store.delete(resolved.id, scopeFilter); if (deleted) { + // A forgotten fact must not keep suppressing its own + // re-statement through the echo ledger. + if (forgottenRow?.text) { + context.manualEchoLedger?.invalidate(agentId, forgottenRow.text); + } context.onMemoriesDeleted?.({ scopeFilter }); return { content: [ @@ -1948,6 +1956,7 @@ export function registerMemoryForgetTool( scopeFilter, ); if (deleted) { + context.manualEchoLedger?.invalidate(agentId, results[0].entry.text); context.onMemoriesDeleted?.({ scopeFilter }); return { content: [ @@ -2179,6 +2188,11 @@ export function registerMemoryUpdateTool( ); } + // The superseding write succeeded: this text will echo through + // the same turn's auto-capture extraction exactly like a plain + // manual store, so it must be recorded on this early-return + // path too. + context.manualEchoLedger?.record(agentId, text); return { content: [ { diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs index b7e767f6..ccca3e84 100644 --- a/test/manual-echo-guard.test.mjs +++ b/test/manual-echo-guard.test.mjs @@ -3,31 +3,45 @@ * candidates that near-duplicate a recent manual memory_store/memory_update * text. * - * Mechanism (design ruling 2026-07-21): when the user dictates a memory - * ("remember this: ..."), the same sentence flows through BOTH the manual - * store lane and auto-capture extraction, minting near-twin rows the dedup - * layer cannot reliably collide (fresh-row vector visibility, category - * splits). The guard keeps an in-memory per-agent ring of recent manual - * texts and drops near-identical extraction candidates BEFORE the admission - * judge — string-only comparison, no LLM, no vector search. + * Mechanism (design ruling 2026-07-21, tightened in review): when the user + * dictates a memory ("remember this: ..."), the same sentence flows through + * BOTH the manual store lane and auto-capture extraction, minting near-twin + * rows the dedup layer cannot reliably collide. The guard keeps a per-agent + * ring of recent manual texts and drops near-identical extraction candidates + * BEFORE the admission judge — string-only comparison, no LLM, no vector + * search. * - * Match test: normalized containment (either direction, min token guard) or - * token-set Jaccard >= 0.75 on the candidate content. + * Match test is ONE-SIDED and conservative: exact match, the manual text + * containing the candidate, or every candidate content token (after glue-word + * stripping) already present in the manual text. A candidate carrying ANY + * extra content — negation, changed value, temporal qualifier, added facts — + * always survives. Entries expire (TTL), are consumed on match, and are + * invalidated when their memory is forgotten. * * Fixtures are entirely synthetic; no real fleet data. */ -import { describe, it } from "node:test"; +import { describe, it, beforeEach, afterEach } from "node:test"; import assert from "node:assert/strict"; +import http from "node:http"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; import jitiFactory from "jiti"; -const jiti = jitiFactory(import.meta.url, { interopDefault: true }); +const testDir = path.dirname(fileURLToPath(import.meta.url)); +const pluginSdkStubPath = path.resolve(testDir, "helpers", "openclaw-plugin-sdk-stub.mjs"); +const jiti = jitiFactory(import.meta.url, { + interopDefault: true, + alias: { "openclaw/plugin-sdk": pluginSdkStubPath }, +}); const { ManualEchoLedger, isNearIdenticalEcho, normalizeEchoText, - MANUAL_ECHO_JACCARD_THRESHOLD, MANUAL_ECHO_RING_SIZE, + MANUAL_ECHO_TTL_MS, } = jiti("../src/manual-echo-guard.ts"); describe("normalizeEchoText", () => { @@ -50,7 +64,7 @@ describe("isNearIdenticalEcho", () => { assert.equal(isNearIdenticalEcho(manual, manual), true); }); - it("matches when the candidate contains the manual text", () => { + it("matches the sentence-wrapped echo (glue words only added)", () => { assert.equal( isNearIdenticalEcho( `User stated that the office plant needs watering every Friday.`, @@ -67,23 +81,73 @@ describe("isNearIdenticalEcho", () => { ); }); - it("matches high token overlap above the Jaccard threshold", () => { + it("matches the sentence-wrapped echo shape via token-subset containment", () => { + assert.equal( + isNearIdenticalEcho( + "User's favorite teacup is the red one", + "favorite teacup: the red one", + ), + true, + ); + }); + + it("keeps a candidate that adds a new content token (no fuzzy overlap match)", () => { assert.equal( isNearIdenticalEcho( "office plant needs deep watering every friday morning", "the office plant needs watering every friday morning", ), - true, + false, ); }); - it("matches the sentence-wrapped echo shape via token-subset containment", () => { + it("keeps a negated candidate (correction, not echo)", () => { assert.equal( isNearIdenticalEcho( - "User's favorite teacup is the red one", - "favorite teacup: the red one", + "alice no longer works at acme", + "alice works at acme", ), - true, + false, + ); + }); + + it("keeps a positive candidate against a negated manual text", () => { + assert.equal( + isNearIdenticalEcho( + "alice works at acme", + "alice does not work at acme anymore", + ), + false, + ); + }); + + it("keeps a temporally qualified candidate", () => { + assert.equal( + isNearIdenticalEcho( + "alice works at acme until friday", + "alice works at acme", + ), + false, + ); + }); + + it("keeps a changed-value candidate", () => { + assert.equal( + isNearIdenticalEcho( + "alice works at initech", + "alice works at acme", + ), + false, + ); + }); + + it("keeps a candidate carrying additional facts", () => { + assert.equal( + isNearIdenticalEcho( + "alice works at acme and volunteers at the animal shelter", + "alice works at acme", + ), + false, ); }); @@ -112,8 +176,28 @@ describe("isNearIdenticalEcho", () => { ); }); - it("exposes the documented threshold", () => { - assert.equal(MANUAL_ECHO_JACCARD_THRESHOLD, 0.75); + describe("CJK fallback", () => { + const manualCjk = "最喜欢的茶杯是红色的那个"; + + it("matches an exact CJK echo", () => { + assert.equal(isNearIdenticalEcho(manualCjk, manualCjk), true); + }); + + it("matches a wrapped CJK echo (small glue, no markers)", () => { + assert.equal(isNearIdenticalEcho(`用户说${manualCjk}`, manualCjk), true); + }); + + it("matches a shortened CJK echo", () => { + assert.equal(isNearIdenticalEcho("最喜欢的茶杯是红色", manualCjk), true); + }); + + it("keeps a temporally qualified CJK candidate", () => { + assert.equal(isNearIdenticalEcho(`${manualCjk}直到周五`, manualCjk), false); + }); + + it("keeps a negated CJK candidate", () => { + assert.equal(isNearIdenticalEcho(`不再${manualCjk}`, manualCjk), false); + }); }); }); @@ -141,6 +225,43 @@ describe("ManualEchoLedger", () => { assert.ok(ledger.match(undefined, "the kneeling chair height is 104cm")); }); + it("consumes an entry on match: one manual store suppresses one echo", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + assert.ok(ledger.match("agent-one", "favorite teacup: the red one")); + assert.equal( + ledger.match("agent-one", "favorite teacup: the red one"), + null, + "a second identical statement is a deliberate re-assertion, not an echo", + ); + }); + + it("expires entries after the TTL", () => { + const ledger = new ManualEchoLedger(); + const t0 = 1_000_000; + ledger.record("agent-one", "standing desk height is 112cm", t0); + assert.ok( + ledger.match("agent-one", "standing desk height is 112cm", t0 + MANUAL_ECHO_TTL_MS - 1), + ); + ledger.record("agent-one", "standing desk height is 112cm", t0); + assert.equal( + ledger.match("agent-one", "standing desk height is 112cm", t0 + MANUAL_ECHO_TTL_MS + 1), + null, + "an expired manual text must not suppress a later re-statement", + ); + }); + + it("invalidate() drops the forgotten text but keeps others", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + ledger.record("agent-one", "the office plant needs watering every friday"); + ledger.invalidate("agent-one", "Favorite Teacup: the RED one"); + assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); + assert.ok( + ledger.match("agent-one", "the office plant needs watering every friday"), + ); + }); + it("caps the ring and evicts the oldest entry", () => { const ledger = new ManualEchoLedger(); ledger.record("agent-one", "the very first manual fact about topic zero"); @@ -171,3 +292,251 @@ describe("ManualEchoLedger", () => { assert.ok(ledger.match("agent-two", "favorite teacup: the red one")); }); }); + +describe("echo guard through the full auto-capture path", () => { + const EMBED_DIMS = 64; + let workspaceDir; + let embeddingServer; + let llmServer; + let extractionPrompts; + let llmEchoText; + + function hashToIndex(text, dims) { + let h = 0; + for (let i = 0; i < text.length; i++) h = (h * 31 + text.charCodeAt(i)) >>> 0; + return h % dims; + } + + beforeEach(async () => { + workspaceDir = mkdtempSync(path.join(tmpdir(), "manual-echo-e2e-")); + extractionPrompts = []; + llmEchoText = null; + embeddingServer = http.createServer(async (req, res) => { + const chunks = []; + for await (const chunk of req) chunks.push(chunk); + const payload = JSON.parse(Buffer.concat(chunks).toString("utf8")); + const inputs = Array.isArray(payload.input) ? payload.input : [payload.input]; + res.writeHead(200, { "Content-Type": "application/json" }); + res.end(JSON.stringify({ + object: "list", + data: inputs.map((input, index) => { + const v = new Array(EMBED_DIMS).fill(0); + v[hashToIndex(String(input), EMBED_DIMS)] = 1; + return { object: "embedding", index, embedding: v }; + }), + model: "mock-embedding-model", + usage: { prompt_tokens: 0, total_tokens: 0 }, + })); + }); + llmServer = http.createServer(async (req, res) => { + const chunks = []; + for await (const chunk of req) chunks.push(chunk); + const payload = JSON.parse(Buffer.concat(chunks).toString("utf8")); + const prompt = String(payload.messages?.map((m) => m.content).join("\n") ?? ""); + if (prompt.includes("## Recent Conversation")) extractionPrompts.push(prompt); + res.writeHead(200, { "Content-Type": "application/json" }); + res.end(JSON.stringify({ + id: "chatcmpl-test", object: "chat.completion", created: 1, model: "mock-memory-model", + choices: [{ + index: 0, finish_reason: "stop", + message: { + role: "assistant", + content: JSON.stringify({ + memories: llmEchoText + ? [{ + category: "preferences", + abstract: "echo of the manual text", + overview: `## Preference\n- ${llmEchoText}`, + content: llmEchoText, + }] + : [], + }), + }, + }], + })); + }); + await new Promise((resolve) => embeddingServer.listen(0, "127.0.0.1", resolve)); + await new Promise((resolve) => llmServer.listen(0, "127.0.0.1", resolve)); + }); + + afterEach(async () => { + await new Promise((resolve) => embeddingServer.close(resolve)); + await new Promise((resolve) => llmServer.close(resolve)); + rmSync(workspaceDir, { recursive: true, force: true }); + }); + + function registerPlugin() { + const pluginModule = jiti("../index.ts"); + const plugin = pluginModule.default || pluginModule; + (pluginModule.resetRegistration ?? (() => {}))(); + const eventHandlers = new Map(); + const tools = new Map(); + const logs = { info: [], warn: [], debug: [] }; + const api = { + pluginConfig: { + dbPath: path.join(workspaceDir, "memory-db"), + autoCapture: true, + autoRecall: false, + smartExtraction: true, + extractMinMessages: 2, + extractionThrottle: { skipLowValue: false, maxExtractionsPerHour: 200 }, + sessionCompression: { enabled: false }, + selfImprovement: { enabled: false, beforeResetNote: false, ensureLearningFiles: false }, + embedding: { + apiKey: "test-key", model: "mock-embedding-model", + baseURL: `http://127.0.0.1:${embeddingServer.address().port}/v1`, + dimensions: EMBED_DIMS, + }, + llm: { + apiKey: "test-key", model: "mock-memory-model", + baseURL: `http://127.0.0.1:${llmServer.address().port}`, + }, + }, + resolvePath(target) { + if (typeof target !== "string") return target; + if (path.isAbsolute(target)) return target; + return path.join(workspaceDir, target); + }, + logger: { + info(m) { logs.info.push(String(m)); }, + warn(m) { logs.warn.push(String(m)); }, + debug(m) { logs.debug.push(String(m)); }, + }, + registerTool(factory, meta) { + const name = meta?.name; + if (name) tools.set(name, factory); + }, + registerCli() {}, + registerService() {}, + on(eventName, handler, meta) { + const list = eventHandlers.get(eventName) || []; + list.push({ handler, meta }); + eventHandlers.set(eventName, list); + }, + registerHook(eventName, handler, opts) { + const list = eventHandlers.get(eventName) || []; + list.push({ handler, meta: opts }); + eventHandlers.set(eventName, list); + }, + }; + plugin.register(api); + const hooks = eventHandlers.get("agent_end") || []; + assert.ok(hooks.length >= 1, "expected an agent_end handler"); + const getTool = (name) => { + const factory = tools.get(name); + assert.ok(factory, `tool ${name} should be registered`); + return factory({}); + }; + return { hook: hooks[0].handler, getTool, logs }; + } + + async function fireAgentEnd(hook, messages, ctx) { + hook({ success: true, messages }, ctx); + const run = hook.__lastRun; + assert.ok(run && typeof run.then === "function", "expected a background capture run"); + await run; + } + + it("settles an echo-only batch instead of deferring it for a retry", async () => { + const { hook, getTool, logs } = registerPlugin(); + const MANUAL = "preferred rehearsal room is the basement studio"; + llmEchoText = `User stated the preferred rehearsal room is the basement studio`; + + const store = getTool("memory_store"); + await store.execute("call-1", { text: MANUAL }); + + const ctx = { sessionKey: "agent:main:main", agentId: "main" }; + await fireAgentEnd(hook, [ + { role: "user", content: `remember this: ${MANUAL}` }, + { role: "assistant", content: "stored it" }, + { role: "user", content: "thanks, noted for the band" }, + ], ctx); + + assert.equal(extractionPrompts.length, 1, "extraction should run once"); + assert.ok( + logs.info.some((line) => line.includes("settled with no persisted rows")), + `an echo-only batch must settle (consume its input), not defer for retry; got info logs: ${logs.info.join(" | ")}`, + ); + + llmEchoText = null; + await fireAgentEnd(hook, [ + { role: "user", content: `remember this: ${MANUAL}` }, + { role: "assistant", content: "stored it" }, + { role: "user", content: "thanks, noted for the band" }, + { role: "user", content: "also the amp cables live in the gray tote" }, + ], ctx); + const repeatedEcho = extractionPrompts + .slice(1) + .filter((p) => p.includes(MANUAL) && p.includes("remember this")).length; + assert.ok( + extractionPrompts.length >= 1, + `follow-up state sanity (${repeatedEcho} echo re-runs observed)`, + ); + }); + + it("records the superseding text on the temporal memory_update path (handler level)", async () => { + const { registerAllMemoryTools } = jiti("../src/tools.ts"); + const ledger = new ManualEchoLedger(); + const EXISTING_ID = "11111111-2222-4333-8444-555555555555"; + const ORIGINAL = "favorite rehearsal drink: sparkling water"; + const UPDATED = "favorite rehearsal drink: mint tea"; + const existingEntry = { + id: EXISTING_ID, + text: ORIGINAL, + category: "preference", + scope: "agent:main", + importance: 0.7, + timestamp: Date.now() - 60_000, + metadata: JSON.stringify({ + memory_category: "preferences", + l0_abstract: ORIGINAL, + l1_overview: `- ${ORIGINAL}`, + l2_content: ORIGINAL, + source: "manual", + state: "confirmed", + }), + }; + const context = { + agentId: "main", + workspaceDir: workspaceDir, + mdMirror: null, + manualEchoLedger: ledger, + scopeManager: { + getAccessibleScopes: (agentId) => ["global", `agent:${agentId}`], + getScopeFilter: (agentId) => ["global", `agent:${agentId}`], + isAccessible: (scope, agentId) => ["global", `agent:${agentId}`].includes(scope), + getDefaultScope: (agentId) => `agent:${agentId}`, + }, + retriever: { getConfig() { return { mode: "hybrid" }; } }, + store: { + async getById(id) { return id === EXISTING_ID ? existingEntry : null; }, + async vectorSearch() { return []; }, + async list() { return [existingEntry]; }, + async listFactKeyCandidates() { return [existingEntry]; }, + async store(entry) { return { ...entry, id: "99999999-8888-4777-8666-555555555554", timestamp: Date.now() }; }, + async update() { return existingEntry; }, + }, + embedder: { async embedPassage() { return [0.1, 0.2, 0.3]; } }, + }; + const creators = new Map(); + registerAllMemoryTools( + { + registerTool(factory, meta) { creators.set(meta.name, factory); }, + logger: { info() {}, warn() {}, debug() {} }, + }, + context, + { enableManagementTools: true }, + ); + const update = creators.get("memory_update")({}); + const updated = await update.execute(null, { memoryId: EXISTING_ID, text: UPDATED }); + assert.equal( + updated?.details?.action, + "superseded", + `the preferences update must take the temporal supersede path (got ${JSON.stringify(updated?.details)})`, + ); + assert.ok( + ledger.match("main", UPDATED), + "the successful temporal supersede must record the NEW text in the echo ledger before its early return", + ); + }); +}); From feae22ada25eb51fbd78398a42f85297478d36a4 Mon Sep 17 00:00:00 2001 From: Gorkem Date: Tue, 1 Sep 2026 12:33:19 +0300 Subject: [PATCH 3/6] fix(echo-guard): persist consume with multiple live entries, substantive-residual CJK checks, order-preserving containment, deletion-lane invalidation Round-2 review: 1. match() now persists the post-splice array in every outcome: with two or more live entries the consume mutated a detached filter() copy while the Map kept the matched entry, so one manual store could suppress repeated re-statements for its whole TTL. Regression: two live entries, the same candidate matched twice consumes exactly one entry. 2. CJK residuals must be proven non-substantive: the shortened branch rejects a match when the REMOVED manual residual carries a marker (stripping the negated wrapper yields the opposite claim), and the wrapped branch accepts only residuals composed entirely of known reporting-glue fragments (short and marker-free was not enough: a three-character residual can be a new fact). Both review cases covered exactly. 3. English containment is order-preserving: candidate content tokens must appear in the manual text as an ordered subsequence, so reversed relationships (alice reports to bob vs bob reports to alice) survive. 4. Every deletion lane invalidates the ledger: CLI delete fetches the row pre-delete and invalidates its text across all agent buckets, CLI delete-bulk clears the ledger wholesale (fail-open), and the ledger is wired into contexts only while smart extraction is enabled, which also spares memory_forget its pre-delete getById fetch when disabled. --- cli.ts | 18 +++++ dist/cli.js | 14 ++++ dist/index.js | 6 +- dist/src/manual-echo-guard.js | 123 +++++++++++++++++++++++++------- index.ts | 6 +- src/manual-echo-guard.ts | 121 +++++++++++++++++++++++++------ test/manual-echo-guard.test.mjs | 65 +++++++++++++++++ 7 files changed, 304 insertions(+), 49 deletions(-) diff --git a/cli.ts b/cli.ts index 0b60a3f4..81c3ce44 100644 --- a/cli.ts +++ b/cli.ts @@ -44,6 +44,10 @@ interface CLIContext { // memories, so the host plugin can invalidate any in-process read caches (e.g. // reflection slice caches) keyed on the affected scope(s) before they go stale. onMemoriesDeleted?: (info: { scopeFilter?: string[] }) => void; + // Manual-store echo ledger (present only while smart extraction is enabled): + // the deletion lanes must drop a deleted row's suppression state, so a + // forgotten fact can never keep suppressing its own re-statement. + manualEchoLedger?: import("./src/manual-echo-guard.js").ManualEchoLedger; oauthTestHooks?: { openUrl?: (url: string) => void | Promise; authorizeUrl?: (url: string) => void | Promise; @@ -1477,9 +1481,19 @@ export function registerMemoryCLI(program: Command, context: CLIContext): void { scopeFilter = [options.scope]; } + // Fetched before the delete (afterwards the row is gone), only when a + // ledger is wired: the deleted text's echo suppression must not + // outlive the row (targeted invalidation across every agent bucket, + // since the CLI cannot name the writing agent). + const deletedRow = context.manualEchoLedger + ? await context.store.getById(id, scopeFilter).catch(() => null) + : null; const deleted = await context.store.delete(id, scopeFilter); if (deleted) { + if (deletedRow?.text) { + context.manualEchoLedger?.invalidateEverywhere(deletedRow.text); + } context.onMemoriesDeleted?.({ scopeFilter }); console.log(`Memory ${id} deleted successfully.`); printReadConsistencyHint(context.store); @@ -1527,6 +1541,10 @@ export function registerMemoryCLI(program: Command, context: CLIContext): void { } else { const deletedCount = await context.store.bulkDelete(options.scope, beforeTimestamp); if (deletedCount > 0) { + // Pre-fetching every deleted row's text would defeat the point of + // a bulk delete; clearing the whole ledger is fail-open (worst + // case: one uncaught echo lands as a duplicate for dedup). + context.manualEchoLedger?.clearAll(); context.onMemoriesDeleted?.({ scopeFilter: options.scope }); } console.log(`Deleted ${deletedCount} memories.`); diff --git a/dist/cli.js b/dist/cli.js index ec05e9e1..10ef4af9 100644 --- a/dist/cli.js +++ b/dist/cli.js @@ -1199,8 +1199,18 @@ export function registerMemoryCLI(program, context) { if (options.scope) { scopeFilter = [options.scope]; } + // Fetched before the delete (afterwards the row is gone), only when a + // ledger is wired: the deleted text's echo suppression must not + // outlive the row (targeted invalidation across every agent bucket, + // since the CLI cannot name the writing agent). + const deletedRow = context.manualEchoLedger + ? await context.store.getById(id, scopeFilter).catch(() => null) + : null; const deleted = await context.store.delete(id, scopeFilter); if (deleted) { + if (deletedRow?.text) { + context.manualEchoLedger?.invalidateEverywhere(deletedRow.text); + } context.onMemoriesDeleted?.({ scopeFilter }); console.log(`Memory ${id} deleted successfully.`); printReadConsistencyHint(context.store); @@ -1247,6 +1257,10 @@ export function registerMemoryCLI(program, context) { else { const deletedCount = await context.store.bulkDelete(options.scope, beforeTimestamp); if (deletedCount > 0) { + // Pre-fetching every deleted row's text would defeat the point of + // a bulk delete; clearing the whole ledger is fail-open (worst + // case: one uncaught echo lands as a duplicate for dedup). + context.manualEchoLedger?.clearAll(); context.onMemoriesDeleted?.({ scopeFilter: options.scope }); } console.log(`Deleted ${deletedCount} memories.`); diff --git a/dist/index.js b/dist/index.js index 6c3e47d1..f4ab5a8d 100644 --- a/dist/index.js +++ b/dist/index.js @@ -2621,7 +2621,10 @@ const memoryLanceDBProPlugin = { workspaceBoundary: config.workspaceBoundary, selfImprovementMaxEntries: config.selfImprovement?.maxEntries, manualStoreSupersede: config.manualStoreSupersede === true, - manualEchoLedger, + // The echo ledger only ever matters when smart extraction can echo a + // manual store back; leaving it out otherwise also spares + // memory_forget its pre-delete getById fetch. + manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, // Mirrors the CLI context wiring below: keep in-process reflection caches // consistent after a live memory_forget delete too, not just CLI delete/delete-bulk. onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), @@ -2664,6 +2667,7 @@ const memoryLanceDBProPlugin = { store, retriever, scopeManager, + manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), migrator, embedder, diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js index 1cedb39b..911d3802 100644 --- a/dist/src/manual-echo-guard.js +++ b/dist/src/manual-echo-guard.js @@ -76,17 +76,20 @@ export function normalizeEchoText(text) { function tokenList(normalized) { return normalized.split(" ").filter((t) => t.length > 0); } -function contentTokens(normalized) { - const out = new Set(); +function orderedContentTokens(normalized) { + const out = []; for (const token of tokenList(normalized)) { if (token.length <= 1 && !CJK_RE.test(token)) continue; if (ECHO_STOPWORDS.has(token)) continue; - out.add(token); + out.push(token); } return out; } +function contentTokens(normalized) { + return new Set(orderedContentTokens(normalized)); +} function markerAsymmetry(aTokens, bTokens) { const a = new Set(aTokens); const b = new Set(bTokens); @@ -99,6 +102,30 @@ function markerAsymmetry(aTokens, bTokens) { function containsCjkMarker(fragment) { return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); } +/** + * The ONLY residual a wrapped CJK echo may carry: reporting glue and + * particles. Anything else in the residual — a marker, a verb, a new fact + * like 并养猫 — is substantive content, and "short and marker-free" was + * provably not enough to exclude it. + */ +const CJK_WRAPPER_GLUE_FRAGMENTS = [ + "用户说", "用户", "说过", "说", "提到", "表示", "了", "的", "是", + "ユーザーは", "ユーザー", "と言った", "と言いました", "です", "ます", + "사용자는", "사용자", "라고", "입니다", +]; +function isCjkGlueOnly(residual) { + let rest = residual; + for (let pass = 0; pass < 8 && rest.length > 0; pass++) { + const before = rest; + for (const glue of CJK_WRAPPER_GLUE_FRAGMENTS) { + while (rest.includes(glue)) + rest = rest.replace(glue, ""); + } + if (rest === before) + break; + } + return rest.length === 0; +} function isCjkEcho(candidate, manual) { const cand = candidate.replace(/\s+/g, ""); const man = manual.replace(/\s+/g, ""); @@ -106,17 +133,20 @@ function isCjkEcho(candidate, manual) { return false; if (cand === man) return true; - // Shortened echo: the candidate re-states a piece of the manual text and - // adds nothing. Length-gated so tiny fragments cannot over-match. - if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) - return true; - // Wrapped echo: the candidate is the manual text plus a small amount of - // glue ("用户说…"). Allowed only when the residual is short AND carries no - // negation/temporal marker, so a qualified or corrected statement - // ("…直到周五", "不再…") is never treated as an echo. + // Shortened echo: the candidate re-states a piece of the manual text. The + // REMOVED part must carry no marker: stripping 用户不 off 用户不喜欢喝茶和咖啡 + // yields the OPPOSITE claim, not an echo of it. + if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) { + const removed = man.replace(cand, ""); + return !containsCjkMarker(removed); + } + // Wrapped echo: the candidate is the manual text plus reporting glue + // (用户说…). The residual must consist ONLY of known glue fragments — + // being short and marker-free is not enough, since a three-character + // residual can be a brand-new fact (并养猫). if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { const residual = cand.replace(man, ""); - return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && !containsCjkMarker(residual); + return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && isCjkGlueOnly(residual); } return false; } @@ -146,16 +176,29 @@ export function isNearIdenticalEcho(candidateText, manualText) { return true; // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red - // one"). After glue-word stripping, EVERY candidate content token must - // already be in the manual text — one-sided by design, so a candidate + // one"). After glue-word stripping, the candidate's content tokens must + // appear in the manual text IN THE SAME RELATIVE ORDER (an ordered + // subsequence, not a bag-of-words subset): set membership alone would + // collapse "alice reports to bob" onto "bob reports to alice" and discard + // the reversed relationship. One-sided by design either way — a candidate // with any extra content token (changed value, qualifier, added fact) is // new information and survives. - const candidateContent = contentTokens(candidate); - if (candidateContent.size === 0) + const candidateContentList = orderedContentTokens(candidate); + if (candidateContentList.length === 0) return false; - for (const token of candidateContent) { - if (!manualContent.has(token)) + const manualContentList = orderedContentTokens(manual); + let cursor = 0; + for (const token of candidateContentList) { + let found = -1; + for (let i = cursor; i < manualContentList.length; i++) { + if (manualContentList[i] === token) { + found = i; + break; + } + } + if (found < 0) return false; + cursor = found + 1; } return true; } @@ -189,22 +232,28 @@ export class ManualEchoLedger { const ring = this.byAgent.get(key); if (!ring || ring.length === 0) return null; + // `live` is a fresh array, so every outcome below must PERSIST it: an + // in-place splice of an unpersisted copy would leave the Map holding the + // matched entry and let one manual store suppress repeated re-statements + // for its whole TTL (review round 2, finding 1). const live = ring.filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); - if (live.length !== ring.length) { - if (live.length === 0) { - this.byAgent.delete(key); - return null; - } - this.byAgent.set(key, live); + if (live.length === 0) { + this.byAgent.delete(key); + return null; } for (let i = live.length - 1; i >= 0; i--) { if (isNearIdenticalEcho(candidateText, live[i].text)) { const [hit] = live.splice(i, 1); if (live.length === 0) this.byAgent.delete(key); + else + this.byAgent.set(key, live); return hit.text; } } + if (live.length !== ring.length) { + this.byAgent.set(key, live); + } return null; } /** @@ -230,4 +279,30 @@ export class ManualEchoLedger { clear(agentId) { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } + /** + * Deletion-lane invalidation when the deleter cannot name the writing + * agent (CLI delete by id): the fact is gone from the store, so no bucket + * may keep suppressing its re-statement. + */ + invalidateEverywhere(text) { + if (typeof text !== "string" || text.trim().length === 0) + return; + const target = normalizeEchoText(text); + for (const [key, ring] of [...this.byAgent.entries()]) { + const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); + if (kept.length === 0) + this.byAgent.delete(key); + else if (kept.length !== ring.length) + this.byAgent.set(key, kept); + } + } + /** + * Wholesale reset for bulk deletion lanes, where pre-fetching every + * deleted row's text would defeat the point of a bulk delete. Clearing is + * fail-open: the worst case is one uncaught echo (a duplicate row for + * dedup), never a lost memory. + */ + clearAll() { + this.byAgent.clear(); + } } diff --git a/index.ts b/index.ts index ee733640..8082a355 100644 --- a/index.ts +++ b/index.ts @@ -3502,7 +3502,10 @@ const memoryLanceDBProPlugin = { workspaceBoundary: config.workspaceBoundary, selfImprovementMaxEntries: config.selfImprovement?.maxEntries, manualStoreSupersede: config.manualStoreSupersede === true, - manualEchoLedger, + // The echo ledger only ever matters when smart extraction can echo a + // manual store back; leaving it out otherwise also spares + // memory_forget its pre-delete getById fetch. + manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, // Mirrors the CLI context wiring below: keep in-process reflection caches // consistent after a live memory_forget delete too, not just CLI delete/delete-bulk. onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), @@ -3556,6 +3559,7 @@ const memoryLanceDBProPlugin = { store, retriever, scopeManager, + manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), migrator, embedder, diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts index f75d8f79..b828d8e3 100644 --- a/src/manual-echo-guard.ts +++ b/src/manual-echo-guard.ts @@ -84,16 +84,20 @@ function tokenList(normalized: string): string[] { return normalized.split(" ").filter((t) => t.length > 0); } -function contentTokens(normalized: string): Set { - const out = new Set(); +function orderedContentTokens(normalized: string): string[] { + const out: string[] = []; for (const token of tokenList(normalized)) { if (token.length <= 1 && !CJK_RE.test(token)) continue; if (ECHO_STOPWORDS.has(token)) continue; - out.add(token); + out.push(token); } return out; } +function contentTokens(normalized: string): Set { + return new Set(orderedContentTokens(normalized)); +} + function markerAsymmetry(aTokens: string[], bTokens: string[]): boolean { const a = new Set(aTokens); const b = new Set(bTokens); @@ -107,21 +111,49 @@ function containsCjkMarker(fragment: string): boolean { return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); } +/** + * The ONLY residual a wrapped CJK echo may carry: reporting glue and + * particles. Anything else in the residual — a marker, a verb, a new fact + * like 并养猫 — is substantive content, and "short and marker-free" was + * provably not enough to exclude it. + */ +const CJK_WRAPPER_GLUE_FRAGMENTS = [ + "用户说", "用户", "说过", "说", "提到", "表示", "了", "的", "是", + "ユーザーは", "ユーザー", "と言った", "と言いました", "です", "ます", + "사용자는", "사용자", "라고", "입니다", +]; + +function isCjkGlueOnly(residual: string): boolean { + let rest = residual; + for (let pass = 0; pass < 8 && rest.length > 0; pass++) { + const before = rest; + for (const glue of CJK_WRAPPER_GLUE_FRAGMENTS) { + while (rest.includes(glue)) rest = rest.replace(glue, ""); + } + if (rest === before) break; + } + return rest.length === 0; +} + function isCjkEcho(candidate: string, manual: string): boolean { const cand = candidate.replace(/\s+/g, ""); const man = manual.replace(/\s+/g, ""); if (cand.length === 0 || man.length === 0) return false; if (cand === man) return true; - // Shortened echo: the candidate re-states a piece of the manual text and - // adds nothing. Length-gated so tiny fragments cannot over-match. - if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) return true; - // Wrapped echo: the candidate is the manual text plus a small amount of - // glue ("用户说…"). Allowed only when the residual is short AND carries no - // negation/temporal marker, so a qualified or corrected statement - // ("…直到周五", "不再…") is never treated as an echo. + // Shortened echo: the candidate re-states a piece of the manual text. The + // REMOVED part must carry no marker: stripping 用户不 off 用户不喜欢喝茶和咖啡 + // yields the OPPOSITE claim, not an echo of it. + if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) { + const removed = man.replace(cand, ""); + return !containsCjkMarker(removed); + } + // Wrapped echo: the candidate is the manual text plus reporting glue + // (用户说…). The residual must consist ONLY of known glue fragments — + // being short and marker-free is not enough, since a three-character + // residual can be a brand-new fact (并养猫). if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { const residual = cand.replace(man, ""); - return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && !containsCjkMarker(residual); + return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && isCjkGlueOnly(residual); } return false; } @@ -152,14 +184,27 @@ export function isNearIdenticalEcho(candidateText: string, manualText: string): // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red - // one"). After glue-word stripping, EVERY candidate content token must - // already be in the manual text — one-sided by design, so a candidate + // one"). After glue-word stripping, the candidate's content tokens must + // appear in the manual text IN THE SAME RELATIVE ORDER (an ordered + // subsequence, not a bag-of-words subset): set membership alone would + // collapse "alice reports to bob" onto "bob reports to alice" and discard + // the reversed relationship. One-sided by design either way — a candidate // with any extra content token (changed value, qualifier, added fact) is // new information and survives. - const candidateContent = contentTokens(candidate); - if (candidateContent.size === 0) return false; - for (const token of candidateContent) { - if (!manualContent.has(token)) return false; + const candidateContentList = orderedContentTokens(candidate); + if (candidateContentList.length === 0) return false; + const manualContentList = orderedContentTokens(manual); + let cursor = 0; + for (const token of candidateContentList) { + let found = -1; + for (let i = cursor; i < manualContentList.length; i++) { + if (manualContentList[i] === token) { + found = i; + break; + } + } + if (found < 0) return false; + cursor = found + 1; } return true; } @@ -197,21 +242,26 @@ export class ManualEchoLedger { const key = agentId?.trim() || DEFAULT_AGENT_BUCKET; const ring = this.byAgent.get(key); if (!ring || ring.length === 0) return null; + // `live` is a fresh array, so every outcome below must PERSIST it: an + // in-place splice of an unpersisted copy would leave the Map holding the + // matched entry and let one manual store suppress repeated re-statements + // for its whole TTL (review round 2, finding 1). const live = ring.filter((e) => now - e.at < MANUAL_ECHO_TTL_MS); - if (live.length !== ring.length) { - if (live.length === 0) { - this.byAgent.delete(key); - return null; - } - this.byAgent.set(key, live); + if (live.length === 0) { + this.byAgent.delete(key); + return null; } for (let i = live.length - 1; i >= 0; i--) { if (isNearIdenticalEcho(candidateText, live[i].text)) { const [hit] = live.splice(i, 1); if (live.length === 0) this.byAgent.delete(key); + else this.byAgent.set(key, live); return hit.text; } } + if (live.length !== ring.length) { + this.byAgent.set(key, live); + } return null; } @@ -236,4 +286,29 @@ export class ManualEchoLedger { clear(agentId: string | undefined): void { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } + + /** + * Deletion-lane invalidation when the deleter cannot name the writing + * agent (CLI delete by id): the fact is gone from the store, so no bucket + * may keep suppressing its re-statement. + */ + invalidateEverywhere(text: string): void { + if (typeof text !== "string" || text.trim().length === 0) return; + const target = normalizeEchoText(text); + for (const [key, ring] of [...this.byAgent.entries()]) { + const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); + if (kept.length === 0) this.byAgent.delete(key); + else if (kept.length !== ring.length) this.byAgent.set(key, kept); + } + } + + /** + * Wholesale reset for bulk deletion lanes, where pre-fetching every + * deleted row's text would defeat the point of a bulk delete. Clearing is + * fail-open: the worst case is one uncaught echo (a duplicate row for + * dedup), never a lost memory. + */ + clearAll(): void { + this.byAgent.clear(); + } } diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs index ccca3e84..96cba09d 100644 --- a/test/manual-echo-guard.test.mjs +++ b/test/manual-echo-guard.test.mjs @@ -198,6 +198,22 @@ describe("isNearIdenticalEcho", () => { it("keeps a negated CJK candidate", () => { assert.equal(isNearIdenticalEcho(`不再${manualCjk}`, manualCjk), false); }); + + it("keeps a shortened CJK candidate whose removed residual carried the negation", () => { + assert.equal( + isNearIdenticalEcho("喜欢喝茶和咖啡", "用户不喜欢喝茶和咖啡"), + false, + "stripping the negated wrapper off the manual text yields the OPPOSITE claim, never an echo", + ); + }); + + it("keeps a wrapped CJK candidate whose residual is a new fact, not glue", () => { + assert.equal( + isNearIdenticalEcho("我住在北京市海淀区并养猫", "我住在北京市海淀区"), + false, + "a short marker-free residual can still be a brand-new fact; only known glue may wrap an echo", + ); + }); }); }); @@ -283,6 +299,55 @@ describe("ManualEchoLedger", () => { assert.equal(ledger.match("agent-one", " "), null); }); + it("keeps a reversed relationship (order-preserving containment)", () => { + assert.equal( + isNearIdenticalEcho("Alice reports to Bob", "Bob reports to Alice"), + false, + "bag-of-words containment would collapse the reversed relationship", + ); + assert.equal( + isNearIdenticalEcho("User stated that Alice reports to Bob", "Alice reports to Bob"), + true, + "the same-order wrap echo must still collapse", + ); + }); + + it("consume persists with MULTIPLE live entries: one hit removes exactly the matched entry", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "the office plant needs watering every friday"); + ledger.record("agent-one", "favorite teacup: the red one"); + assert.ok(ledger.match("agent-one", "favorite teacup: the red one"), "first hit consumes the teacup entry"); + assert.equal( + ledger.match("agent-one", "favorite teacup: the red one"), + null, + "the consumed entry must be gone FROM THE MAP, not just from a detached copy", + ); + assert.ok( + ledger.match("agent-one", "the office plant needs watering every friday"), + "the other live entry must survive the first consume", + ); + }); + + it("invalidateEverywhere() drops the text from every agent bucket", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + ledger.record("agent-two", "favorite teacup: the red one"); + ledger.record("agent-two", "the office plant needs watering every friday"); + ledger.invalidateEverywhere("Favorite Teacup: the RED one"); + assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); + assert.equal(ledger.match("agent-two", "favorite teacup: the red one"), null); + assert.ok(ledger.match("agent-two", "the office plant needs watering every friday")); + }); + + it("clearAll() empties every bucket (bulk-delete lane)", () => { + const ledger = new ManualEchoLedger(); + ledger.record("agent-one", "favorite teacup: the red one"); + ledger.record("agent-two", "the office plant needs watering every friday"); + ledger.clearAll(); + assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); + assert.equal(ledger.match("agent-two", "the office plant needs watering every friday"), null); + }); + it("clear() empties one agent's ring only", () => { const ledger = new ManualEchoLedger(); ledger.record("agent-one", "favorite teacup: the red one"); From dbb78cc3d5be31b683c97f2a51062c78bee9d8ce Mon Sep 17 00:00:00 2001 From: Gorkem Date: Wed, 2 Sep 2026 20:38:23 +0300 Subject: [PATCH 4/6] fix(echo-guard): keep semantic predicates as content, require exact content equality for wrap echoes, invalidate replaced text on update and supersede Review round 3 follow-ups: - ECHO_STOPWORDS is reporting glue only; has/have/had, wants/want, likes/like, prefers/prefer are content tokens again, so a candidate that swaps a predicate is never an echo - the wrap-echo path requires the candidate's content tokens to EQUAL the manual text's, in order, instead of forming an ordered subsequence: a candidate that drops distinguishing content ("User prefers Python" against "User prefers Go over Python for backend services") is a different assertion and survives; the manual-side minimum doubles as the candidate floor since equal sequences have equal length - memory_store supersede, memory_update temporal supersede, and the plain memory_update path invalidate the replaced row text in the ledger before recording the new text, so a reversal back to a replaced statement is never treated as an echo of a fact the store no longer holds - the CLI ledger wiring (pre-delete getById, invalidateEverywhere, clearAll) is removed: the CLI runs in its own process and could never reach the gateway's in-memory ledger; the class documents the per-process scope and the TTL-bounded window it implies - the ledger construction comment in index.ts matches the conditional wiring (recording happens only when a smart extractor exists) - regressions: the three predicate/subsequence cases, a predicate-carrying wrap echo that must still collapse, and replaced-text invalidation on all three write paths --- cli.ts | 18 ---- dist/cli.js | 14 --- dist/index.js | 6 +- dist/src/manual-echo-guard.js | 100 ++++++++------------ dist/src/tools.js | 15 ++- index.ts | 6 +- src/manual-echo-guard.ts | 96 ++++++++----------- src/tools.ts | 15 ++- test/manual-echo-guard.test.mjs | 134 +++++++++++++++++++++++---- test/manual-store-supersede.test.mjs | 24 +++++ 10 files changed, 249 insertions(+), 179 deletions(-) diff --git a/cli.ts b/cli.ts index 81c3ce44..0b60a3f4 100644 --- a/cli.ts +++ b/cli.ts @@ -44,10 +44,6 @@ interface CLIContext { // memories, so the host plugin can invalidate any in-process read caches (e.g. // reflection slice caches) keyed on the affected scope(s) before they go stale. onMemoriesDeleted?: (info: { scopeFilter?: string[] }) => void; - // Manual-store echo ledger (present only while smart extraction is enabled): - // the deletion lanes must drop a deleted row's suppression state, so a - // forgotten fact can never keep suppressing its own re-statement. - manualEchoLedger?: import("./src/manual-echo-guard.js").ManualEchoLedger; oauthTestHooks?: { openUrl?: (url: string) => void | Promise; authorizeUrl?: (url: string) => void | Promise; @@ -1481,19 +1477,9 @@ export function registerMemoryCLI(program: Command, context: CLIContext): void { scopeFilter = [options.scope]; } - // Fetched before the delete (afterwards the row is gone), only when a - // ledger is wired: the deleted text's echo suppression must not - // outlive the row (targeted invalidation across every agent bucket, - // since the CLI cannot name the writing agent). - const deletedRow = context.manualEchoLedger - ? await context.store.getById(id, scopeFilter).catch(() => null) - : null; const deleted = await context.store.delete(id, scopeFilter); if (deleted) { - if (deletedRow?.text) { - context.manualEchoLedger?.invalidateEverywhere(deletedRow.text); - } context.onMemoriesDeleted?.({ scopeFilter }); console.log(`Memory ${id} deleted successfully.`); printReadConsistencyHint(context.store); @@ -1541,10 +1527,6 @@ export function registerMemoryCLI(program: Command, context: CLIContext): void { } else { const deletedCount = await context.store.bulkDelete(options.scope, beforeTimestamp); if (deletedCount > 0) { - // Pre-fetching every deleted row's text would defeat the point of - // a bulk delete; clearing the whole ledger is fail-open (worst - // case: one uncaught echo lands as a duplicate for dedup). - context.manualEchoLedger?.clearAll(); context.onMemoriesDeleted?.({ scopeFilter: options.scope }); } console.log(`Deleted ${deletedCount} memories.`); diff --git a/dist/cli.js b/dist/cli.js index 10ef4af9..ec05e9e1 100644 --- a/dist/cli.js +++ b/dist/cli.js @@ -1199,18 +1199,8 @@ export function registerMemoryCLI(program, context) { if (options.scope) { scopeFilter = [options.scope]; } - // Fetched before the delete (afterwards the row is gone), only when a - // ledger is wired: the deleted text's echo suppression must not - // outlive the row (targeted invalidation across every agent bucket, - // since the CLI cannot name the writing agent). - const deletedRow = context.manualEchoLedger - ? await context.store.getById(id, scopeFilter).catch(() => null) - : null; const deleted = await context.store.delete(id, scopeFilter); if (deleted) { - if (deletedRow?.text) { - context.manualEchoLedger?.invalidateEverywhere(deletedRow.text); - } context.onMemoriesDeleted?.({ scopeFilter }); console.log(`Memory ${id} deleted successfully.`); printReadConsistencyHint(context.store); @@ -1257,10 +1247,6 @@ export function registerMemoryCLI(program, context) { else { const deletedCount = await context.store.bulkDelete(options.scope, beforeTimestamp); if (deletedCount > 0) { - // Pre-fetching every deleted row's text would defeat the point of - // a bulk delete; clearing the whole ledger is fail-open (worst - // case: one uncaught echo lands as a duplicate for dedup). - context.manualEchoLedger?.clearAll(); context.onMemoriesDeleted?.({ scopeFilter: options.scope }); } console.log(`Deleted ${deletedCount} memories.`); diff --git a/dist/index.js b/dist/index.js index f4ab5a8d..1b265f45 100644 --- a/dist/index.js +++ b/dist/index.js @@ -1991,8 +1991,9 @@ function _initPluginState(api) { // its own. let smartExtractor = null; // Echo guard: shared between the manual store/update tools (record side) - // and the smart extractor (drop side); lives here so the tools keep - // recording even when smart extraction is disabled. + // and the smart extractor (drop side). Constructed unconditionally, but + // wired into the tools only when a smart extractor exists: an echo can + // only arise when extraction is able to re-mint the dictated text. const manualEchoLedger = new ManualEchoLedger(); let admissionController = null; let admissionControllerReflectionLane = null; @@ -2667,7 +2668,6 @@ const memoryLanceDBProPlugin = { store, retriever, scopeManager, - manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), migrator, embedder, diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js index 911d3802..7c1a92a6 100644 --- a/dist/src/manual-echo-guard.js +++ b/dist/src/manual-echo-guard.js @@ -8,25 +8,38 @@ * and drops near-identical extraction candidates BEFORE the admission judge: * deterministic, string-only, no LLM calls, no vector search. * - * Matching is deliberately ONE-SIDED and conservative: a candidate is an - * echo only when it adds NOTHING substantive beyond the recorded manual - * text (exact match, the manual text containing the candidate, or every - * candidate content token already present in the manual text). A candidate + * Matching is deliberately conservative: a candidate is an echo only when + * it asserts NOTHING beyond, and nothing different from, the recorded + * manual text: an exact match, the manual text containing the candidate + * verbatim, or the candidate carrying exactly the manual text's content + * tokens in the same order once reporting glue is stripped. A candidate * carrying extra content — a negation ("no longer"), a changed value, a * temporal qualifier ("until friday"), or additional facts — is new - * information and always survives; the worst case of the guard staying - * quiet is the pre-guard status quo (one duplicate row for dedup). + * information and always survives, and so is one that swaps or drops a + * semantic predicate ("wants" against "has", "prefers Python" against + * "prefers Go over Python"). The worst case of the guard staying quiet is + * the pre-guard status quo (one duplicate row for dedup); the worst case of + * it firing wrongly is a silently lost memory, so every ambiguity resolves + * toward keeping the candidate. * * Entries are short-lived and consumed: each recorded manual text expires * after MANUAL_ECHO_TTL_MS and suppresses at most ONE candidate (the * immediate re-extraction of the same turn). A later identical statement is - * a deliberate user re-assertion, not an echo. + * a deliberate user re-assertion, not an echo. A statement that gets + * replaced (memory_update, supersede) or forgotten is invalidated at once, + * so a reversal back to it is never mistaken for an echo of a fact the + * store no longer holds. * * Scoped per agent (not per session): the store tool and the auto-capture * hook derive their session keys differently, but both resolve the same * agent id, and an echo of ANY recent manual text of the same agent is a * correct drop regardless of session boundaries. TTL + consumption bound * staleness; the ring bounds size. + * + * The ledger is in-memory and per process: only the gateway that recorded + * a text can invalidate it. Deletions made from another process (the + * memory-pro CLI) are not visible here; the TTL bounds that window to + * MANUAL_ECHO_TTL_MS, after which the stale entry expires on its own. */ export const MANUAL_ECHO_RING_SIZE = 8; export const MANUAL_ECHO_TTL_MS = 10 * 60 * 1000; @@ -36,9 +49,12 @@ const MIN_CJK_CONTAINMENT_CHARS = 6; const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; /** - * Glue vocabulary the extractor wraps a dictated fact in ("User stated - * that ..."). Stripped before token containment so the canonical wrap echo - * still collapses; negation and temporal markers are deliberately NOT here. + * Reporting glue the extractor wraps a dictated fact in ("User stated + * that ..."): articles, copulas, pronouns, prepositions, and reporting + * verbs. Stripped before token comparison so the canonical wrap echo still + * collapses. Semantic predicates (has, wants, likes, prefers, ...) are NOT + * glue: they decide what a sentence asserts, so they stay content tokens. + * Negation and temporal markers are deliberately not here either. */ const ECHO_STOPWORDS = new Set([ "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", @@ -46,8 +62,7 @@ const ECHO_STOPWORDS = new Set([ "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", "your", "user", "users", "stated", "said", "says", "saying", "mentioned", - "noted", "prefers", "prefer", "likes", "like", "wants", "want", "has", - "have", "had", "also", + "noted", "also", ]); /** * A marker on exactly one side of the pair means the two texts assert @@ -176,31 +191,20 @@ export function isNearIdenticalEcho(candidateText, manualText) { return true; // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red - // one"). After glue-word stripping, the candidate's content tokens must - // appear in the manual text IN THE SAME RELATIVE ORDER (an ordered - // subsequence, not a bag-of-words subset): set membership alone would - // collapse "alice reports to bob" onto "bob reports to alice" and discard - // the reversed relationship. One-sided by design either way — a candidate - // with any extra content token (changed value, qualifier, added fact) is - // new information and survives. + // one"). After glue-word stripping, the candidate must carry EXACTLY the + // manual text's content tokens, in the same order. Adding a token (a + // changed value, a qualifier, a new fact) is new information; dropping + // one is a different assertion ("User prefers Python" against "User + // prefers Go over Python for backend services" reverses the preference), + // so a subsequence match is not enough. Order matters too: bag-of-words + // equality would collapse "alice reports to bob" onto "bob reports to + // alice". The manual-side minimum above doubles as the candidate floor, + // since equal sequences have equal length. const candidateContentList = orderedContentTokens(candidate); - if (candidateContentList.length === 0) - return false; const manualContentList = orderedContentTokens(manual); - let cursor = 0; - for (const token of candidateContentList) { - let found = -1; - for (let i = cursor; i < manualContentList.length; i++) { - if (manualContentList[i] === token) { - found = i; - break; - } - } - if (found < 0) - return false; - cursor = found + 1; - } - return true; + if (candidateContentList.length !== manualContentList.length) + return false; + return candidateContentList.every((token, i) => token === manualContentList[i]); } export class ManualEchoLedger { byAgent = new Map(); @@ -279,30 +283,4 @@ export class ManualEchoLedger { clear(agentId) { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } - /** - * Deletion-lane invalidation when the deleter cannot name the writing - * agent (CLI delete by id): the fact is gone from the store, so no bucket - * may keep suppressing its re-statement. - */ - invalidateEverywhere(text) { - if (typeof text !== "string" || text.trim().length === 0) - return; - const target = normalizeEchoText(text); - for (const [key, ring] of [...this.byAgent.entries()]) { - const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); - if (kept.length === 0) - this.byAgent.delete(key); - else if (kept.length !== ring.length) - this.byAgent.set(key, kept); - } - } - /** - * Wholesale reset for bulk deletion lanes, where pre-fetching every - * deleted row's text would defeat the point of a bulk delete. Clearing is - * fail-open: the worst case is one uncaught echo (a duplicate row for - * dedup), never a lost memory. - */ - clearAll() { - this.byAgent.clear(); - } } diff --git a/dist/src/tools.js b/dist/src/tools.js index cee386ce..e7c5302a 100644 --- a/dist/src/tools.js +++ b/dist/src/tools.js @@ -1318,6 +1318,14 @@ export function registerMemoryStoreTool(api, context) { // invalidation instead of silently reporting it as superseded. console.warn(`memory-pro: failed to invalidate superseded record ${failure.id.slice(0, 8)}: ${failure.reason}`); } + // The replaced statements leave the echo ledger with the rows: + // a reversal back to a superseded text is new information once + // the store no longer holds that fact. + for (const target of lastDiscovery.targets) { + if (supersededIds.includes(target.entry.id)) { + context.manualEchoLedger?.invalidate(agentId, target.entry.text); + } + } context.manualEchoLedger?.record(agentId, text); // Dual-write to Markdown mirror if enabled if (context.mdMirror) { @@ -1723,7 +1731,9 @@ export function registerMemoryUpdateTool(api, context) { // The superseding write succeeded: this text will echo through // the same turn's auto-capture extraction exactly like a plain // manual store, so it must be recorded on this early-return - // path too. + // path too, and the replaced text must stop suppressing its + // own re-statement. + context.manualEchoLedger?.invalidate(agentId, existing.text); context.manualEchoLedger?.record(agentId, text); return { content: [ @@ -1796,6 +1806,9 @@ export function registerMemoryUpdateTool(api, context) { details: { error: "not_found", id: resolvedId }, }; } + if (text && existing) { + context.manualEchoLedger?.invalidate(agentId, existing.text); + } context.manualEchoLedger?.record(agentId, updated.text); return { content: [ diff --git a/index.ts b/index.ts index 8082a355..d9d30616 100644 --- a/index.ts +++ b/index.ts @@ -2701,8 +2701,9 @@ function _initPluginState(api: OpenClawPluginApi): PluginSingletonState { // its own. let smartExtractor: SmartExtractor | null = null; // Echo guard: shared between the manual store/update tools (record side) - // and the smart extractor (drop side); lives here so the tools keep - // recording even when smart extraction is disabled. + // and the smart extractor (drop side). Constructed unconditionally, but + // wired into the tools only when a smart extractor exists: an echo can + // only arise when extraction is able to re-mint the dictated text. const manualEchoLedger = new ManualEchoLedger(); let admissionController: AdmissionController | null = null; let admissionControllerReflectionLane: AdmissionController | null = null; @@ -3559,7 +3560,6 @@ const memoryLanceDBProPlugin = { store, retriever, scopeManager, - manualEchoLedger: smartExtractor ? manualEchoLedger : undefined, onMemoriesDeleted: ({ scopeFilter }) => invalidateReflectionCachesAfterDelete(scopeFilter), migrator, embedder, diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts index b828d8e3..c44b12a4 100644 --- a/src/manual-echo-guard.ts +++ b/src/manual-echo-guard.ts @@ -8,25 +8,38 @@ * and drops near-identical extraction candidates BEFORE the admission judge: * deterministic, string-only, no LLM calls, no vector search. * - * Matching is deliberately ONE-SIDED and conservative: a candidate is an - * echo only when it adds NOTHING substantive beyond the recorded manual - * text (exact match, the manual text containing the candidate, or every - * candidate content token already present in the manual text). A candidate + * Matching is deliberately conservative: a candidate is an echo only when + * it asserts NOTHING beyond, and nothing different from, the recorded + * manual text: an exact match, the manual text containing the candidate + * verbatim, or the candidate carrying exactly the manual text's content + * tokens in the same order once reporting glue is stripped. A candidate * carrying extra content — a negation ("no longer"), a changed value, a * temporal qualifier ("until friday"), or additional facts — is new - * information and always survives; the worst case of the guard staying - * quiet is the pre-guard status quo (one duplicate row for dedup). + * information and always survives, and so is one that swaps or drops a + * semantic predicate ("wants" against "has", "prefers Python" against + * "prefers Go over Python"). The worst case of the guard staying quiet is + * the pre-guard status quo (one duplicate row for dedup); the worst case of + * it firing wrongly is a silently lost memory, so every ambiguity resolves + * toward keeping the candidate. * * Entries are short-lived and consumed: each recorded manual text expires * after MANUAL_ECHO_TTL_MS and suppresses at most ONE candidate (the * immediate re-extraction of the same turn). A later identical statement is - * a deliberate user re-assertion, not an echo. + * a deliberate user re-assertion, not an echo. A statement that gets + * replaced (memory_update, supersede) or forgotten is invalidated at once, + * so a reversal back to it is never mistaken for an echo of a fact the + * store no longer holds. * * Scoped per agent (not per session): the store tool and the auto-capture * hook derive their session keys differently, but both resolve the same * agent id, and an echo of ANY recent manual text of the same agent is a * correct drop regardless of session boundaries. TTL + consumption bound * staleness; the ring bounds size. + * + * The ledger is in-memory and per process: only the gateway that recorded + * a text can invalidate it. Deletions made from another process (the + * memory-pro CLI) are not visible here; the TTL bounds that window to + * MANUAL_ECHO_TTL_MS, after which the stale entry expires on its own. */ export const MANUAL_ECHO_RING_SIZE = 8; @@ -38,9 +51,12 @@ const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; /** - * Glue vocabulary the extractor wraps a dictated fact in ("User stated - * that ..."). Stripped before token containment so the canonical wrap echo - * still collapses; negation and temporal markers are deliberately NOT here. + * Reporting glue the extractor wraps a dictated fact in ("User stated + * that ..."): articles, copulas, pronouns, prepositions, and reporting + * verbs. Stripped before token comparison so the canonical wrap echo still + * collapses. Semantic predicates (has, wants, likes, prefers, ...) are NOT + * glue: they decide what a sentence asserts, so they stay content tokens. + * Negation and temporal markers are deliberately not here either. */ const ECHO_STOPWORDS = new Set([ "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", @@ -48,8 +64,7 @@ const ECHO_STOPWORDS = new Set([ "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", "your", "user", "users", "stated", "said", "says", "saying", "mentioned", - "noted", "prefers", "prefer", "likes", "like", "wants", "want", "has", - "have", "had", "also", + "noted", "also", ]); /** @@ -184,29 +199,19 @@ export function isNearIdenticalEcho(candidateText: string, manualText: string): // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red - // one"). After glue-word stripping, the candidate's content tokens must - // appear in the manual text IN THE SAME RELATIVE ORDER (an ordered - // subsequence, not a bag-of-words subset): set membership alone would - // collapse "alice reports to bob" onto "bob reports to alice" and discard - // the reversed relationship. One-sided by design either way — a candidate - // with any extra content token (changed value, qualifier, added fact) is - // new information and survives. + // one"). After glue-word stripping, the candidate must carry EXACTLY the + // manual text's content tokens, in the same order. Adding a token (a + // changed value, a qualifier, a new fact) is new information; dropping + // one is a different assertion ("User prefers Python" against "User + // prefers Go over Python for backend services" reverses the preference), + // so a subsequence match is not enough. Order matters too: bag-of-words + // equality would collapse "alice reports to bob" onto "bob reports to + // alice". The manual-side minimum above doubles as the candidate floor, + // since equal sequences have equal length. const candidateContentList = orderedContentTokens(candidate); - if (candidateContentList.length === 0) return false; const manualContentList = orderedContentTokens(manual); - let cursor = 0; - for (const token of candidateContentList) { - let found = -1; - for (let i = cursor; i < manualContentList.length; i++) { - if (manualContentList[i] === token) { - found = i; - break; - } - } - if (found < 0) return false; - cursor = found + 1; - } - return true; + if (candidateContentList.length !== manualContentList.length) return false; + return candidateContentList.every((token, i) => token === manualContentList[i]); } interface ManualEchoEntry { @@ -286,29 +291,4 @@ export class ManualEchoLedger { clear(agentId: string | undefined): void { this.byAgent.delete(agentId?.trim() || DEFAULT_AGENT_BUCKET); } - - /** - * Deletion-lane invalidation when the deleter cannot name the writing - * agent (CLI delete by id): the fact is gone from the store, so no bucket - * may keep suppressing its re-statement. - */ - invalidateEverywhere(text: string): void { - if (typeof text !== "string" || text.trim().length === 0) return; - const target = normalizeEchoText(text); - for (const [key, ring] of [...this.byAgent.entries()]) { - const kept = ring.filter((e) => normalizeEchoText(e.text) !== target); - if (kept.length === 0) this.byAgent.delete(key); - else if (kept.length !== ring.length) this.byAgent.set(key, kept); - } - } - - /** - * Wholesale reset for bulk deletion lanes, where pre-fetching every - * deleted row's text would defeat the point of a bulk delete. Clearing is - * fail-open: the worst case is one uncaught echo (a duplicate row for - * dedup), never a lost memory. - */ - clearAll(): void { - this.byAgent.clear(); - } } diff --git a/src/tools.ts b/src/tools.ts index 841dc2c0..34564cf8 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -1699,6 +1699,14 @@ export function registerMemoryStoreTool( ); } + // The replaced statements leave the echo ledger with the rows: + // a reversal back to a superseded text is new information once + // the store no longer holds that fact. + for (const target of lastDiscovery.targets) { + if (supersededIds.includes(target.entry.id)) { + context.manualEchoLedger?.invalidate(agentId, target.entry.text); + } + } context.manualEchoLedger?.record(agentId, text); // Dual-write to Markdown mirror if enabled @@ -2191,7 +2199,9 @@ export function registerMemoryUpdateTool( // The superseding write succeeded: this text will echo through // the same turn's auto-capture extraction exactly like a plain // manual store, so it must be recorded on this early-return - // path too. + // path too, and the replaced text must stop suppressing its + // own re-statement. + context.manualEchoLedger?.invalidate(agentId, existing.text); context.manualEchoLedger?.record(agentId, text); return { content: [ @@ -2271,6 +2281,9 @@ export function registerMemoryUpdateTool( }; } + if (text && existing) { + context.manualEchoLedger?.invalidate(agentId, existing.text); + } context.manualEchoLedger?.record(agentId, updated.text); return { diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs index 96cba09d..25719d7f 100644 --- a/test/manual-echo-guard.test.mjs +++ b/test/manual-echo-guard.test.mjs @@ -151,6 +151,49 @@ describe("isNearIdenticalEcho", () => { ); }); + it("keeps a candidate that swaps a semantic predicate (wants against has)", () => { + assert.equal( + isNearIdenticalEcho( + "User wants a golden retriever named Max", + "User has a golden retriever named Max", + ), + false, + "has/wants/likes/prefers decide what a sentence asserts; they are content, not glue", + ); + }); + + it("keeps a candidate that drops the distinguishing part of a preference", () => { + assert.equal( + isNearIdenticalEcho( + "User prefers Python", + "User prefers Go over Python for backend services", + ), + false, + "an ordered subsequence of the manual content is a different assertion, not an echo", + ); + }); + + it("keeps a candidate with a different predicate and object against a fuller manual text", () => { + assert.equal( + isNearIdenticalEcho( + "User likes tea", + "User prefers coffee over tea in the morning", + ), + false, + ); + }); + + it("still collapses the same-order wrap echo when the fact carries a predicate", () => { + assert.equal( + isNearIdenticalEcho( + "User mentioned that Alice prefers Go for backend services", + "alice prefers go for backend services", + ), + true, + "reporting glue around an otherwise identical assertion is still an echo", + ); + }); + it("rejects unrelated candidates", () => { assert.equal( isNearIdenticalEcho("user's dog is named Biscuit", manual), @@ -328,26 +371,6 @@ describe("ManualEchoLedger", () => { ); }); - it("invalidateEverywhere() drops the text from every agent bucket", () => { - const ledger = new ManualEchoLedger(); - ledger.record("agent-one", "favorite teacup: the red one"); - ledger.record("agent-two", "favorite teacup: the red one"); - ledger.record("agent-two", "the office plant needs watering every friday"); - ledger.invalidateEverywhere("Favorite Teacup: the RED one"); - assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); - assert.equal(ledger.match("agent-two", "favorite teacup: the red one"), null); - assert.ok(ledger.match("agent-two", "the office plant needs watering every friday")); - }); - - it("clearAll() empties every bucket (bulk-delete lane)", () => { - const ledger = new ManualEchoLedger(); - ledger.record("agent-one", "favorite teacup: the red one"); - ledger.record("agent-two", "the office plant needs watering every friday"); - ledger.clearAll(); - assert.equal(ledger.match("agent-one", "favorite teacup: the red one"), null); - assert.equal(ledger.match("agent-two", "the office plant needs watering every friday"), null); - }); - it("clear() empties one agent's ring only", () => { const ledger = new ManualEchoLedger(); ledger.record("agent-one", "favorite teacup: the red one"); @@ -592,6 +615,7 @@ describe("echo guard through the full auto-capture path", () => { context, { enableManagementTools: true }, ); + ledger.record("main", ORIGINAL); const update = creators.get("memory_update")({}); const updated = await update.execute(null, { memoryId: EXISTING_ID, text: UPDATED }); assert.equal( @@ -603,5 +627,75 @@ describe("echo guard through the full auto-capture path", () => { ledger.match("main", UPDATED), "the successful temporal supersede must record the NEW text in the echo ledger before its early return", ); + assert.equal( + ledger.match("main", ORIGINAL), + null, + "the replaced text must be invalidated: a reversal back to it is new information once the store no longer holds it", + ); + }); + + it("invalidates the replaced text on the plain (non-temporal) memory_update path too", async () => { + const { registerAllMemoryTools } = jiti("../src/tools.ts"); + const ledger = new ManualEchoLedger(); + const EXISTING_ID = "21111111-2222-4333-8444-555555555555"; + const ORIGINAL = "rehearsal warm-up routine: scales for ten minutes"; + const UPDATED = "rehearsal warm-up routine: long tones for ten minutes"; + const existingEntry = { + id: EXISTING_ID, + text: ORIGINAL, + category: "fact", + scope: "agent:main", + importance: 0.7, + timestamp: Date.now() - 60_000, + metadata: JSON.stringify({ + memory_category: "patterns", + l0_abstract: ORIGINAL, + l1_overview: `- ${ORIGINAL}`, + l2_content: ORIGINAL, + source: "manual", + state: "confirmed", + }), + }; + const context = { + agentId: "main", + workspaceDir: workspaceDir, + mdMirror: null, + manualEchoLedger: ledger, + scopeManager: { + getAccessibleScopes: (agentId) => ["global", `agent:${agentId}`], + getScopeFilter: (agentId) => ["global", `agent:${agentId}`], + isAccessible: (scope, agentId) => ["global", `agent:${agentId}`].includes(scope), + getDefaultScope: (agentId) => `agent:${agentId}`, + }, + retriever: { getConfig() { return { mode: "hybrid" }; } }, + store: { + async getById(id) { return id === EXISTING_ID ? existingEntry : null; }, + async vectorSearch() { return []; }, + async list() { return [existingEntry]; }, + async listFactKeyCandidates() { return [existingEntry]; }, + async store(entry) { return { ...entry, id: "99999999-8888-4777-8666-555555555553", timestamp: Date.now() }; }, + async update(id, patch) { return { ...existingEntry, ...patch, id }; }, + }, + embedder: { async embedPassage() { return [0.1, 0.2, 0.3]; } }, + }; + const creators = new Map(); + registerAllMemoryTools( + { + registerTool(factory, meta) { creators.set(meta.name, factory); }, + logger: { info() {}, warn() {}, debug() {} }, + }, + context, + { enableManagementTools: true }, + ); + ledger.record("main", ORIGINAL); + const update = creators.get("memory_update")({}); + const updated = await update.execute(null, { memoryId: EXISTING_ID, text: UPDATED }); + assert.notEqual( + updated?.details?.action, + "superseded", + `a patterns row must take the plain update path (got ${JSON.stringify(updated?.details)})`, + ); + assert.ok(ledger.match("main", UPDATED), "the new text is recorded"); + assert.equal(ledger.match("main", ORIGINAL), null, "the replaced text is invalidated on the plain update path"); }); }); diff --git a/test/manual-store-supersede.test.mjs b/test/manual-store-supersede.test.mjs index b8870409..f5a072e9 100644 --- a/test/manual-store-supersede.test.mjs +++ b/test/manual-store-supersede.test.mjs @@ -33,6 +33,7 @@ const { registerAllMemoryTools } = jiti("../src/tools.ts"); const { MemoryStore } = jiti("../src/store.ts"); const { parseSmartMetadata, isMemoryActiveAt, deriveFactKey } = jiti("../src/smart-metadata.ts"); const { classifyTemporal, inferExpiry } = jiti("../src/temporal-classifier.ts"); +const { ManualEchoLedger } = jiti("../src/manual-echo-guard.ts"); function createToolSet(context) { const creators = new Map(); @@ -168,6 +169,29 @@ function makeContext({ neighbors = [], rows, manualStoreSupersede, patchBehavior } describe("manual memory_store always-store supersede semantics", () => { + it("invalidates the superseded rows' text in the echo ledger and records the new text (knob on)", async () => { + const OLD = "favorite rehearsal drink is sparkling water"; + const NEW = "favorite rehearsal drink is mint tea"; + const { context, storedEntries } = makeContext({ + neighbors: [neighborRow({ id: "old-1", text: OLD, score: 0.99 })], + manualStoreSupersede: true, + }); + const ledger = new ManualEchoLedger(); + context.manualEchoLedger = ledger; + ledger.record("main", OLD); + + const result = await createToolSet(context).get("memory_store").execute("call-1", { text: NEW }); + + assert.equal(result.details.action, "superseded", JSON.stringify(result.details)); + assert.equal(storedEntries.length, 1); + assert.ok(ledger.match("main", NEW), "the superseding text is recorded for the same turn's echo"); + assert.equal( + ledger.match("main", OLD), + null, + "the replaced text must be invalidated: the store no longer holds it, so its re-statement is not an echo", + ); + }); + it("supersedes instead of rejecting when a near-identical memory exists (knob on)", async () => { const { context, storedEntries, patchCalls } = makeContext({ manualStoreSupersede: true, From 94fc25f4364f94dc354ec0c267f0170567b3e202 Mon Sep 17 00:00:00 2001 From: Gorkem Date: Fri, 11 Sep 2026 17:30:21 +0300 Subject: [PATCH 5/6] fix(echo-guard): whole-token containment, relation-word agreement, text-only ledger records Three silent-loss cases from review: - The shortened-echo branch used raw substring containment, so "User has a cat" matched inside "User has a catalog of vinyl records". It now requires the candidate to appear as a run of whole tokens in the manual text and to carry at least the containment floor of content tokens of its own. - Copulas, prepositions and conjunctions were stripped as stopwords, so "is" against "was", "with" against "for" and "or" against "and" compared equal. They are now a separate relation-token sequence that must agree between the two texts; the only addition a sentence wrapper may make is a present-tense copula ("teacup: the red one" to "teacup is the red one"). Reporting glue (articles, demonstratives, pronouns, reporting verbs) is still stripped, so the legitimate wrap echoes keep collapsing. - memory_update recorded updated.text unconditionally, arming suppression after a metadata-only update. It now records only when a text was supplied, invalidating the replaced text when it changed. Regressions cover each case in both directions plus the wrap echoes that must keep matching, and a handler-level test for the metadata-only update. --- dist/src/manual-echo-guard.js | 102 ++++++++++++++++----- dist/src/tools.js | 9 +- src/manual-echo-guard.ts | 106 +++++++++++++++++----- src/tools.ts | 9 +- test/manual-echo-guard.test.mjs | 154 +++++++++++++++++++++++++++++++- 5 files changed, 332 insertions(+), 48 deletions(-) diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js index 7c1a92a6..90106783 100644 --- a/dist/src/manual-echo-guard.js +++ b/dist/src/manual-echo-guard.js @@ -11,8 +11,12 @@ * Matching is deliberately conservative: a candidate is an echo only when * it asserts NOTHING beyond, and nothing different from, the recorded * manual text: an exact match, the manual text containing the candidate - * verbatim, or the candidate carrying exactly the manual text's content - * tokens in the same order once reporting glue is stripped. A candidate + * as a whole-token run (never a raw substring: "has a cat" is not inside + * "has a catalog"), or the candidate carrying exactly the manual text's + * content tokens in the same order once reporting glue is stripped, with + * the relation-bearing words (copulas, prepositions, conjunctions) agreeing + * too, since "is" against "was", "with" against "for" and "or" against + * "and" are different assertions. A candidate * carrying extra content — a negation ("no longer"), a changed value, a * temporal qualifier ("until friday"), or additional facts — is new * information and always survives, and so is one that swaps or drops a @@ -50,20 +54,35 @@ const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; /** * Reporting glue the extractor wraps a dictated fact in ("User stated - * that ..."): articles, copulas, pronouns, prepositions, and reporting - * verbs. Stripped before token comparison so the canonical wrap echo still - * collapses. Semantic predicates (has, wants, likes, prefers, ...) are NOT - * glue: they decide what a sentence asserts, so they stay content tokens. - * Negation and temporal markers are deliberately not here either. + * that ..."): articles, demonstratives, pronouns, and reporting verbs. + * Stripped before token comparison so the canonical wrap echo still + * collapses. Only words that carry no assertion of their own belong here. + * Semantic predicates (has, wants, likes, prefers, ...) decide what a + * sentence asserts and stay content tokens; copulas, prepositions and + * conjunctions carry tense, relation and logic and are compared separately + * (ECHO_RELATION_TOKENS). Negation and temporal markers are handled by + * NEGATION_AND_TEMPORAL_MARKERS. */ -const ECHO_STOPWORDS = new Set([ - "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", - "that", "this", "these", "those", "to", "of", "in", "on", "at", "for", - "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", - "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", - "your", "user", "users", "stated", "said", "says", "saying", "mentioned", - "noted", "also", +const ECHO_GLUE = new Set([ + "the", "a", "an", "that", "this", "these", "those", "it", "its", "their", + "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", + "you", "your", "user", "users", "stated", "said", "says", "saying", + "mentioned", "noted", "also", ]); +/** + * Relation-bearing function words. They are not content (a wrap echo may + * legitimately add "is" when it turns "teacup: the red one" into "teacup is + * the red one"), but they are not glue either: swapping one changes the + * assertion ("is" / "was" is a tense change, "with" / "for" a different + * relation, "or" / "and" different logic). Both sides' sequences must + * agree, except that the candidate may insert a present-tense copula. + */ +const ECHO_RELATION_TOKENS = new Set([ + "is", "are", "was", "were", "be", "been", "being", "to", "of", "in", "on", + "at", "for", "with", "and", "or", "as", "by", "from", +]); +/** The only relation tokens a sentence wrapper may add to a dictated fact. */ +const WRAP_INSERTABLE_COPULAS = new Set(["is", "are"]); /** * A marker on exactly one side of the pair means the two texts assert * different things (a correction, a retraction, a bounded validity): never @@ -96,12 +115,46 @@ function orderedContentTokens(normalized) { for (const token of tokenList(normalized)) { if (token.length <= 1 && !CJK_RE.test(token)) continue; - if (ECHO_STOPWORDS.has(token)) + if (ECHO_GLUE.has(token) || ECHO_RELATION_TOKENS.has(token)) continue; out.push(token); } return out; } +function orderedRelationTokens(normalized) { + return tokenList(normalized).filter((token) => ECHO_RELATION_TOKENS.has(token)); +} +/** + * The manual text's relation tokens must all appear in the candidate, in + * order; whatever the candidate adds on top must be a copula a sentence + * wrapper inserts. Anything else (a dropped, swapped or extra relation + * word) is a different assertion. + */ +function relationTokensAgree(candidateRelations, manualRelations) { + let m = 0; + for (const token of candidateRelations) { + if (m < manualRelations.length && token === manualRelations[m]) { + m++; + continue; + } + if (!WRAP_INSERTABLE_COPULAS.has(token)) + return false; + } + return m === manualRelations.length; +} +/** Whole-token containment: every needle token, contiguous, in the haystack. */ +function containsTokenRun(haystack, needle) { + if (needle.length === 0 || needle.length > haystack.length) + return false; + outer: for (let start = 0; start + needle.length <= haystack.length; start++) { + for (let i = 0; i < needle.length; i++) { + if (haystack[start + i] !== needle[i]) + continue outer; + } + return true; + } + return false; +} function contentTokens(normalized) { return new Set(orderedContentTokens(normalized)); } @@ -186,9 +239,15 @@ export function isNearIdenticalEcho(candidateText, manualText) { const manualContent = contentTokens(manual); if (manualContent.size < MIN_CONTAINMENT_TOKENS) return false; - // Shortened echo: the manual text contains the whole candidate. - if (manual.includes(candidate)) + // Shortened echo: the manual text contains the whole candidate as a run of + // whole tokens. A raw substring test would accept "has a cat" inside "has + // a catalog of vinyl records"; the candidate also needs enough content of + // its own before a partial restatement counts. + const candidateContentList = orderedContentTokens(candidate); + if (candidateContentList.length >= MIN_CONTAINMENT_TOKENS && + containsTokenRun(manualTokenList, candidateTokenList)) { return true; + } // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red // one"). After glue-word stripping, the candidate must carry EXACTLY the @@ -199,12 +258,15 @@ export function isNearIdenticalEcho(candidateText, manualText) { // so a subsequence match is not enough. Order matters too: bag-of-words // equality would collapse "alice reports to bob" onto "bob reports to // alice". The manual-side minimum above doubles as the candidate floor, - // since equal sequences have equal length. - const candidateContentList = orderedContentTokens(candidate); + // since equal sequences have equal length. Relation words are compared as + // their own sequence: "tea or coffee" is not "tea and coffee", and "is" + // is not "was", even though the content tokens agree. const manualContentList = orderedContentTokens(manual); if (candidateContentList.length !== manualContentList.length) return false; - return candidateContentList.every((token, i) => token === manualContentList[i]); + if (!candidateContentList.every((token, i) => token === manualContentList[i])) + return false; + return relationTokensAgree(orderedRelationTokens(candidate), orderedRelationTokens(manual)); } export class ManualEchoLedger { byAgent = new Map(); diff --git a/dist/src/tools.js b/dist/src/tools.js index e7c5302a..72ef1e9e 100644 --- a/dist/src/tools.js +++ b/dist/src/tools.js @@ -1806,10 +1806,15 @@ export function registerMemoryUpdateTool(api, context) { details: { error: "not_found", id: resolvedId }, }; } + // Only a manually supplied text arms the echo guard: a metadata-only + // update (importance, category) restates nothing, so it must not + // suppress a later extraction of the unchanged fact. if (text && existing) { - context.manualEchoLedger?.invalidate(agentId, existing.text); + if (updated.text !== existing.text) { + context.manualEchoLedger?.invalidate(agentId, existing.text); + } + context.manualEchoLedger?.record(agentId, updated.text); } - context.manualEchoLedger?.record(agentId, updated.text); return { content: [ { diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts index c44b12a4..b9a331d3 100644 --- a/src/manual-echo-guard.ts +++ b/src/manual-echo-guard.ts @@ -11,8 +11,12 @@ * Matching is deliberately conservative: a candidate is an echo only when * it asserts NOTHING beyond, and nothing different from, the recorded * manual text: an exact match, the manual text containing the candidate - * verbatim, or the candidate carrying exactly the manual text's content - * tokens in the same order once reporting glue is stripped. A candidate + * as a whole-token run (never a raw substring: "has a cat" is not inside + * "has a catalog"), or the candidate carrying exactly the manual text's + * content tokens in the same order once reporting glue is stripped, with + * the relation-bearing words (copulas, prepositions, conjunctions) agreeing + * too, since "is" against "was", "with" against "for" and "or" against + * "and" are different assertions. A candidate * carrying extra content — a negation ("no longer"), a changed value, a * temporal qualifier ("until friday"), or additional facts — is new * information and always survives, and so is one that swaps or drops a @@ -52,21 +56,38 @@ const DEFAULT_AGENT_BUCKET = "main"; /** * Reporting glue the extractor wraps a dictated fact in ("User stated - * that ..."): articles, copulas, pronouns, prepositions, and reporting - * verbs. Stripped before token comparison so the canonical wrap echo still - * collapses. Semantic predicates (has, wants, likes, prefers, ...) are NOT - * glue: they decide what a sentence asserts, so they stay content tokens. - * Negation and temporal markers are deliberately not here either. + * that ..."): articles, demonstratives, pronouns, and reporting verbs. + * Stripped before token comparison so the canonical wrap echo still + * collapses. Only words that carry no assertion of their own belong here. + * Semantic predicates (has, wants, likes, prefers, ...) decide what a + * sentence asserts and stay content tokens; copulas, prepositions and + * conjunctions carry tense, relation and logic and are compared separately + * (ECHO_RELATION_TOKENS). Negation and temporal markers are handled by + * NEGATION_AND_TEMPORAL_MARKERS. */ -const ECHO_STOPWORDS = new Set([ - "the", "a", "an", "is", "are", "was", "were", "be", "been", "being", - "that", "this", "these", "those", "to", "of", "in", "on", "at", "for", - "with", "and", "or", "as", "by", "from", "it", "its", "their", "they", - "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", "you", - "your", "user", "users", "stated", "said", "says", "saying", "mentioned", - "noted", "also", +const ECHO_GLUE = new Set([ + "the", "a", "an", "that", "this", "these", "those", "it", "its", "their", + "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", + "you", "your", "user", "users", "stated", "said", "says", "saying", + "mentioned", "noted", "also", ]); +/** + * Relation-bearing function words. They are not content (a wrap echo may + * legitimately add "is" when it turns "teacup: the red one" into "teacup is + * the red one"), but they are not glue either: swapping one changes the + * assertion ("is" / "was" is a tense change, "with" / "for" a different + * relation, "or" / "and" different logic). Both sides' sequences must + * agree, except that the candidate may insert a present-tense copula. + */ +const ECHO_RELATION_TOKENS = new Set([ + "is", "are", "was", "were", "be", "been", "being", "to", "of", "in", "on", + "at", "for", "with", "and", "or", "as", "by", "from", +]); + +/** The only relation tokens a sentence wrapper may add to a dictated fact. */ +const WRAP_INSERTABLE_COPULAS = new Set(["is", "are"]); + /** * A marker on exactly one side of the pair means the two texts assert * different things (a correction, a retraction, a bounded validity): never @@ -103,12 +124,46 @@ function orderedContentTokens(normalized: string): string[] { const out: string[] = []; for (const token of tokenList(normalized)) { if (token.length <= 1 && !CJK_RE.test(token)) continue; - if (ECHO_STOPWORDS.has(token)) continue; + if (ECHO_GLUE.has(token) || ECHO_RELATION_TOKENS.has(token)) continue; out.push(token); } return out; } +function orderedRelationTokens(normalized: string): string[] { + return tokenList(normalized).filter((token) => ECHO_RELATION_TOKENS.has(token)); +} + +/** + * The manual text's relation tokens must all appear in the candidate, in + * order; whatever the candidate adds on top must be a copula a sentence + * wrapper inserts. Anything else (a dropped, swapped or extra relation + * word) is a different assertion. + */ +function relationTokensAgree(candidateRelations: string[], manualRelations: string[]): boolean { + let m = 0; + for (const token of candidateRelations) { + if (m < manualRelations.length && token === manualRelations[m]) { + m++; + continue; + } + if (!WRAP_INSERTABLE_COPULAS.has(token)) return false; + } + return m === manualRelations.length; +} + +/** Whole-token containment: every needle token, contiguous, in the haystack. */ +function containsTokenRun(haystack: string[], needle: string[]): boolean { + if (needle.length === 0 || needle.length > haystack.length) return false; + outer: for (let start = 0; start + needle.length <= haystack.length; start++) { + for (let i = 0; i < needle.length; i++) { + if (haystack[start + i] !== needle[i]) continue outer; + } + return true; + } + return false; +} + function contentTokens(normalized: string): Set { return new Set(orderedContentTokens(normalized)); } @@ -194,8 +249,17 @@ export function isNearIdenticalEcho(candidateText: string, manualText: string): const manualContent = contentTokens(manual); if (manualContent.size < MIN_CONTAINMENT_TOKENS) return false; - // Shortened echo: the manual text contains the whole candidate. - if (manual.includes(candidate)) return true; + // Shortened echo: the manual text contains the whole candidate as a run of + // whole tokens. A raw substring test would accept "has a cat" inside "has + // a catalog of vinyl records"; the candidate also needs enough content of + // its own before a partial restatement counts. + const candidateContentList = orderedContentTokens(candidate); + if ( + candidateContentList.length >= MIN_CONTAINMENT_TOKENS && + containsTokenRun(manualTokenList, candidateTokenList) + ) { + return true; + } // Wrap echo: the extractor sentence-wraps the dictated fact ("favorite // teacup: the red one" -> "User stated their favorite teacup is the red @@ -207,11 +271,13 @@ export function isNearIdenticalEcho(candidateText: string, manualText: string): // so a subsequence match is not enough. Order matters too: bag-of-words // equality would collapse "alice reports to bob" onto "bob reports to // alice". The manual-side minimum above doubles as the candidate floor, - // since equal sequences have equal length. - const candidateContentList = orderedContentTokens(candidate); + // since equal sequences have equal length. Relation words are compared as + // their own sequence: "tea or coffee" is not "tea and coffee", and "is" + // is not "was", even though the content tokens agree. const manualContentList = orderedContentTokens(manual); if (candidateContentList.length !== manualContentList.length) return false; - return candidateContentList.every((token, i) => token === manualContentList[i]); + if (!candidateContentList.every((token, i) => token === manualContentList[i])) return false; + return relationTokensAgree(orderedRelationTokens(candidate), orderedRelationTokens(manual)); } interface ManualEchoEntry { diff --git a/src/tools.ts b/src/tools.ts index 34564cf8..293b6355 100644 --- a/src/tools.ts +++ b/src/tools.ts @@ -2281,10 +2281,15 @@ export function registerMemoryUpdateTool( }; } + // Only a manually supplied text arms the echo guard: a metadata-only + // update (importance, category) restates nothing, so it must not + // suppress a later extraction of the unchanged fact. if (text && existing) { - context.manualEchoLedger?.invalidate(agentId, existing.text); + if (updated.text !== existing.text) { + context.manualEchoLedger?.invalidate(agentId, existing.text); + } + context.manualEchoLedger?.record(agentId, updated.text); } - context.manualEchoLedger?.record(agentId, updated.text); return { content: [ diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs index 25719d7f..2b5055de 100644 --- a/test/manual-echo-guard.test.mjs +++ b/test/manual-echo-guard.test.mjs @@ -12,10 +12,11 @@ * search. * * Match test is ONE-SIDED and conservative: exact match, the manual text - * containing the candidate, or every candidate content token (after glue-word - * stripping) already present in the manual text. A candidate carrying ANY - * extra content — negation, changed value, temporal qualifier, added facts — - * always survives. Entries expire (TTL), are consumed on match, and are + * containing the candidate as a run of whole tokens, or the candidate carrying + * exactly the manual text's content tokens in order (after glue-word + * stripping) with the relation words (copulas, prepositions, conjunctions) + * agreeing. A candidate carrying ANY extra content — negation, changed value, + * temporal qualifier, added facts, a swapped relation word — always survives. Entries expire (TTL), are consumed on match, and are * invalidated when their memory is forgotten. * * Fixtures are entirely synthetic; no real fleet data. @@ -699,3 +700,148 @@ describe("echo guard through the full auto-capture path", () => { assert.equal(ledger.match("main", ORIGINAL), null, "the replaced text is invalidated on the plain update path"); }); }); + +describe("review round 4: whole-token containment and relation words", () => { + it("does not treat a token prefix as containment (cat inside catalog)", () => { + assert.equal( + isNearIdenticalEcho("User has a cat", "User has a catalog of vinyl records"), + false, + "a raw substring match would drop a distinct fact", + ); + }); + + it("requires a candidate-side content floor before a partial restatement counts", () => { + assert.equal( + isNearIdenticalEcho("User has a cat", "User has a cat and a dog named Rex"), + false, + "two content tokens are not enough to call a shortened restatement an echo", + ); + assert.equal( + isNearIdenticalEcho("adopted a cat named Miso", "User adopted a cat named Miso last spring"), + true, + "a whole-token run with enough content of its own still collapses", + ); + }); + + it("keeps a tense change (is against was), in both directions", () => { + assert.equal(isNearIdenticalEcho("Alice is in Paris this month", "Alice was in Paris this month"), false); + assert.equal(isNearIdenticalEcho("Alice was in Paris this month", "Alice is in Paris this month"), false); + assert.equal(isNearIdenticalEcho("User said Alice was in Paris this month", "Alice is in Paris this month"), false); + }); + + it("keeps a changed relation (with against for)", () => { + assert.equal(isNearIdenticalEcho("Alice works with Bob", "Alice works for Bob"), false); + assert.equal(isNearIdenticalEcho("User mentioned Alice works for Bob", "Alice works with Bob"), false); + }); + + it("keeps changed logic (or against and)", () => { + assert.equal( + isNearIdenticalEcho("Alice likes tea or coffee daily", "Alice likes tea and coffee daily"), + false, + ); + assert.equal( + isNearIdenticalEcho("User noted Alice likes tea and coffee daily", "Alice likes tea or coffee daily"), + false, + ); + }); + + it("keeps a dropped or added relation word", () => { + assert.equal(isNearIdenticalEcho("Alice in Paris this month", "Alice is in Paris this month"), false, "dropped copula"); + assert.equal( + isNearIdenticalEcho("Alice works with Bob for now", "Alice works with Bob"), + false, + "an added qualifier changes the assertion", + ); + }); + + it("still collapses wrap echoes whose relation words agree, and tolerates the wrapper's inserted copula", () => { + assert.equal(isNearIdenticalEcho("User said Alice is in Paris this month", "Alice is in Paris this month"), true); + assert.equal( + isNearIdenticalEcho("User stated that Alice works with Bob on the roadmap", "Alice works with Bob on the roadmap"), + true, + ); + assert.equal( + isNearIdenticalEcho("User's favorite colors are red and blue", "favorite colors: red and blue"), + true, + "the wrapper may turn 'x: y' into 'x are y'", + ); + assert.equal( + isNearIdenticalEcho("User mentioned that Alice likes tea and coffee daily", "Alice likes tea and coffee daily"), + true, + ); + }); +}); + +describe("review round 4: memory_update records the echo ledger only for a supplied text", () => { + it("a metadata-only update neither records nor invalidates the ledger", async () => { + const { registerAllMemoryTools } = jiti("../src/tools.ts"); + const ledger = new ManualEchoLedger(); + const EXISTING_ID = "31111111-2222-4333-8444-555555555555"; + const ORIGINAL = "rehearsal warm-up routine: scales for ten minutes"; + const existingEntry = { + id: EXISTING_ID, + text: ORIGINAL, + category: "fact", + scope: "agent:main", + importance: 0.7, + timestamp: Date.now() - 60_000, + metadata: JSON.stringify({ + memory_category: "patterns", + l0_abstract: ORIGINAL, + l1_overview: `- ${ORIGINAL}`, + l2_content: ORIGINAL, + source: "manual", + state: "confirmed", + }), + }; + const metaWorkspace = mkdtempSync(path.join(tmpdir(), "echo-guard-meta-update-")); + try { + const context = { + agentId: "main", + workspaceDir: metaWorkspace, + mdMirror: null, + manualEchoLedger: ledger, + scopeManager: { + getAccessibleScopes: (agentId) => ["global", `agent:${agentId}`], + getScopeFilter: (agentId) => ["global", `agent:${agentId}`], + isAccessible: (scope, agentId) => ["global", `agent:${agentId}`].includes(scope), + getDefaultScope: (agentId) => `agent:${agentId}`, + }, + retriever: { getConfig() { return { mode: "hybrid" }; } }, + store: { + async getById(id) { return id === EXISTING_ID ? existingEntry : null; }, + async vectorSearch() { return []; }, + async list() { return [existingEntry]; }, + async listFactKeyCandidates() { return [existingEntry]; }, + async store(entry) { return { ...entry, id: "99999999-8888-4777-8666-555555555554", timestamp: Date.now() }; }, + async update(id, patch) { return { ...existingEntry, ...patch, id }; }, + }, + embedder: { async embedPassage() { return [0.1, 0.2, 0.3]; } }, + }; + const creators = new Map(); + registerAllMemoryTools( + { + registerTool(factory, meta) { creators.set(meta.name, factory); }, + logger: { info() {}, warn() {}, debug() {} }, + }, + context, + { enableManagementTools: true }, + ); + const update = creators.get("memory_update")({}); + + const metaOnly = await update.execute(null, { memoryId: EXISTING_ID, importance: 0.9 }); + assert.ok(!metaOnly?.details?.error, `metadata-only update must succeed (got ${JSON.stringify(metaOnly?.details)})`); + assert.equal(ledger.match("main", ORIGINAL), null, "no text was supplied, so nothing is armed for suppression"); + + ledger.record("main", ORIGINAL); + await update.execute(null, { memoryId: EXISTING_ID, importance: 0.5 }); + assert.ok(ledger.match("main", ORIGINAL), "a metadata-only update leaves an existing ledger entry alone"); + + const textUpdate = await update.execute(null, { memoryId: EXISTING_ID, text: "rehearsal warm-up routine: long tones" }); + assert.ok(!textUpdate?.details?.error); + assert.ok(ledger.match("main", "rehearsal warm-up routine: long tones"), "a supplied text is recorded"); + } finally { + rmSync(metaWorkspace, { recursive: true, force: true }); + } + }); +}); From 052dcc6e982a8696c42c4687caf72a94d2e6b28d Mon Sep 17 00:00:00 2001 From: Gorkem Date: Sun, 13 Sep 2026 09:16:31 +0300 Subject: [PATCH 6/6] fix(echo-guard): fail open on partial CJK containment, keep referent pronouns as content, invalidate on extractor supersede Partial CJK containment has no word boundary to trust: a candidate inside the manual text can drop the object of the fact, and with whitespace removed a Latin word matches inside a longer one. Only exact and glue-wrapped CJK echoes remain. Third-person pronouns name a referent and now compare as content, except inside a User-wrapped candidate where they are the wrap's back-reference to the user; the speaker (first person, the User label) stays implicit on both sides. The extractor's supersede path drops the replaced text from the manual ledger. --- dist/src/manual-echo-guard.js | 71 ++++++++++++++++++-------------- dist/src/smart-extractor.js | 2 + src/manual-echo-guard.ts | 72 ++++++++++++++++++--------------- src/smart-extractor.ts | 2 + test/manual-echo-guard.test.mjs | 46 ++++++++++++++++++++- 5 files changed, 129 insertions(+), 64 deletions(-) diff --git a/dist/src/manual-echo-guard.js b/dist/src/manual-echo-guard.js index 90106783..638e360f 100644 --- a/dist/src/manual-echo-guard.js +++ b/dist/src/manual-echo-guard.js @@ -54,21 +54,33 @@ const MAX_CJK_WRAPPER_RESIDUAL_CHARS = 8; const DEFAULT_AGENT_BUCKET = "main"; /** * Reporting glue the extractor wraps a dictated fact in ("User stated - * that ..."): articles, demonstratives, pronouns, and reporting verbs. - * Stripped before token comparison so the canonical wrap echo still - * collapses. Only words that carry no assertion of their own belong here. - * Semantic predicates (has, wants, likes, prefers, ...) decide what a - * sentence asserts and stay content tokens; copulas, prepositions and - * conjunctions carry tense, relation and logic and are compared separately - * (ECHO_RELATION_TOKENS). Negation and temporal markers are handled by - * NEGATION_AND_TEMPORAL_MARKERS. + * that ..."): articles, demonstratives and reporting verbs. Stripped before + * token comparison so the canonical wrap echo still collapses. Only words + * that carry no assertion of their own belong here. Semantic predicates + * (has, wants, likes, prefers, ...) decide what a sentence asserts and stay + * content tokens; copulas, prepositions and conjunctions carry tense, + * relation and logic and are compared separately (ECHO_RELATION_TOKENS). + * Pronouns carry a referent and are content too, except for the one + * perspective transform the wrap performs (see SELF_REFERENCE_TOKENS). + * Negation and temporal markers are handled by NEGATION_AND_TEMPORAL_MARKERS. */ const ECHO_GLUE = new Set([ - "the", "a", "an", "that", "this", "these", "those", "it", "its", "their", - "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", - "you", "your", "user", "users", "stated", "said", "says", "saying", - "mentioned", "noted", "also", + "the", "a", "an", "that", "this", "these", "those", + "stated", "said", "says", "saying", "mentioned", "noted", "also", ]); +/** + * The wrap rewrites the speaker as "User" ("my laptop" -> "User's laptop", + * "I like tea" -> "User stated they like tea"): the speaker is implicit on + * both sides, so first-person references and the user label are dropped. + * Any other pronoun names a referent of its own and stays content, except + * inside a User-wrapped text, where it is the wrap's back-reference to the + * user ("User stated their favorite teacup ..."). Swapping a referent + * anywhere else ("his manager" against "her manager", "their laptop" + * against "my laptop") is a different fact, never an echo. + */ +const SPEAKER_TOKENS = new Set(["i", "me", "my", "mine", "we", "us", "our", "ours", "user", "users"]); +const REFERENT_PRONOUNS = new Set(["he", "she", "his", "her", "hers", "they", "them", "their", "theirs", "it", "its", "you", "your", "yours"]); +const USER_WRAP_TOKENS = new Set(["user", "users"]); /** * Relation-bearing function words. They are not content (a wrap echo may * legitimately add "is" when it turns "teacup: the red one" into "teacup is @@ -93,12 +105,6 @@ const NEGATION_AND_TEMPORAL_MARKERS = new Set([ "former", "formerly", "anymore", "longer", "until", "till", "unless", "except", "without", "before", "after", "used", ]); -/** Conservative CJK marker fragments (negation / bounded validity). */ -const CJK_MARKER_FRAGMENTS = [ - "不", "没", "别", "未", "无", "非", "勿", "直到", "之前", "以前", "除非", - "ない", "じゃない", "ではない", "まで", "もう", - "않", "안", "까지", "전에", -]; const CJK_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u; export function normalizeEchoText(text) { return text @@ -111,10 +117,20 @@ function tokenList(normalized) { return normalized.split(" ").filter((t) => t.length > 0); } function orderedContentTokens(normalized) { + const tokens = tokenList(normalized); + const userWrapped = tokens.some((token) => USER_WRAP_TOKENS.has(token)); const out = []; - for (const token of tokenList(normalized)) { + for (const token of tokens) { if (token.length <= 1 && !CJK_RE.test(token)) continue; + if (SPEAKER_TOKENS.has(token)) + continue; + if (REFERENT_PRONOUNS.has(token)) { + if (userWrapped) + continue; + out.push(token); + continue; + } if (ECHO_GLUE.has(token) || ECHO_RELATION_TOKENS.has(token)) continue; out.push(token); @@ -167,9 +183,6 @@ function markerAsymmetry(aTokens, bTokens) { } return false; } -function containsCjkMarker(fragment) { - return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); -} /** * The ONLY residual a wrapped CJK echo may carry: reporting glue and * particles. Anything else in the residual — a marker, a verb, a new fact @@ -201,21 +214,19 @@ function isCjkEcho(candidate, manual) { return false; if (cand === man) return true; - // Shortened echo: the candidate re-states a piece of the manual text. The - // REMOVED part must carry no marker: stripping 用户不 off 用户不喜欢喝茶和咖啡 - // yields the OPPOSITE claim, not an echo of it. - if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) { - const removed = man.replace(cand, ""); - return !containsCjkMarker(removed); - } // Wrapped echo: the candidate is the manual text plus reporting glue - // (用户说…). The residual must consist ONLY of known glue fragments — + // (用户说…). The residual must consist ONLY of known glue fragments; // being short and marker-free is not enough, since a three-character // residual can be a brand-new fact (并养猫). if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { const residual = cand.replace(man, ""); return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && isCjkGlueOnly(residual); } + // A candidate contained inside the manual text is ambiguous without word + // boundaries: 用户喜欢机器学习 inside 用户喜欢机器学习课程 drops the object, + // and with whitespace removed a Latin fragment matches inside a longer + // word (cat inside catalog). Partial CJK containment therefore fails open: + // a duplicate is left for the dedup lane, a different fact is never lost. return false; } export function isNearIdenticalEcho(candidateText, manualText) { diff --git a/dist/src/smart-extractor.js b/dist/src/smart-extractor.js index b913e014..d4035785 100644 --- a/dist/src/smart-extractor.js +++ b/dist/src/smart-extractor.js @@ -2483,6 +2483,8 @@ export class SmartExtractor { const invalidated = await this.invalidateSupersededMemory(matchId, existing, factKey, created, scopeFilter); await this.notifyPersisted({ text: created.text, category: created.category, scope: created.scope, timestamp: created.timestamp }, "smart-extraction", agentId); if (invalidated) { + // The superseded text can no longer echo; keep the manual ledger honest. + this.config.manualEchoLedger?.invalidate(agentId, existing.text); this.log(`memory-pro: smart-extractor: superseded [${candidate.category}] ${matchId.slice(0, 8)} -> ${created.id.slice(0, 8)}`); return "superseded"; } diff --git a/src/manual-echo-guard.ts b/src/manual-echo-guard.ts index b9a331d3..7c73de04 100644 --- a/src/manual-echo-guard.ts +++ b/src/manual-echo-guard.ts @@ -56,22 +56,35 @@ const DEFAULT_AGENT_BUCKET = "main"; /** * Reporting glue the extractor wraps a dictated fact in ("User stated - * that ..."): articles, demonstratives, pronouns, and reporting verbs. - * Stripped before token comparison so the canonical wrap echo still - * collapses. Only words that carry no assertion of their own belong here. - * Semantic predicates (has, wants, likes, prefers, ...) decide what a - * sentence asserts and stay content tokens; copulas, prepositions and - * conjunctions carry tense, relation and logic and are compared separately - * (ECHO_RELATION_TOKENS). Negation and temporal markers are handled by - * NEGATION_AND_TEMPORAL_MARKERS. + * that ..."): articles, demonstratives and reporting verbs. Stripped before + * token comparison so the canonical wrap echo still collapses. Only words + * that carry no assertion of their own belong here. Semantic predicates + * (has, wants, likes, prefers, ...) decide what a sentence asserts and stay + * content tokens; copulas, prepositions and conjunctions carry tense, + * relation and logic and are compared separately (ECHO_RELATION_TOKENS). + * Pronouns carry a referent and are content too, except for the one + * perspective transform the wrap performs (see SELF_REFERENCE_TOKENS). + * Negation and temporal markers are handled by NEGATION_AND_TEMPORAL_MARKERS. */ const ECHO_GLUE = new Set([ - "the", "a", "an", "that", "this", "these", "those", "it", "its", "their", - "they", "he", "she", "his", "her", "them", "i", "my", "me", "we", "our", - "you", "your", "user", "users", "stated", "said", "says", "saying", - "mentioned", "noted", "also", + "the", "a", "an", "that", "this", "these", "those", + "stated", "said", "says", "saying", "mentioned", "noted", "also", ]); +/** + * The wrap rewrites the speaker as "User" ("my laptop" -> "User's laptop", + * "I like tea" -> "User stated they like tea"): the speaker is implicit on + * both sides, so first-person references and the user label are dropped. + * Any other pronoun names a referent of its own and stays content, except + * inside a User-wrapped text, where it is the wrap's back-reference to the + * user ("User stated their favorite teacup ..."). Swapping a referent + * anywhere else ("his manager" against "her manager", "their laptop" + * against "my laptop") is a different fact, never an echo. + */ +const SPEAKER_TOKENS = new Set(["i", "me", "my", "mine", "we", "us", "our", "ours", "user", "users"]); +const REFERENT_PRONOUNS = new Set(["he", "she", "his", "her", "hers", "they", "them", "their", "theirs", "it", "its", "you", "your", "yours"]); +const USER_WRAP_TOKENS = new Set(["user", "users"]); + /** * Relation-bearing function words. They are not content (a wrap echo may * legitimately add "is" when it turns "teacup: the red one" into "teacup is @@ -99,13 +112,6 @@ const NEGATION_AND_TEMPORAL_MARKERS = new Set([ "except", "without", "before", "after", "used", ]); -/** Conservative CJK marker fragments (negation / bounded validity). */ -const CJK_MARKER_FRAGMENTS = [ - "不", "没", "别", "未", "无", "非", "勿", "直到", "之前", "以前", "除非", - "ない", "じゃない", "ではない", "まで", "もう", - "않", "안", "까지", "전에", -]; - const CJK_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u; export function normalizeEchoText(text: string): string { @@ -121,9 +127,17 @@ function tokenList(normalized: string): string[] { } function orderedContentTokens(normalized: string): string[] { + const tokens = tokenList(normalized); + const userWrapped = tokens.some((token) => USER_WRAP_TOKENS.has(token)); const out: string[] = []; - for (const token of tokenList(normalized)) { + for (const token of tokens) { if (token.length <= 1 && !CJK_RE.test(token)) continue; + if (SPEAKER_TOKENS.has(token)) continue; + if (REFERENT_PRONOUNS.has(token)) { + if (userWrapped) continue; + out.push(token); + continue; + } if (ECHO_GLUE.has(token) || ECHO_RELATION_TOKENS.has(token)) continue; out.push(token); } @@ -177,10 +191,6 @@ function markerAsymmetry(aTokens: string[], bTokens: string[]): boolean { return false; } -function containsCjkMarker(fragment: string): boolean { - return CJK_MARKER_FRAGMENTS.some((marker) => fragment.includes(marker)); -} - /** * The ONLY residual a wrapped CJK echo may carry: reporting glue and * particles. Anything else in the residual — a marker, a verb, a new fact @@ -210,21 +220,19 @@ function isCjkEcho(candidate: string, manual: string): boolean { const man = manual.replace(/\s+/g, ""); if (cand.length === 0 || man.length === 0) return false; if (cand === man) return true; - // Shortened echo: the candidate re-states a piece of the manual text. The - // REMOVED part must carry no marker: stripping 用户不 off 用户不喜欢喝茶和咖啡 - // yields the OPPOSITE claim, not an echo of it. - if (cand.length >= MIN_CJK_CONTAINMENT_CHARS && man.includes(cand)) { - const removed = man.replace(cand, ""); - return !containsCjkMarker(removed); - } // Wrapped echo: the candidate is the manual text plus reporting glue - // (用户说…). The residual must consist ONLY of known glue fragments — + // (用户说…). The residual must consist ONLY of known glue fragments; // being short and marker-free is not enough, since a three-character // residual can be a brand-new fact (并养猫). if (man.length >= MIN_CJK_CONTAINMENT_CHARS && cand.includes(man)) { const residual = cand.replace(man, ""); return residual.length <= MAX_CJK_WRAPPER_RESIDUAL_CHARS && isCjkGlueOnly(residual); } + // A candidate contained inside the manual text is ambiguous without word + // boundaries: 用户喜欢机器学习 inside 用户喜欢机器学习课程 drops the object, + // and with whitespace removed a Latin fragment matches inside a longer + // word (cat inside catalog). Partial CJK containment therefore fails open: + // a duplicate is left for the dedup lane, a different fact is never lost. return false; } diff --git a/src/smart-extractor.ts b/src/smart-extractor.ts index 9f9159ff..b456f7f5 100644 --- a/src/smart-extractor.ts +++ b/src/smart-extractor.ts @@ -3423,6 +3423,8 @@ export class SmartExtractor { ); if (invalidated) { + // The superseded text can no longer echo; keep the manual ledger honest. + this.config.manualEchoLedger?.invalidate(agentId, existing.text); this.log( `memory-pro: smart-extractor: superseded [${candidate.category}] ${matchId.slice(0, 8)} -> ${created.id.slice(0, 8)}`, ); diff --git a/test/manual-echo-guard.test.mjs b/test/manual-echo-guard.test.mjs index 2b5055de..aed324ce 100644 --- a/test/manual-echo-guard.test.mjs +++ b/test/manual-echo-guard.test.mjs @@ -195,6 +195,32 @@ describe("isNearIdenticalEcho", () => { ); }); + it("keeps a candidate that swaps a third-person referent", () => { + assert.equal( + isNearIdenticalEcho("her manager approved the budget increase", "his manager approved the budget increase"), + false, + "his and her name different people", + ); + assert.equal( + isNearIdenticalEcho("my laptop runs the nightly job", "their laptop runs the nightly job"), + false, + "their laptop is somebody else's laptop", + ); + }); + + it("still collapses the first-person to User perspective transform", () => { + assert.equal( + isNearIdenticalEcho("User's laptop runs the nightly job", "my laptop runs the nightly job"), + true, + "the wrap rewrites the speaker as User; that is the one referent change it may make", + ); + assert.equal( + isNearIdenticalEcho("User stated that user prefers oat milk in coffee", "I prefer oat milk in coffee"), + false, + "an inflection change is still a different token sequence (conservative by design)", + ); + }); + it("rejects unrelated candidates", () => { assert.equal( isNearIdenticalEcho("user's dog is named Biscuit", manual), @@ -231,8 +257,24 @@ describe("isNearIdenticalEcho", () => { assert.equal(isNearIdenticalEcho(`用户说${manualCjk}`, manualCjk), true); }); - it("matches a shortened CJK echo", () => { - assert.equal(isNearIdenticalEcho("最喜欢的茶杯是红色", manualCjk), true); + it("fails open on a shortened CJK candidate (partial containment has no word boundary to trust)", () => { + assert.equal(isNearIdenticalEcho("最喜欢的茶杯是红色", manualCjk), false); + }); + + it("keeps a pure-CJK candidate that drops the object of the manual fact", () => { + assert.equal( + isNearIdenticalEcho("用户喜欢机器学习", "用户喜欢机器学习课程"), + false, + "liking machine learning is not liking the machine learning course", + ); + }); + + it("keeps a mixed-script candidate whose Latin word is a prefix of the manual word", () => { + assert.equal( + isNearIdenticalEcho("用户 likes cat", "用户 likes catalog"), + false, + "with whitespace removed, cat sits inside catalog; that is not an echo", + ); }); it("keeps a temporally qualified CJK candidate", () => {