From d73cfe2e6541b2c3b7e285de9f20861253ecf3f4 Mon Sep 17 00:00:00 2001 From: Junjun Zhang Date: Fri, 25 Sep 2026 12:03:12 +0800 Subject: [PATCH] feat(agent)!: remove FinOps heterogeneous-model guidance injection The [FinOps Guidance] block appended to tool results recommended Claude/GPT model tiers unconditionally, ignoring the active provider (it fired on GLM-based sessions where those models are not selectable). Removed the sa-131 module, its design doc, and all six wiring points, per direct user directive. Co-Authored-By: ggcode --- docs/design/heterogeneous-model-guide.md | 108 --------- internal/agent/agent.go | 13 - internal/agent/guidance_compact_reset.go | 7 - internal/agent/guidance_compact_reset_test.go | 20 +- internal/agent/heterogeneous_model_guide.go | 224 ------------------ .../agent/heterogeneous_model_guide_test.go | 162 ------------- 6 files changed, 8 insertions(+), 526 deletions(-) delete mode 100644 docs/design/heterogeneous-model-guide.md delete mode 100644 internal/agent/heterogeneous_model_guide.go delete mode 100644 internal/agent/heterogeneous_model_guide_test.go diff --git a/docs/design/heterogeneous-model-guide.md b/docs/design/heterogeneous-model-guide.md deleted file mode 100644 index 71f7aad4f..000000000 --- a/docs/design/heterogeneous-model-guide.md +++ /dev/null @@ -1,108 +0,0 @@ -# Heterogeneous Model Selection Guide (sa-131) - -## Research Basis - -2025-2026 AI Agent trends (Deloitte, Machine Learning Mastery) identify **FinOps for AI Agents** as a critical frontier concept. The economic imperative is to use heterogeneous model architectures: - -- **Frontier models** (Claude Opus, GPT-4) for complex reasoning -- **Mid-tier models** for standard tasks -- **Small language models** for high-frequency execution - -The **Plan-and-Execute pattern** reduces costs by 90% compared to using frontier models for everything (capable model plans, cheaper models execute). - -## Implementation - -### Location -- `internal/agent/heterogeneous_model_guide.go` (170 lines) -- `internal/agent/heterogeneous_model_guide_test.go` (112 lines) -- Integrated in `internal/agent/agent.go` (12 lines added) - -### How It Works - -#### 1. Tool Classification -Tools are categorized into 6 types: - -- **Read**: `read_file`, `multi_file_read` -- **Write**: `edit_file`, `write_file`, `multi_file_edit` -- **Search**: `web_search`, `code_search`, `grep` -- **Reasoning**: `lsp_*` tools (semantic understanding) -- **Execution**: `run_command`, `start_command`, git operations, `file_ops` -- **Other**: everything else - -#### 2. Workload Analysis -After minimum 5 tool calls, analyzes the pattern: - -- **Execution-heavy** (>70% read/write/search/execution): Suggests cheaper model -- **Reasoning-heavy** (>50% LSP tools): No warning (justifies frontier model) -- **Mixed**: No action - -#### 3. Guidance Injection -Non-blocking advisory guidance is injected into tool results when execution-heavy pattern is detected: - -``` -[FinOps Guidance: Heterogeneous Model Selection] - -Your recent actions show an execution-heavy pattern (read/write/search/execution tools dominate). - -Cost Optimization Opportunity: -Consider using a cost-effective model tier for routine file operations: -- Frontier models (Claude Opus, GPT-4): Best for complex reasoning, planning, architectural decisions -- Mid-tier models (Claude Sonnet, GPT-4o-mini): Good balance for standard coding tasks -- Small models: Sufficient for repetitive file edits, grep searches, basic text operations - -Current pattern appears to be routine execution rather than deep reasoning. -If this is primarily mechanical work, you could reduce token costs by 60-90% -while maintaining quality. - -This is guidance only - proceed with the current model if this task requires -frontier-level reasoning capability. -``` - -### Design Decisions - -1. **Zero LLM cost**: Pure pattern matching, no semantic analysis -2. **Non-blocking**: Advisory only, never prevents action -3. **Fires at most once per run**: Prevents repetitive nagging -4. **10-tool lookback window**: Recent behavior matters more -5. **Reasoning-heavy exclusion**: LSP tools justify frontier model cost -6. **Thread-safe**: Uses mutex for concurrent access - -### Integration Points - -- Initialized in `Agent` struct as `heterogeneousModel` field -- Called in tool execution loop after OOD detection -- Guidance appended to `result.Content` if triggered -- Resets on each new run via `reset()` method - -## Testing - -3 test cases: - -1. **TestHeterogeneousModelGuide**: Verifies execution-heavy pattern triggers guidance -2. **TestHeterogeneousModelReasoningHeavy**: Verifies reasoning-heavy (LSP) doesn't trigger -3. **TestHeterogeneousModelMinActions**: Verifies minimum threshold enforcement - -All tests pass. - -## Cost Intelligence Synergy - -This feature complements existing cost intelligence: -- **Token tracking**: `internal/cost/cost.go` -- **Budget enforcement**: `internal/agent/cost_budget.go` -- **Success declaration**: `internal/agent/success_declare.go` - -Together they provide multi-layered FinOps awareness: tracking actual spend, enforcing limits, and suggesting cost-effective model selection. - -## Future Enhancements - -Potential improvements: -1. Learn from historical cost-effectiveness patterns -2. Auto-suggest model switching via config -3. Track actual cost savings realized -4. Integrate with provider's model catalog APIs - -## References - -- Deloitte 2025 AI Agent Trends Report -- Machine Learning Mastery: "FinOps for AI Agents" (2025) -- Plan-and-Execute: arXiv:2308.14342 (2023) diff --git a/internal/agent/agent.go b/internal/agent/agent.go index 84ca04d59..29e00de38 100644 --- a/internal/agent/agent.go +++ b/internal/agent/agent.go @@ -339,7 +339,6 @@ type Agent struct { taintInfluence *taintInfluenceState // tainted data influence detection (IFC: tracks untrusted content flowing into privileged tool calls) falsePremise *falsePremiseState // false premise detection: ungrounded success claims contradicting tool errors (world-model drift) perfBaseline *perfBaselineState // cross-session performance regression detection - heterogeneousModel *heterogeneousModelState // FinOps: heterogeneous model selection guidance (sa-131) lastRunStats *RunStats // stats from the most recent run (for post-run summary display) qualityScorer *ResponseQualityScorer // per-run response quality scoring for provider/model A/B comparison systemPromptInjector func() string // returns extra system prompt text to inject (e.g. lanchat peer warnings) @@ -463,7 +462,6 @@ func NewAgent(p provider.Provider, tools *tool.Registry, systemPrompt string, ma toolEquivDetect: newToolEquivDetectState(), bgOrphan: newBgOrphanState(), actionAnnihil: newActionAnnihilateState(), - heterogeneousModel: newHeterogeneousModelState(), exploreFrag: newExploreFragState(), batchCoupling: newBatchCouplingState(), buildIdempot: newBuildIdempotencyState(), @@ -1377,11 +1375,6 @@ func (a *Agent) RunStreamWithContent(ctx context.Context, content []provider.Con a.buildIdempot.reset() a.orphanFile.reset() a.cfDep.reset() - // #1466-A: the per-run reset block's own #677 note lists the - // same-family misses it fixed - heterogeneousModel was missed too: - // hmMaxWarns=1 burned in run 1 kept the detector silent for every - // later run of the Agent's lifetime. - a.heterogeneousModel.reset() // #1843 case 1: foresightCalib.reset() was never called outside // compaction - "at most 2 per run" (file-header promise) was in fact // per-LIFETIME: mismatches and warnCount accumulated across every @@ -3904,12 +3897,6 @@ func (a *Agent) RunStreamWithContent(ctx context.Context, content []provider.Con // Query convergence tracking: record search queries and code // actions to detect repeated similar searches without progress. a.queryConverge.recordToolCall(tc.Name, string(tc.Arguments), i+1) - // cost-effective model tier selection. Detects execution-heavy - // patterns and suggests using cheaper models for routine work. - // Research basis: 2025-2026 AI Agent trends (Deloitte, Machine Learning Mastery) - if hmGuidance := a.heterogeneousModel.recordToolCall(tc.Name, i+1); hmGuidance != "" { - a.appendGuidance(&result, hmGuidance) - } // Plan drift capture: when exit_plan_mode fires, extract plan items // for later drift detection (spec-driven development tracking). if tc.Name == "exit_plan_mode" { diff --git a/internal/agent/guidance_compact_reset.go b/internal/agent/guidance_compact_reset.go index b872ca070..6329b15ba 100644 --- a/internal/agent/guidance_compact_reset.go +++ b/internal/agent/guidance_compact_reset.go @@ -515,13 +515,6 @@ var guidanceCounterResets = []func(*Agent){ a.phantomVerify.warnings = 0 } }, - func(a *Agent) { - if a.heterogeneousModel != nil { - a.heterogeneousModel.mu.Lock() - a.heterogeneousModel.warnsIssued = 0 - a.heterogeneousModel.mu.Unlock() - } - }, func(a *Agent) { if a.selfMod != nil { a.selfMod.mu.Lock() diff --git a/internal/agent/guidance_compact_reset_test.go b/internal/agent/guidance_compact_reset_test.go index e8b457c82..c7188a757 100644 --- a/internal/agent/guidance_compact_reset_test.go +++ b/internal/agent/guidance_compact_reset_test.go @@ -105,15 +105,14 @@ func TestBClassDetectorsOncePerRun(t *testing.T) { // (int counter, quota map, quota bools) must all reset. func TestResetGuidanceCounters1651Extension(t *testing.T) { a := &Agent{ - trajectoryHealth: &trajectoryHealthState{warnings: 2}, - planAbandon: &planAbandonState{warnings: 1}, - delegationOrch: &delegationState{orphanWarnCount: 1, serialWarnCount: 2, overDelWarned: true}, - fixAmnesia: newFixAmnesiaState(), - heterogeneousModel: &heterogeneousModelState{warnsIssued: 1}, - driftRecurrence: &driftRecurrenceState{fired: true, warned: true}, - crossFileImpact: &crossFileImpactState{fired: true}, - serialRead: &serialReadState{fired: true}, - diskSpace: &diskSpaceState{fired: true}, + trajectoryHealth: &trajectoryHealthState{warnings: 2}, + planAbandon: &planAbandonState{warnings: 1}, + delegationOrch: &delegationState{orphanWarnCount: 1, serialWarnCount: 2, overDelWarned: true}, + fixAmnesia: newFixAmnesiaState(), + driftRecurrence: &driftRecurrenceState{fired: true, warned: true}, + crossFileImpact: &crossFileImpactState{fired: true}, + serialRead: &serialReadState{fired: true}, + diskSpace: &diskSpaceState{fired: true}, } a.fixAmnesia.mu.Lock() a.fixAmnesia.warned["build"] = true @@ -128,9 +127,6 @@ func TestResetGuidanceCounters1651Extension(t *testing.T) { if len(a.fixAmnesia.warned) != 0 { t.Fatal("fixAmnesia warned map must clear") } - if a.heterogeneousModel.warnsIssued != 0 { - t.Fatal("heterogeneousModel quota must reset") - } if a.driftRecurrence.fired || a.driftRecurrence.warned || a.crossFileImpact.fired || a.serialRead.fired || a.diskSpace.fired { t.Fatal("quota bools must reset") } diff --git a/internal/agent/heterogeneous_model_guide.go b/internal/agent/heterogeneous_model_guide.go deleted file mode 100644 index ebe452e26..000000000 --- a/internal/agent/heterogeneous_model_guide.go +++ /dev/null @@ -1,224 +0,0 @@ -package agent - -import ( - "sync" -) - -// Heterogeneous Model Selection Guide -// -// Research basis: 2025-2026 AI Agent trends (Deloitte, Machine Learning Mastery) -// identify FinOps for AI Agents as a critical frontier concept. The economic -// imperative is to use heterogeneous model architectures: -// - Expensive frontier models (Claude Opus, GPT-4) for complex reasoning -// - Mid-tier models for standard tasks -// - Small language models for high-frequency execution -// -// The Plan-and-Execute pattern reduces costs by 90% compared to using -// frontier models for everything (capable model plans, cheaper models execute). -// -// This module provides heuristic guidance to help agents make cost-effective -// model choices at runtime. It detects when the agent is: -// 1. Doing deep reasoning (justifies frontier model cost) -// 2. Doing routine execution (should use cheaper model) -// -// Design: -// - Zero LLM cost - pure pattern matching on tool usage -// - Non-blocking: advisory guidance only -// - Fires at most once per run -// - Resets each run - -const ( - // hmMinToolActions is the minimum number of tool calls to trigger analysis. - hmMinToolActions = 5 - - // hmExecutionThreshold: if >70% of tools are read/search/grep/edit, it's execution - hmExecutionThreshold = 0.70 - - // hmReasoningThreshold: if >50% are lsp/codex/planning tools, it's reasoning - hmReasoningThreshold = 0.50 - - // hmLookbackWindow: number of recent tool calls to analyze - hmLookbackWindow = 10 - - hmMaxWarns = 1 -) - -// Tool categories for heterogeneous model guidance -type hmToolCategory int - -const ( - hmCategoryRead hmToolCategory = iota // read_file, search_files, grep - hmCategoryWrite // edit_file, write_file, multi_file_edit - hmCategorySearch // web_search, code_search - hmCategoryReasoning // lsp_* tools requiring semantic understanding - hmCategoryExecution // shell commands, git operations, file ops - hmCategoryOther // everything else -) - -// hmClassifyTool returns the category of a tool call. -func hmClassifyTool(toolName string) hmToolCategory { - switch { - case toolName == "read_file" || toolName == "multi_file_read": - return hmCategoryRead - case toolName == "edit_file" || toolName == "write_file" || toolName == "multi_file_edit" || toolName == "multi_edit_file": - return hmCategoryWrite - case toolName == "web_search" || toolName == "code_search" || toolName == "search_files" || toolName == "grep": - return hmCategorySearch - case len(toolName) >= 4 && toolName[:4] == "lsp_": - return hmCategoryReasoning - case toolName == "run_command" || toolName == "start_command" || toolName == "git_add" || - toolName == "git_commit" || toolName == "git_checkout" || toolName == "git_stash" || - toolName == "file_ops": - return hmCategoryExecution - default: - return hmCategoryOther - } -} - -// hmToolRecord records a single tool call for analysis. -type hmToolRecord struct { - tool string - category hmToolCategory - iteration int -} - -// heterogeneousModelState tracks FinOps heterogeneous model guidance. -type heterogeneousModelState struct { - mu sync.Mutex - totalTools int - categoryCounts map[hmToolCategory]int - toolHistory []hmToolRecord - warnsIssued int - guidance string -} - -// recordToolCall records a tool call and checks if guidance should be emitted. -// Returns guidance text if triggered, empty string otherwise. -func (s *heterogeneousModelState) recordToolCall(toolName string, iteration int) string { - // Record this tool call - s.mu.Lock() - defer s.mu.Unlock() - - cat := hmClassifyTool(toolName) - s.categoryCounts[cat]++ - s.totalTools++ - - s.toolHistory = append(s.toolHistory, hmToolRecord{ - tool: toolName, - category: cat, - iteration: iteration, - }) - // Keep only the lookback window. #2427: trim AFTER append so the - // steady-state window is exactly hmLookbackWindow entries (the old - // trim-before-append with `>` left an 11-entry window). - if len(s.toolHistory) > hmLookbackWindow { - s.toolHistory = s.toolHistory[len(s.toolHistory)-hmLookbackWindow:] - } - - // Only check after minimum actions - if s.totalTools < hmMinToolActions { - return "" - } - - // Check if we should warn (at most once) - if s.warnsIssued >= hmMaxWarns { - return "" - } - - // Calculate ratios over the sliding window (#2427): the documented - // semantics — "recent tool calls", not lifetime cumulative. A lifetime - // ratio let an early exploration phase (glob/list/todo calls in the - // denominator but not the exec numerator) permanently dilute a later - // pure-execution burst: 8 explore + 12 exec = 0.60 cumulative, below - // the 0.70 threshold forever, while the window is 100% exec — the - // module's own Plan-and-Execute flagship scenario never fired. - // categoryCounts stays maintained for information only. - windowExec, windowReasoning := 0, 0 - for _, rec := range s.toolHistory { - switch rec.category { - case hmCategoryRead, hmCategoryWrite, hmCategorySearch, hmCategoryExecution: - windowExec++ - case hmCategoryReasoning: - windowReasoning++ - } - } - windowLen := len(s.toolHistory) - execRatio := float64(windowExec) / float64(windowLen) - reasoningRatio := float64(windowReasoning) / float64(windowLen) - - // Determine workload type - if execRatio >= hmExecutionThreshold { - s.warnsIssued++ - guidance := hmGenerateGuidance("execution-heavy") - s.guidance = guidance - return guidance - } - if reasoningRatio >= hmReasoningThreshold { - // Reasoning-heavy doesn't trigger a warning - it justifies frontier model - return "" - } - - return "" -} - -// hmGenerateGuidance generates appropriate guidance based on workload type. -func hmGenerateGuidance(workloadType string) string { - if workloadType == "execution-heavy" { - return `[FinOps Guidance: Heterogeneous Model Selection] - -Your recent actions show an execution-heavy pattern (read/write/search/execution tools dominate). - -Cost Optimization Opportunity: -Consider using a cost-effective model tier for routine file operations: -- Frontier models (Claude Opus, GPT-4): Best for complex reasoning, planning, architectural decisions -- Mid-tier models (Claude Sonnet, GPT-4o-mini): Good balance for standard coding tasks -- Small models: Sufficient for repetitive file edits, grep searches, basic text operations - -Current pattern appears to be routine execution rather than deep reasoning. -If this is primarily mechanical work, you could reduce token costs by 60-90% -while maintaining quality. - -This is guidance only - proceed with the current model if this task requires -frontier-level reasoning capability.` - } - return "" -} - -// newHeterogeneousModelState creates a new heterogeneous model guide instance. -func newHeterogeneousModelState() *heterogeneousModelState { - return &heterogeneousModelState{ - categoryCounts: make(map[hmToolCategory]int), - } -} - -// reset clears accumulated state for a new run. -func (s *heterogeneousModelState) reset() { - s.mu.Lock() - defer s.mu.Unlock() - s.totalTools = 0 - s.categoryCounts = make(map[hmToolCategory]int) - s.toolHistory = nil - s.warnsIssued = 0 - s.guidance = "" -} - -// GetHeterogeneousModelGuidance returns any pending heterogeneous model guidance. -func (a *Agent) GetHeterogeneousModelGuidance() string { - if a.heterogeneousModel == nil { - return "" - } - // #2427: guidance is written under s.mu by recordToolCall — take the - // same lock instead of an unlocked read (torn string header race). - a.heterogeneousModel.mu.Lock() - defer a.heterogeneousModel.mu.Unlock() - return a.heterogeneousModel.guidance -} - -// ClearHeterogeneousModelGuidance clears the pending guidance after it's been consumed. -func (a *Agent) ClearHeterogeneousModelGuidance() { - if a.heterogeneousModel != nil { - a.heterogeneousModel.mu.Lock() - a.heterogeneousModel.guidance = "" - a.heterogeneousModel.mu.Unlock() - } -} diff --git a/internal/agent/heterogeneous_model_guide_test.go b/internal/agent/heterogeneous_model_guide_test.go deleted file mode 100644 index 23dcdb995..000000000 --- a/internal/agent/heterogeneous_model_guide_test.go +++ /dev/null @@ -1,162 +0,0 @@ -package agent - -import "testing" - -func TestHmClassifyTool(t *testing.T) { - tests := []struct { - name string - tool string - expected hmToolCategory - }{ - {"read_file", "read_file", hmCategoryRead}, - {"multi_file_read", "multi_file_read", hmCategoryRead}, - {"edit_file", "edit_file", hmCategoryWrite}, - {"write_file", "write_file", hmCategoryWrite}, - {"multi_file_edit", "multi_file_edit", hmCategoryWrite}, - {"web_search", "web_search", hmCategorySearch}, - {"code_search", "code_search", hmCategorySearch}, - {"lsp_definition", "lsp_definition", hmCategoryReasoning}, - {"lsp_references", "lsp_references", hmCategoryReasoning}, - {"run_command", "run_command", hmCategoryExecution}, - {"start_command", "start_command", hmCategoryExecution}, - {"git_add", "git_add", hmCategoryExecution}, - {"unknown_tool", "unknown_tool", hmCategoryOther}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := hmClassifyTool(tt.tool) - if result != tt.expected { - t.Errorf("hmClassifyTool(%q) = %v, want %v", tt.tool, result, tt.expected) - } - }) - } -} - -func TestHeterogeneousModelGuide(t *testing.T) { - state := newHeterogeneousModelState() - - // Initial state should have no guidance - if state.guidance != "" { - t.Errorf("initial state has guidance: %q", state.guidance) - } - - // Add read/write operations (execution-heavy) - for i := 0; i < 8; i++ { - state.recordToolCall("read_file", 1) - } - state.recordToolCall("edit_file", 2) - state.recordToolCall("grep", 3) - - // Should have issued guidance after hitting threshold - guidance := state.guidance - if guidance == "" { - t.Error("expected guidance after execution-heavy pattern, got empty") - } - if !containsString(guidance, "FinOps Guidance") { - t.Errorf("guidance missing expected prefix, got: %q", guidance) - } - - // Subsequent calls should not re-issue guidance (max 1 per session) - state.reset() - for i := 0; i < 10; i++ { - state.recordToolCall("read_file", 1) - } - // Reset the internal counters for testing - state.warnsIssued = 0 - state.guidance = "" - state.recordToolCall("edit_file", 2) - - // After reset, should fire again - guidance = state.guidance - if guidance == "" { - t.Error("expected guidance after reset, got empty") - } -} - -func TestHeterogeneousModelReasoningHeavy(t *testing.T) { - state := newHeterogeneousModelState() - - // Add LSP tools (reasoning-heavy) - should NOT trigger warning - for i := 0; i < 10; i++ { - state.recordToolCall("lsp_definition", 1) - state.recordToolCall("lsp_references", 2) - } - - // Reasoning-heavy patterns don't trigger FinOps guidance - guidance := state.guidance - if guidance != "" { - t.Errorf("expected no guidance for reasoning-heavy pattern, got: %q", guidance) - } -} - -func TestHeterogeneousModelMinActions(t *testing.T) { - state := newHeterogeneousModelState() - - // Below minimum threshold - should not trigger - for i := 0; i < 3; i++ { - state.recordToolCall("read_file", 1) - } - - guidance := state.guidance - if guidance != "" { - t.Errorf("expected no guidance below minimum action threshold, got: %q", guidance) - } -} - -func containsString(s, substr string) bool { - for i := 0; i <= len(s)-len(substr); i++ { - if s[i:i+len(substr)] == substr { - return true - } - } - return false -} - -// TestHeterogeneousModelMixedExplorationThenExecution pins #2427: the -// documented sliding-window semantics. An early exploration phase (glob/ -// todo calls -> hmCategoryOther, counted in a lifetime denominator but not -// in the exec numerator) used to permanently dilute a later pure-execution -// burst: 8 explore + 12 exec = 0.60 lifetime, below the 0.70 threshold -// forever, so the FinOps downgrade hint was silently dropped in exactly the -// Plan-and-Execute scenario the module cites as its flagship. The window of -// the last 10 calls is 100% exec by call 18, so guidance must fire. -func TestHeterogeneousModelMixedExplorationThenExecution(t *testing.T) { - state := newHeterogeneousModelState() - - // Exploration phase: 8 non-exec, non-reasoning calls. - for i := 0; i < 8; i++ { - if g := state.recordToolCall("glob", 1); g != "" { - t.Fatalf("unexpected guidance during exploration phase at call %d", i+1) - } - } - - // Execution burst: with a true 10-call window the last 10 calls are - // all exec once 10 edit_file calls have accumulated (call 18 overall); - // the old lifetime ratio (12/20 = 0.60) never reached 0.70. - fired := false - for i := 0; i < 12; i++ { - if g := state.recordToolCall("edit_file", 2); g != "" { - fired = true - break - } - } - if !fired { - t.Error("expected sliding-window guidance during pure-execution burst after exploration phase; lifetime-ratio bug (#2427) would drop it") - } -} - -// TestHeterogeneousModelWindowTrimBoundary pins the #2427 off-by-one: the -// old trim-before-append (len > window) left an 11-entry steady-state -// window. After trimming post-append, len(toolHistory) must stay exactly -// hmLookbackWindow. -func TestHeterogeneousModelWindowTrimBoundary(t *testing.T) { - state := newHeterogeneousModelState() - - for i := 0; i < 15; i++ { - state.recordToolCall("read_file", 1) - } - if got := len(state.toolHistory); got != hmLookbackWindow { - t.Errorf("toolHistory length = %d, want exactly %d (off-by-one in trim)", got, hmLookbackWindow) - } -}