diff --git a/internal/agent/playbook.go b/internal/agent/playbook.go index a248e8387..d2e42fd59 100644 --- a/internal/agent/playbook.go +++ b/internal/agent/playbook.go @@ -117,20 +117,28 @@ func (pb *Playbook) save() { // classifyTaskType determines the task category from the user prompt. // Order matters: more specific categories are checked first to avoid // misclassification (e.g., "add test" should be "test" not "feature"). +// Keywords are matched on word boundaries (#2745): bare substring matching +// classified "upgrade to the latest version" as test (laTEST), "refactor +// the contest module" as test (conTEST), "address the failing build" as +// feature (ADDRESS), and "rebuild the parser" as build (REBUILD) - the +// misclassified type then polluted the persisted playbook fingerprint and +// system prompt injection. Keywords written with an explicit leading or +// trailing space (" fail", "make ", "ci ", "new ") already encode their +// own anchoring and keep the substring behavior. func classifyTaskType(userPrompt string) string { p := strings.ToLower(userPrompt) switch { - case containsAny(p, "test", "spec", "coverage", "mock"): + case containsAnyWord(p, "test", "spec", "coverage", "mock"): return "test" - case containsAny(p, "build", "compile", "make ", "ci ", "deploy", "release", "publish"): + case containsAnyWord(p, "build", "compile", "make ", "ci ", "deploy", "release", "publish"): return "build" - case containsAny(p, "fix", "bug", "error", "crash", "broken", " fail", "panic", "traceback"): + case containsAnyWord(p, "fix", "bug", "error", "crash", "broken", " fail", "panic", "traceback"): return "bugfix" - case containsAny(p, "refactor", "clean", "rename", "reorganize", "simplify", "extract"): + case containsAnyWord(p, "refactor", "clean", "rename", "reorganize", "simplify", "extract"): return "refactor" - case containsAny(p, "review", "check", "audit", "inspect", "scan", "analyze"): + case containsAnyWord(p, "review", "check", "audit", "inspect", "scan", "analyze"): return "review" - case containsAny(p, "add", "implement", "create", "new ", "support"): + case containsAnyWord(p, "add", "implement", "create", "new ", "support"): return "feature" default: return "other" @@ -428,6 +436,45 @@ func containsAny(s string, substrs ...string) bool { return false } +// containsAnyWord matches whole words only (#2745). A keyword containing +// an explicit space (" fail", "make ", "new ") encodes its own anchoring +// and falls back to substring matching, preserving the original intent. +func containsAnyWord(s string, keywords ...string) bool { + for _, kw := range keywords { + if strings.ContainsAny(kw, " \t") { + if strings.Contains(s, kw) { + return true + } + continue + } + start := 0 + for { + i := strings.Index(s[start:], kw) + if i < 0 { + break + } + at := start + i + end := at + len(kw) + if wordBoundaryAt(s, at) && wordBoundaryAt(s, end) { + return true + } + start = at + 1 + } + } + return false +} + +// wordBoundaryAt reports whether position i in s is a word boundary: +// either string edge, or the neighboring bytes are not word bytes on both +// sides of the boundary. Reuses isWordByte from success_declare.go +// (identifier semantics: [a-z0-9_]). +func wordBoundaryAt(s string, i int) bool { + if i <= 0 || i >= len(s) { + return true + } + return !isWordByte(s[i-1]) || !isWordByte(s[i]) +} + func randomID() string { b := make([]byte, 6) rand.Read(b) diff --git a/internal/agent/zz_issue2745_test.go b/internal/agent/zz_issue2745_test.go new file mode 100644 index 000000000..90d1c6e4a --- /dev/null +++ b/internal/agent/zz_issue2745_test.go @@ -0,0 +1,54 @@ +package agent + +// #2745 regression: classifyTaskType must match keywords on word +// boundaries - bare substring matching misclassified prompts whose words +// merely CONTAIN a keyword (laTEST, conTEST, ADDRESS, REBUILD, +// CHECKlist), polluting the persisted playbook fingerprint and system +// prompt injection downstream. + +import "testing" + +func TestIssue2745_ClassifyTaskTypeWordBoundary(t *testing.T) { + cases := []struct { + prompt, want string + }{ + // The five issue scenarios: substring false-positives are gone. + // (Category per current switch precedence - the fix removes the WRONG + // hit; sentences with no whole keyword left classify "other".) + {"Upgrade the dependency to the latest version", "other"}, // was test (la-test) + {"Refactor the contest scoring module", "refactor"}, // was test (con-test) + {"Create a checklist for onboarding", "feature"}, // was review (check) + {"Address the failing build", "build"}, // was feature (add-ress); whole-word "build" wins by precedence + {"rebuild the parser", "other"}, // was build (re-build); no whole keyword left + // Boundary sanity: real keywords at word edges still classify. + {"run the test suite", "test"}, + {"tests are failing", "bugfix"}, // "test" lacks a boundary inside "tests"; space-anchored " fail" hits bugfix + {"fix the login bug", "bugfix"}, + {"build the project", "build"}, + {"deploy to prod", "build"}, + {"review this diff", "review"}, + {"check the logs", "review"}, + {"add a settings page", "feature"}, + {"create a new module", "feature"}, + {"refactor the store", "refactor"}, + } + for _, tc := range cases { + if got := classifyTaskType(tc.prompt); got != tc.want { + t.Errorf("classifyTaskType(%q) = %q, want %q", tc.prompt, got, tc.want) + } + } +} + +func TestIssue2745_SpaceAnchoredKeywordsKeepSubstringSemantics(t *testing.T) { + // Keywords written with explicit spaces (" fail", "make ", "ci ", + // "new ") encode their own anchoring and keep substring behavior. + if got := classifyTaskType("the daemon will fail soon"); got != "bugfix" { + t.Errorf("space-anchored ' fail' must still match bugfix, got %q", got) + } + if got := classifyTaskType("run make all"); got != "build" { + t.Errorf("space-anchored 'make ' must still match build, got %q", got) + } + if got := classifyTaskType("new feature request"); got != "feature" { + t.Errorf("space-anchored 'new ' must still match feature, got %q", got) + } +}