|
| 1 | +import { describe, expect, test } from "bun:test"; |
| 2 | +import { |
| 3 | + defaultRow, |
| 4 | + FAMILY_IDS, |
| 5 | + grokRow, |
| 6 | + museRow, |
| 7 | + type PromptVarianceFamily, |
| 8 | +} from "./rows.js"; |
| 9 | + |
| 10 | +// CL-8269 RED: the versioned prompt-variance package does not exist yet, so |
| 11 | +// every test below fails at import time until the GREEN lands it. |
| 12 | + |
| 13 | +describe("prompt-variance family rows", () => { |
| 14 | + test("ships exactly the default/muse/grok families", () => { |
| 15 | + expect([...FAMILY_IDS]).toEqual(["default", "muse", "grok"]); |
| 16 | + }); |
| 17 | + |
| 18 | + test("has no glm/claude/gpt rows until their evals land (CL-8265/7772/7775)", () => { |
| 19 | + for (const id of FAMILY_IDS) { |
| 20 | + expect(["glm", "claude", "gpt"]).not.toContain(id); |
| 21 | + } |
| 22 | + }); |
| 23 | + |
| 24 | + test("every row carries the render/residual/deny/omit shape", () => { |
| 25 | + for (const row of [defaultRow, museRow, grokRow]) { |
| 26 | + expect(row.render).toBe("tail"); |
| 27 | + expect(typeof row.residual).toBe("string"); |
| 28 | + expect(Array.isArray([...row.advertisedToolDeny])).toBe(true); |
| 29 | + expect(Array.isArray([...row.sectionOmit])).toBe(true); |
| 30 | + } |
| 31 | + }); |
| 32 | + |
| 33 | + test("row ids match their family", () => { |
| 34 | + const ids: PromptVarianceFamily[] = [ |
| 35 | + defaultRow.id, |
| 36 | + museRow.id, |
| 37 | + grokRow.id, |
| 38 | + ]; |
| 39 | + expect(ids).toEqual(["default", "muse", "grok"]); |
| 40 | + }); |
| 41 | + |
| 42 | + test("muse row is the shipped CL-7869 tool-discipline text", () => { |
| 43 | + expect(museRow.residual).toContain("Tool discipline:"); |
| 44 | + expect(museRow.residual).toContain("Batch independent tool calls"); |
| 45 | + expect(museRow.residual).toContain("Never re-read a file"); |
| 46 | + expect(museRow.residual).toContain("Do not narrate; act."); |
| 47 | + }); |
| 48 | + |
| 49 | + test("grok row is the existing finish-bias residual", () => { |
| 50 | + expect(grokRow.residual).toContain("Finish bias (xAI / Grok worker):"); |
| 51 | + expect(grokRow.residual).toContain("prefer the structured report"); |
| 52 | + expect(grokRow.residual).toContain("re-open paths you already read"); |
| 53 | + expect(grokRow.residual).toContain("done-definition is met"); |
| 54 | + expect(grokRow.residual).toContain("never run_shell"); |
| 55 | + }); |
| 56 | + |
| 57 | + test("default row carries no residual", () => { |
| 58 | + expect(defaultRow.residual).toBe(""); |
| 59 | + }); |
| 60 | + |
| 61 | + test("grok leaves deny skill_search only; other rows deny nothing", () => { |
| 62 | + expect([...grokRow.advertisedToolDeny]).toEqual(["skill_search"]); |
| 63 | + expect([...defaultRow.advertisedToolDeny]).toEqual([]); |
| 64 | + expect([...museRow.advertisedToolDeny]).toEqual([]); |
| 65 | + }); |
| 66 | + |
| 67 | + test("no row denies use_skill — brief-named skills always load directly", () => { |
| 68 | + for (const row of [defaultRow, museRow, grokRow]) { |
| 69 | + expect(row.advertisedToolDeny).not.toContain("use_skill"); |
| 70 | + } |
| 71 | + }); |
| 72 | + |
| 73 | + test("sectionOmit starts empty on every row", () => { |
| 74 | + for (const row of [defaultRow, museRow, grokRow]) { |
| 75 | + expect([...row.sectionOmit]).toEqual([]); |
| 76 | + } |
| 77 | + }); |
| 78 | +}); |
0 commit comments