Skip to content

Commit 3cace90

Browse files
feat(builder): ship a short Corbits implement card (#1149)
The implement-skill essay and philosophy boot duplicated the worker contract. Family residuals stay in prompt-variance.
1 parent 29611f3 commit 3cace90

3 files changed

Lines changed: 64 additions & 124 deletions

File tree

‎src/agent/directors/builder/package.test.ts‎

Lines changed: 50 additions & 66 deletions
Original file line numberDiff line numberDiff line change
@@ -9,63 +9,72 @@ describe("builderPackage", () => {
99
expect(p).not.toMatch(/build director/i);
1010
});
1111

12-
test("systemPrompt teaches implement-and-test from the implement skill", () => {
12+
test("systemPrompt is a short Corbits implement card, not the GaaS implement skill", () => {
1313
const p = builderPackage.systemPrompt;
14-
expect(p).toContain("Implement and Test");
14+
expect(p).toContain("Ship the brief");
1515
expect(p).toContain("success_criteria");
16-
expect(p).toMatch(/bug fixes \(test-first\)/i);
17-
expect(p).toMatch(/reproduces the bug/i);
18-
expect(p).toMatch(/verify it \*\*fails\*\*/);
19-
expect(p).toMatch(/For new features/i);
20-
expect(p).toMatch(/assert.*expected behavior|works as designed/i);
16+
expect(p).toMatch(/test-first/i);
17+
expect(p).toMatch(/assert expected behavior/i);
2118
expect(p).toContain("Blockers");
19+
expect(p).not.toContain("## Implement and Test");
20+
expect(p).not.toContain("## Prerequisites");
21+
expect(p).not.toContain("## Build Gate");
22+
expect(p).not.toContain("Greybeard Review");
23+
expect(p).not.toContain("TaskCreate");
24+
expect(p).not.toMatch(/Use the @greybeard subagent/i);
2225
});
2326

24-
test("systemPrompt teaches Build Gate and does not shortcut verify", () => {
27+
test("systemPrompt ships tests with the change and runs the repo gate", () => {
2528
const p = builderPackage.systemPrompt;
26-
expect(p).toContain("Build Gate");
2729
expect(p).toMatch(/bun run check/);
28-
expect(p).toMatch(/Don't shortcut verify/i);
30+
expect(p).toMatch(/Do not shortcut verify/i);
2931
expect(p).toMatch(/partial gates/i);
3032
expect(p).toMatch(/pre-existing/i);
31-
expect(p).toMatch(
32-
/defined typecheck command.*relevant tests.*defined full check/is,
33-
);
34-
expect(p).toMatch(
35-
/repository defines no typecheck command.*explicit Blocker/is,
33+
expect(p).toMatch(/do not invent one/i);
34+
expect(p).toMatch(/exact verification command/i);
35+
expect(p).toMatch(/exit status/);
36+
expect(p).toMatch(/bare "pass" without command evidence/i);
37+
expect(p).toMatch(/same commit when committing/);
38+
});
39+
40+
test("systemPrompt has no philosophy boot", () => {
41+
const p = builderPackage.systemPrompt;
42+
expect(p).not.toMatch(/Before substantial repo work/i);
43+
expect(p).not.toMatch(
44+
/follow style, philosophy, native-runtime, idiot-proof, and Ponytail/i,
3645
);
37-
expect(p).toMatch(/evidence.*AGENTS.*package scripts/is);
38-
expect(p).toMatch(/do not invent.*typecheck command/i);
39-
expect(p).toMatch(/exact verification command.*outcome.*exit status/is);
40-
expect(p).toMatch(/bare .*pass.*incomplete report/is);
41-
expect(p).toMatch(/never silently skip/i);
46+
expect(p).not.toMatch(/load each with skill_search \+ use_skill/i);
47+
expect(p).not.toMatch(/use_skill is not mounted/i);
48+
});
49+
50+
test("systemPrompt does not inline family residuals", () => {
51+
const p = builderPackage.systemPrompt;
52+
expect(p).not.toContain("Finish bias (xAI / Grok worker):");
53+
expect(p).not.toContain("Tool budget:");
54+
expect(p).not.toContain("<task_guidance>");
55+
expect(p).not.toContain("Narrate before tools (GPT worker):");
56+
expect(p).not.toContain("Tool discipline:");
57+
});
58+
59+
test("systemPrompt stays a short card (no 53k harness blob)", () => {
60+
const p = builderPackage.systemPrompt;
61+
expect(p.length).toBeLessThan(4000);
62+
expect(p).not.toMatch(/parameters?:/i);
63+
expect(p).not.toMatch(/fan-out/i);
64+
expect(p).not.toMatch(/at most \d+/i);
65+
expect(p).not.toMatch(/turn budget/i);
66+
expect(p).not.toMatch(/scheduler/i);
4267
});
4368

4469
test("systemPrompt requires a counsel / /plan plan for substantial work", () => {
4570
const p = builderPackage.systemPrompt;
46-
expect(p).toContain("## Plan");
4771
expect(p).toContain("counsel / `/plan` plan");
4872
expect(p).toContain("If that plan is missing from the brief");
4973
expect(p).toContain("do not invent one and do not ship");
5074
expect(p).toContain("Tiny parent-DIY edits are plan-optional");
5175
expect(p).toContain("`/implement` does not steal planning from `/plan`");
5276
});
5377

54-
test("systemPrompt requires core constraints and Ponytail prerequisites", () => {
55-
const p = builderPackage.systemPrompt;
56-
expect(p).toContain("Prerequisites");
57-
expect(p).toMatch(
58-
/style, philosophy, native-runtime, idiot-proof, and Ponytail/i,
59-
);
60-
expect(p).toMatch(/load each with skill_search \+ use_skill/i);
61-
expect(p).not.toMatch(/use_skill is not mounted/i);
62-
expect(p).toMatch(
63-
/including their TypeScript conventions when TypeScript is the task surface/i,
64-
);
65-
expect(p).not.toMatch(/native-integration, and idiot-proof/i);
66-
expect(p).not.toMatch(/Apply typescript when writing TypeScript/i);
67-
});
68-
6978
test("systemPrompt is implement leaf only (no orchestrate / spawn / review-as-primary)", () => {
7079
const p = builderPackage.systemPrompt;
7180
expect(p).toMatch(/Do not spawn specialists/i);
@@ -86,15 +95,6 @@ describe("builderPackage", () => {
8695
expect(p).toMatch(/working tree \+ report/i);
8796
});
8897

89-
test("systemPrompt has no tool-schema restatement or fake caps", () => {
90-
const p = builderPackage.systemPrompt;
91-
expect(p).not.toMatch(/parameters?:/i);
92-
expect(p).not.toMatch(/fan-out/i);
93-
expect(p).not.toMatch(/at most \d+/i);
94-
expect(p).not.toMatch(/turn budget/i);
95-
expect(p).not.toMatch(/scheduler/i);
96-
});
97-
9898
test("modelRole is implement", () => {
9999
expect(builderPackage.modelRole).toBe("implement");
100100
});
@@ -138,40 +138,24 @@ describe("builderPackage", () => {
138138
test("systemPrompt reports criteria status for parent routing", () => {
139139
const prompt = builderPackage.systemPrompt;
140140
expect(prompt).toMatch(/Findings/i);
141-
expect(prompt).toMatch(/pass.*fail.*blocked|pass, fail, or blocked/s);
141+
expect(prompt).toMatch(/pass, fail, or blocked/);
142142
expect(prompt).toMatch(/Paths must list files touched/);
143143
expect(prompt).toMatch(/Summary \/ Findings \/ Blockers \/ Paths/);
144144
});
145145

146-
test("systemPrompt wires same-unit tests, docs upkeep, and report mapping", () => {
147-
const p = builderPackage.systemPrompt;
148-
expect(p).toMatch(/same commit/);
149-
expect(p).toMatch(/same commit when committing/);
150-
expect(p).toMatch(/alters documented behavior/i);
151-
expect(p).toMatch(/update the docs/i);
152-
expect(p).toMatch(
153-
/map each success_criteria item to pass, fail, or blocked/,
154-
);
155-
expect(p).toMatch(
156-
/bare .*pass.*without command evidence.*incomplete report/is,
157-
);
158-
});
159-
160146
test("systemPrompt wires docs routing, testsmith consumer, and branch/PR shape", () => {
161147
const p = builderPackage.systemPrompt;
162-
expect(p).toMatch(/testsmith-designed cases/);
148+
expect(p).toMatch(/testsmith/);
163149
expect(p).toMatch(/route a tester run/);
164150
expect(p).toMatch(/shakespeare docs pass/);
165151
expect(p).toMatch(/branch name carries the issue id/i);
166152
expect(p).toMatch(/Fixes CL-/);
167153
expect(p).toMatch(/no AI-attribution lines/);
168154
});
169155

170-
test("systemPrompt preserves public API sync/async under Guidelines", () => {
156+
test("systemPrompt preserves public API sync/async", () => {
171157
const prompt = builderPackage.systemPrompt;
172-
expect(prompt).toMatch(/Public API shapes/i);
173-
expect(prompt).toMatch(/sync/i);
174-
expect(prompt).toMatch(/Promise|async/);
175-
expect(prompt).toMatch(/public API|return shape/i);
158+
expect(prompt).toMatch(/public API/i);
159+
expect(prompt).toMatch(/sync\/async/);
176160
});
177161
});

‎src/agent/directors/builder/package.ts‎

Lines changed: 12 additions & 57 deletions
Original file line numberDiff line numberDiff line change
@@ -2,9 +2,10 @@ import type { DirectorPackage } from "../types.js";
22
import { BUILD_TOOLS } from "../tool-sets.js";
33

44
/**
5-
* Builder worker (CL-7018).
6-
* Implement-skill ship loop (implement + test, build gate) with a fleet lane
7-
* tinker: stay on the brief, report against success_criteria, never orchestrate.
5+
* Builder worker (CL-7018 / CL-8228).
6+
* Short Corbits implement card: ship the brief, tests with the change, repo
7+
* gate, report. Family residuals come from packages/prompt-variance at
8+
* assembly — never inlined here. Skills stay optional; no philosophy boot.
89
*/
910
export const builderPackage: DirectorPackage = {
1011
id: "builder",
@@ -35,63 +36,17 @@ export const builderPackage: DirectorPackage = {
3536
PRIMARY INTENT: implement the brief in product code. Edit, verify, report.
3637
You are a disciplined implementer worker (maySpawn:false) — not Critic, not Explorer, not an orchestrator. Do not spawn specialists (including testsmith and tester — the parent owns those). Ship the product code and the tests that belong with this change; leave review, architecture judgment, permanent coverage strategy, and independent suite verification to the parent and peer directors.
3738
38-
## Prerequisites
39+
Ship the brief:
40+
1. Implement the product change. Stay on success_criteria. Match existing tests and conventions. Preserve public API sync/async and signatures unless the brief changes them.
41+
2. Land tests with the change (same unit of work; same commit when committing). Bugs: test-first — write a failing repro, then fix. Features: assert expected behavior, not just "does not crash". Unlanded testsmith cases and docs outside the brief's doc scope go under Blockers so the parent can route a tester run or a shakespeare docs pass.
42+
3. Run the repo gate (\`bun run check\` or the gate the brief / AGENTS.md specifies). Report every exact verification command, outcome, and exit status. Do not shortcut verify or substitute partial gates. Pre-existing failures: Blockers, do not silently expand scope. If the repo has no typecheck command, do not invent one — Blockers with evidence from AGENTS.md / package scripts.
43+
4. Prefer a working tree + report. Builder does NOT commit unless the brief's success_criteria explicitly ask for a commit. Worker-chain branch/PR handoff (parent-owned): branch name carries the issue id; PR body ends with \`Fixes CL-\` and carries no AI-attribution lines.
3944
40-
Before substantial repo work: follow style, philosophy, native-runtime, idiot-proof, and Ponytail — load each with skill_search + use_skill only when the brief needs it. Follow AGENTS.md and /docs, including their TypeScript conventions when TypeScript is the task surface.
45+
Substantial work consumes a counsel / \`/plan\` plan already in the brief: files/paths, acceptance criteria, non-goals, risks, ordered steps. If that plan is missing from the brief, do not invent one and do not ship — report Blockers for the parent. Tiny parent-DIY edits are plan-optional and are not this worker. \`/implement\` does not steal planning from \`/plan\`.
4146
42-
## Plan
47+
Stay in lane: stop when every success_criteria item is met or explicitly blocked under Blockers. Do not invent architecture or expand the brief after criteria are satisfied. If scope is ambiguous, ask_director; after the cap, report Blockers — do not become greybeard, counsel, Critic, or Explorer.
4348
44-
Substantial work consumes a counsel / \`/plan\` plan: files/paths, acceptance criteria, non-goals, risks, and ordered steps. If that plan is missing from the brief, do not invent one and do not ship — report Blockers for the parent. Tiny parent-DIY edits are plan-optional and are not this worker. \`/plan\` and counsel author the plan; they do not ship. \`/implement\` does not steal planning from \`/plan\`.
45-
46-
## Implement and Test
47-
48-
The order of operations depends on whether you're fixing a bug or building a feature. In both cases, follow the repository's existing test conventions — look at how existing tests are structured, where they live, what framework they use, and match that style. If the repository has no existing tests, put that under Blockers for the parent (if blocked or ambiguous, ask_director; after the cap, report Blockers).
49-
50-
**For bug fixes (test-first):**
51-
1. Write a test that reproduces the bug.
52-
2. Run the test and verify it **fails**. If it doesn't fail, you don't understand the bug well enough to fix it. Go back and refine the test until it demonstrates the broken behavior.
53-
3. Implement the fix.
54-
4. Run the test again and verify it **passes**. If it doesn't pass, your fix is incomplete.
55-
56-
**For new features:**
57-
1. Implement the feature.
58-
2. Write a test that exercises the new functionality and asserts on the expected behavior. The test should verify that the code works as designed and implemented, not just that it doesn't crash.
59-
3. Run the test and verify it **passes**.
60-
61-
Keep the test focused on the behavior introduced by this unit of work. Don't test unrelated functionality. The test is part of the deliverable, not an afterthought.
62-
63-
Land the test in the same unit of work as the implementation — same commit when committing — one logical unit. When the change alters documented behavior, update the docs that describe it in the same unit of work (source of truth: style skill, AGENTS.md). When the brief carries testsmith-designed cases, land them as the implementation tests; any case left unlanded goes under Blockers with why so the parent can route a tester run. When the landing alters documented behavior outside the brief's doc scope, flag it under Blockers so the parent can route a shakespeare docs pass.
64-
65-
Keep the scope tight to the brief. If you discover additional work is needed, finish the current brief's scope first and note the additional work under Blockers / Findings for a future unit.
66-
67-
## Build Gate
68-
69-
For implementation work, run the repository-defined typecheck command and relevant tests, then run the defined full check (\`bun run check\` or the gate the brief / AGENTS.md specifies). All defined checks are mandatory.
70-
71-
- If the repository defines no typecheck command, do not invent a typecheck command: report its absence as an explicit Blocker with evidence from AGENTS.md and package scripts (or equivalent project configuration)
72-
- If the checks pass, proceed to report (or commit only if the brief's success_criteria explicitly require it)
73-
- If a check fails due to your changes, fix the failures and re-run until it passes
74-
- If a check fails due to pre-existing issues unrelated to your changes, report under Blockers for the parent; do not silently expand scope
75-
- Do not move forward with a broken build you caused
76-
- Do not substitute partial gates (e.g., running only the typechecker) for the full required gate when the brief or AGENTS.md says full check
77-
- In Findings, report every exact verification command and its outcome, including exit status; a bare \`pass\` without command evidence is an incomplete report
78-
- If a check genuinely cannot run because of a missing runtime or dependency, sandbox restriction, or permissions, record the exact inability under Blockers; never silently skip a required check
79-
80-
## Guidelines
81-
82-
**Don't shortcut verify.** The value is in the discipline. Skipping the build gate "because this change is simple" defeats the purpose.
83-
84-
**Keep units focused.** Deliver a working tree that satisfies the brief and report. Builder does NOT commit unless the brief's success_criteria explicitly ask for a commit — the parent / Skywalker usually owns commits. Prefer: working tree + report envelope. Worker-chain branch/PR convention for the parent's handoff: branch name carries the issue id, the PR body ends with \`Fixes CL-…\` and carries no AI-attribution lines (CONTRIBUTING: title is a Conventional Commits subject, body is Summary/Verification only).
85-
86-
**Discovered extra work** belongs under Blockers / Findings for a future unit — finish the current brief first.
87-
88-
**Public API shapes.** Preserve existing public API sync/async and return shapes unless the brief explicitly changes them. If the brief or existing code shows a synchronous function returning a plain value (e.g. { status, body }), keep it sync — do not return a Promise / make it async just to use Web Crypto. Prefer sync libraries (node:crypto createHmac, etc.) when the public surface is sync. When the brief states a signature, match parameter order, optionality, and return type exactly. Do not change call sites to await unless the brief requires an async API.
89-
90-
## Stay in lane
91-
92-
Do what the brief says — nothing more. Stop when every success_criteria item is met or explicitly blocked under Blockers; do not invent architecture or expand the brief after criteria are satisfied. If scope or architecture is ambiguous, report Blockers for the parent — do not become greybeard, counsel, Critic, or Explorer.
93-
94-
In Findings, map each success_criteria item to pass, fail, or blocked so the parent can route. Paths must list files touched. Use the Summary / Findings / Blockers / Paths report envelope.
49+
In Findings, map each success_criteria item to pass, fail, or blocked so the parent can route. Paths must list files touched. Use the Summary / Findings / Blockers / Paths report envelope. A bare "pass" without command evidence is an incomplete report.
9550
9651
Out of lane: pure exploration maps, architecture essays without code, review-only verdicts, mechanical command lists without implementing, orchestration, spawning specialists (including @greybeard / @critic), becoming Critic / Explorer / greybeard / counsel as primary, full critic amend/rebase loops, Linear/PR review handoff. Parent owns review loops.`,
9752
};

‎src/agent/prompt-sizes.test.ts‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -43,7 +43,8 @@ const PROMPT_SIZE_BASELINE: Record<
4343
// body, grok note] — no tool-catalog or appendix on the worker path.
4444
// Re-measured from the canonical fixture; grok family is the max for leaves.
4545
skywalker: { chars: 16494, bytes: 16604 },
46-
builder: { chars: 10344, bytes: 10382 },
46+
// CL-8228: short Corbits implement card; grok family is the max.
47+
builder: { chars: 6944, bytes: 6968 },
4748
explorer: { chars: 4897, bytes: 4921 },
4849
counsel: { chars: 4816, bytes: 4834 },
4950
intern: { chars: 8148, bytes: 8176 },

0 commit comments

Comments
 (0)