From 538ccff159efa42866a1a98223ec7d1e63578b75 Mon Sep 17 00:00:00 2001 From: Sawyer Date: Mon, 21 Sep 2026 21:50:22 -0700 Subject: [PATCH] feat(skywalker): shrink primary prompt to a dispatcher card Idle, mailbox, and poll belong in the harness. Family residuals come from prompt-variance instead of inlined novels. --- docs/ARCHITECTURE.md | 10 +- docs/IMPLEMENTATION.md | 2 +- src/agent/directors/skywalker/package.test.ts | 201 +++++------------- src/agent/directors/skywalker/package.ts | 161 +++----------- src/agent/prompt-sizes.test.ts | 3 +- src/prompts.test.ts | 4 - 6 files changed, 95 insertions(+), 286 deletions(-) diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index cf00a5fcd..89bb3cf24 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -280,9 +280,9 @@ Every shipped specialist is a **director package** — a prompt-first `DirectorP **Primary** -| Director | Owns | Does not own | -| --------- | -------------------------------------------------------------- | ------------------------------------------------------------- | -| skywalker | Orchestrate only — classify, dispatch, track fleet, synthesize | Product tree edits; being the implementer/reviewer by default | +| Director | Owns | Does not own | +| --------- | -------------------------------------------------------------------------- | ------------------------------------------------------------------------------------ | +| skywalker | Dispatcher — classify, DIY tiny edits, spawn named specialists, synthesize | Substantial product work without spawning; being the implementer/reviewer by default | **Engineering directors** @@ -351,9 +351,9 @@ Data-only agent plugins (`src/plugins/data-only-agent.ts`) synthesize `agentPlug ### System Prompt (`src/agent/prompts.ts`) -The primary session identity is **Skywalker** (`buildChatRole` → `createSkywalkerSystemPrompt`). Product name remains Corbits Code; when asked its name, the primary answers Skywalker. Role: orchestrate — classify, DIY tiny/single-file/one-route product edits, dispatch closed directors via `spawn_agent` (mailbox mail on TUI; `wait_agents` on exec primary) for substantial work, track the fleet, synthesize. Product mutation tools (`write_file` / `edit_file` / `delete_file`) are mounted on the primary session (CORE and `SKYWALKER_TOOLS`) so Skywalker can DIY bounded edits; spawn remains the default for substantial, multi-file, parallel, or specialist work. Shell file-writes stay denied by auto-shell policy. MCP tools are not re-filtered by a product-write deny list (that list is gone). There is no static per-worker write-path lock; concurrent lanes sharing a cwd are instead flagged (not blocked) as a `conflict` intervention. A frontier model already knows how to code; the static prompt carries harness-specific facts and the closed-fleet orchestration policy. The base is three individually-exported sections: +The primary session identity is **Skywalker** (`buildChatRole` → `createSkywalkerSystemPrompt`). Product name remains Corbits Code; when asked its name, the primary answers Skywalker. Role: dispatcher — classify, DIY tiny/single-file/one-route product edits (Builder neighborhood), spawn named closed directors via `spawn_agent` for substantial work, track the fleet, synthesize. The Skywalker card is a ~3–5k operator-surface prompt; idle, mailbox delivery, and poll-avoidance live in the harness (tool descriptions and occupancy), not the card. Family residuals append via `@corbits/prompt-variance` at assembly (model-family policy) — do not inline family text in the card. Product mutation tools (`write_file` / `edit_file` / `delete_file`) are mounted on the primary session (CORE and `SKYWALKER_TOOLS`) so Skywalker can DIY bounded edits; spawn remains the default for substantial, multi-file, parallel, or specialist work. Shell file-writes stay denied by auto-shell policy. MCP tools are not re-filtered by a product-write deny list (that list is gone). There is no static per-worker write-path lock; concurrent lanes sharing a cwd are instead flagged (not blocked) as a `conflict` intervention. A frontier model already knows how to code; the static prompt carries harness-specific facts and the closed-fleet routing policy. The base is three individually-exported sections: -- `buildChatRole` — Skywalker primary identity (orchestrate; DIY tiny/bounded product edits; spawn for substantial work). +- `buildChatRole` — Skywalker dispatcher card (operator surface; classify; DIY tiny/bounded product edits; spawn named specialists). - `buildHarnessFacts` — the non-derivable rules: shell file-writes are blocked, path tools are the DIY surface on primary (spawn builder/docs directors for substantial work), dependency installs and off-limits paths need approval, images are native multimodal input, core tools plus the advertised catalog (including `skill_search`) are resident (MCP and other unadvertised tools load via `tool_search`; use `search_agents` before dispatching specialists), workflows run only from slash-command steps, and session memory lives at `.corbits/MEMORY.md`. - `buildGuidelines` — be concise, prefer `spawn_agent` (mailbox mail on TUI; exec-primary `wait_agents`) for substantial product work with one focused task per worker and fan-out by independent lanes, DIY tiny/bounded edits on the parent, answer questions and diagnose visual/product feedback before editing, work autonomously for explicit coding tasks, use `lsp` for symbol work, and require implementation agents to run the repository-defined typecheck, relevant tests, and every defined full verification command. Agents report exact commands, outcomes, and exit statuses; a repository with no typecheck command produces an explicit Blocker backed by project-configuration evidence rather than an invented command or silent skip. - `buildPromptDisciplineBlock` — a shared, prohibition-form section appended exactly once to every built prompt (chat and sub-agent, every provider family). Primary vs worker wording differs for product writes: workers are told to use `read_file`/`edit_file`/`write_file`; Skywalker is told to DIY tiny/bounded edits with those path tools and spawn directors for substantial work. Shared rules: never `cat`/`sed`/heredoc/`echo` for file work, no setting or exporting environment variables (recurring needs belong in project settings), `web_fetch`/`web_search` instead of `curl`/`wget`/hand-rolled queries, one operation per `run_shell` call, turn semantics (a tool-less reply is the final answer, no repeat searches, stop and change approach after three failed attempts, batch independent reads in parallel), and TTY output rules (short bold headers, one-line bullets, backticks for paths/commands, no wide tables). diff --git a/docs/IMPLEMENTATION.md b/docs/IMPLEMENTATION.md index 2edcdfbc7..82d636686 100644 --- a/docs/IMPLEMENTATION.md +++ b/docs/IMPLEMENTATION.md @@ -161,7 +161,7 @@ Twenty packages under `src/agent/directors//` register in `DIRECTOR_REGISTRY 2. `packageToProfile` maps envelope (`tools.allow`/`deny`) to `AgentProfile.capabilities` and `spawn.maySpawn` → `orchestrator`. System prompts are prefixed with a stable identity block (`formatDirectorSystemPrompt`: agent id, model role, optional skills). 3. Nested spawn: packages with `spawn.allowlist` forward that list into nested `spawn_agent` (`spawnAllowlist` on nestedDispatch). Off-list `agent` is refused. `spawn_agent(agent=skywalker)` is refused (primary is not a spawned worker). Primary omits the list so plugin profiles stay reachable. 4. `directorProfiles()` is the spawn catalog (`default-agents.ts`) — closed set minus skywalker. Plugin and local `.agents/agents/` profiles still load, but closed `DIRECTOR_IDS` cannot be overridden or aliased. -5. Primary chat role is Skywalker: `buildChatRole()` → `createSkywalkerSystemPrompt()`. Product mutation tools (`write_file` / `edit_file` / `delete_file`) live in CORE (and `SKYWALKER_TOOLS`) so they are advertised on the primary without a `tool_search` round-trip. DIY tiny/bounded edits on the parent; spawn builder/docs directors for substantial work — a prompt judgment call, not a toolset strip. `PRIMARY_DENIED_PRODUCT_TOOLS` is gone. Shell file-writes stay denied; MCP tools are not re-filtered by a product-write deny list. There is no static per-profile write-path lock (CL-6952). +5. Primary chat role is Skywalker: `buildChatRole()` → `createSkywalkerSystemPrompt()`. The Skywalker card is a ~3–5k dispatcher (classify, tiny DIY, spawn named specialists); idle/mailbox/poll live in the harness, and family residuals come from `@corbits/prompt-variance`. Product mutation tools (`write_file` / `edit_file` / `delete_file`) live in CORE (and `SKYWALKER_TOOLS`) so they are advertised on the primary without a `tool_search` round-trip. DIY tiny/bounded edits on the parent; spawn builder/docs directors for substantial work — a prompt judgment call, not a toolset strip. `PRIMARY_DENIED_PRODUCT_TOOLS` is gone. Shell file-writes stay denied; MCP tools are not re-filtered by a product-write deny list. There is no static per-profile write-path lock (CL-6952). **Grok infer envelope.** `loadSessionChatPrompt` always appends `loadAgentContextExtensions` (`AGENTS.md`, capped at `MAX_AGENTS_MD_BYTES`) and advertises CORE+CATALOG full schemas via `advertisedToolNamesForSessionMode`. That assembly is family-agnostic — Grok does not substitute the trimmed director prompt (`buildSubAgentSystemPrompt` + `formatDirectorSystemPrompt`). Workers already use that trimmed path (no `AGENTS.md`, mounted-tool schemas only). Keep the infer envelope on Grok; do not strip `AGENTS.md` or core schemas. Measure in `src/agent/prompt-sizes.ts` (`assembleSkywalkerInferEnvelope` vs `assembleDirectorPrompt("skywalker", "grok")`). diff --git a/src/agent/directors/skywalker/package.test.ts b/src/agent/directors/skywalker/package.test.ts index 1312334f7..8c8aa8487 100644 --- a/src/agent/directors/skywalker/package.test.ts +++ b/src/agent/directors/skywalker/package.test.ts @@ -19,6 +19,27 @@ describe("skywalkerPackage", () => { expect(createSkywalkerSystemPrompt()).toBe(skywalkerPackage.systemPrompt); }); + test("dispatcher card stays in the 3–5k band", () => { + const p = skywalkerPackage.systemPrompt; + expect(p.length).toBeGreaterThanOrEqual(3000); + expect(p.length).toBeLessThanOrEqual(5000); + }); + + test("idle, mailbox, and poll stay out of the card", () => { + const p = skywalkerPackage.systemPrompt; + expect(p).not.toMatch(/idle/i); + expect(p).not.toMatch(/mailbox/i); + expect(p).not.toMatch(/\bpoll\b/i); + expect(p).not.toContain("wait_agents"); + }); + + test("card is not a Karen clone", () => { + const p = skywalkerPackage.systemPrompt; + expect(p).not.toMatch(/karen/i); + expect(p).not.toContain("section 9"); + expect(p).not.toContain("You orchestrate."); + }); + test("spawn allowlist is the full closed set", () => { expect(skywalkerPackage.spawn.allowlist).toHaveLength(19); expect(skywalkerPackage.spawn.allowlist).toEqual([ @@ -85,65 +106,48 @@ describe("skywalkerPackage", () => { ); }); - test("systemPrompt parent tools tell the parent not to run long-blocking jobs", () => { + test("systemPrompt classifies, DIY in Builder neighborhood, and routes named specialists", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Parent tools"); - expect(p).toContain("long-blocking"); - expect(p).toContain("tool.boundary"); - expect(p).toContain("Dispatch intern"); - expect(p).toContain("or builder (substantial code)"); + expect(p).toContain("# Classify"); + expect(p).toContain("COMMUNICATION"); + expect(p).toContain("IMPLEMENTATION"); + expect(p).toContain("ORCHESTRATION"); + expect(p).toContain("Builder neighborhood"); + expect(p).toContain("operator surface"); + expect(p).toContain("# Routing"); + expect(p).toContain("builder = ship product code + tests"); + expect(p).toContain("No catch-all worker"); + expect(p).not.toContain("one-agent-per-task"); + expect(p).not.toContain("one agent per task"); }); - test("systemPrompt has effort scaling / named non-overlapping lanes (no numeric soft ceiling)", () => { + test("systemPrompt gives each worker one focused task", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Effort scaling"); - expect(p).toContain("fan-out"); - expect(p).toContain("0–1 worker"); - expect(p).toContain("named, non-overlapping lanes"); expect(p).toContain("one focused task"); expect(p).toContain("one lane per PR/path/ownership"); expect(p).toContain("Do not pack a multi-step workflow into one worker"); + expect(p).toContain("keep it tight"); + expect(p).toContain("One job per spawn"); expect(p).not.toContain("2–4 workers"); expect(p).not.toContain("at most 4"); expect(p).not.toContain("Prefer synthesizing early returns"); - // CL-6953: admission queue owns queueing (src/subagent/admission.ts) — - // the prompt keeps the fan-out judgment, not the mechanism restatement. expect(p).not.toContain("queues excess"); - expect(p).toContain("Do not invent a numeric cap"); }); - test("systemPrompt prefers spawn_agent then idle (idle-orchestrator)", () => { + test("systemPrompt parent tools tell the parent not to run long-blocking jobs", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("spawn_agent"); - // CL-6953: collection path (wait_agents vs mailbox) lives in the runtime - // mount + spawn_agent tool description — not the static prompt. - expect(p).not.toContain("wait_agents"); - expect(p).toContain("Idle-orchestrator"); - expect(p).not.toContain("task()"); - expect(p).toContain("Spawn then idle; do not poll"); - expect(p).not.toContain("do not poll wait_agents"); - // CL-6953: wait_agents mounting lives in the runtime (exec/runner.ts) and - // tool descriptions — the prompt keeps spawn-then-idle, not the mount fact. - expect(p).not.toContain("wait_agents is mounted on exec-primary runs only"); - expect(p).toContain("mailbox mail arrives as inbound"); - expect(p).toContain( - "When the fleet goes dry the runtime re-enters with collected reports", - ); - expect(p).not.toContain("do not tight-loop wait_agents"); - expect(p).not.toContain( - "Present the plan when the change is large or ambiguous", - ); + expect(p).toContain("long-blocking"); + expect(p).toContain("dispatch intern"); + expect(p).toContain("tester"); + expect(p).toContain("builder"); }); - test("systemPrompt requires frequent operator updates and staying free for Enter", () => { + test("systemPrompt requires frequent operator updates", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Operator updates"); + expect(p).toContain("# Operator surface"); expect(p).toContain("only surface that talks to the operator"); expect(p).toContain("frequent short status updates"); - expect(p).toContain("reply to the operator"); - expect(p).toContain("end the turn"); - expect(p).toContain("mailbox mail"); - expect(p).toContain("Enter mid-run"); + expect(p).toContain("send_input"); }); test("systemPrompt does not forbid steering workers when the operator messages mid-run", () => { @@ -152,64 +156,14 @@ describe("skywalkerPackage", () => { expect(p).not.toContain("Do not hold the reply on fleet collection"); }); - test("systemPrompt anti-cascade keeps digs out of fleets", () => { - const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Anti-cascade"); - expect(p).toContain("COMMUNICATION first"); - expect(p).toContain("Never spawn parallel"); - expect(p).toContain("one explorer worker"); - expect(p).toContain("search the repo yourself after a worker stops"); - expect(p).toContain("Do not reclassify COMMUNICATION as ORCHESTRATION"); - expect(p).toContain("synthesize what returned"); - expect(p).toContain("do **not** re-fan-out another diagnostic wave"); - expect(p).toContain( - "Permission asks and long run_shell clocks on worker rows are not a signal to spawn more diggers", - ); - expect(p).toContain( - "`incomplete-report` from plan/counsel is not an attachable plan", - ); - expect(p).not.toContain("Then start the next worker"); - expect(p).not.toContain("if the job still needs doing"); - }); - - test("systemPrompt fail-then-successor is distinct from operator-cancel wait", () => { - const p = skywalkerPackage.systemPrompt; - expect(p).toContain("incomplete-report"); - expect(p).toContain("MAY `spawn_agent` **one** successor"); - expect(p).toContain("Cap is one successor for that stall"); - expect(p).toContain("changed** brief"); - expect(p).toContain("wait for the operator"); - expect(p).toContain("Do not auto-retry"); - expect(p).toContain( - "Identical re-dispatch of the same brief stays refused", - ); - expect(p).toContain("Operator-cancel is not a re-dispatch"); - expect(p).not.toContain("Then start the next worker"); - expect(p).not.toContain("if the job still needs doing"); - expect(p).not.toContain("interrupted-incomplete"); - }); - - test("systemPrompt treats parent interrupt as resume, not successor spawn", () => { - const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Parent-initiated interrupt"); - expect(p).toContain("resume_agent"); - expect(p).toContain("still-live worker"); - expect(p).toContain("no longer resumable"); - expect(p).toContain("interrupt_agent"); - expect(p).toContain("stop_reason: interrupted"); - }); - test("systemPrompt simple path skips explorer+critic for tiny work", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("DIY on the parent"); - expect(p).toContain("skip spawn, skip explorer, skip plan, skip critic"); + expect(p).toContain("Skip spawn, skip explorer, skip plan, skip critic"); expect(p).toContain("write_file/edit_file"); - expect(p).toContain("Do not always explorer→plan→implement→critic"); }); test("systemPrompt routes URL reads through web_fetch on primary", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Fetch URLs"); expect(p).toContain("web_fetch"); expect(p).toContain("already mounted"); expect(p).toContain("curl/wget"); @@ -217,7 +171,6 @@ describe("skywalkerPackage", () => { test("systemPrompt teaches spawn handoff packet for dispatch", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Spawn handoff"); expect(p).toContain("success_criteria"); expect(p).toContain("do_not"); expect(p).toContain("Child starts blank"); @@ -226,14 +179,9 @@ describe("skywalkerPackage", () => { expect(p).toContain("required for implement/review"); expect(p).toContain("keep it tight"); expect(p).toContain("One job per spawn"); - // CL-6953 / CL-6807: single contract statement (spawn graph); the routing - // and handoff restatements are gone. expect(p).not.toContain("Runtime requires success_criteria"); expect(p).not.toContain("Brief completeness"); expect(p).not.toContain("Prefer typed spawn"); - expect(p.indexOf("Critic stays clean-room")).toBeGreaterThan( - p.indexOf("# Verify after ship"), - ); }); test("systemPrompt report envelope names each section header explicitly", () => { @@ -254,13 +202,7 @@ describe("skywalkerPackage", () => { const p = skywalkerPackage.systemPrompt; expect(p).toContain("ask_director"); expect(p).toContain("send_input"); - expect(p).toContain("Do not poll list_agents"); - expect(p).not.toContain("list_agents shows awaiting_director"); - expect(p).toContain("idle-send"); expect(p).toMatch(/target = (that worker's |worker )session id/); - expect(p).not.toMatch( - /wait_agents returns status running plus a question/i, - ); expect(p).toMatch( /Escalate with ask_operator only when you cannot resolve it/, ); @@ -276,44 +218,19 @@ describe("skywalkerPackage", () => { test("systemPrompt requires critic after every builder implementation", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain("Verify after ship"); expect(p).toContain("tester"); - expect(p).toContain( - "correctness/brief gaps and hygiene the diff introduced", - ); - expect(p).toContain("That hygiene lens is not over-engineering theater"); - expect(p).toMatch( - /after every delegated \*\*builder\*\* implementation.*run \*\*critic\*\*/is, - ); - expect(p).toMatch( - /substantial implementation limited to one internal file.*still requires Critic/is, - ); + expect(p).toMatch(/after every delegated builder landing.*run critic/is); expect(p).toMatch(/Builder self-report.*never sufficient to skip/is); - expect(p).toMatch( - /After every delegated builder landing.*run a critic.*architecture.*add greybeard/is, - ); - expect(p).not.toMatch( - /critic \(or greybeard when architecture is in play\)/i, - ); - expect(p).toMatch( - /Skip a new Critic dispatch only for parent-DIY work or when existing independent review evidence already covers both the resulting diff and its success criteria/i, - ); - expect(p).not.toMatch(/Multi-file or public-API changes: after builder/i); - expect(p).not.toMatch(/After multi-file builder landings/i); - expect(p).not.toMatch(/self-report is thin/i); + expect(p).toContain("add greybeard when architecture is in play"); + expect(p).toMatch(/Skip a new critic only for parent-DIY/i); }); test("systemPrompt spawn-target for substantial code is builder, not implement", () => { const p = skywalkerPackage.systemPrompt; expect(p).toContain("spawn builder"); - expect(p).toContain("spawn (builder for code"); expect(p).toContain("builder = ship product code + tests"); expect(p).not.toContain("implement = ship product code + tests"); expect(p).not.toMatch(/\bspawn implement\b/); - expect(p).toContain("explorer → plan → implement → critic"); - expect(p).toContain("Do not always explorer→plan→implement→critic"); - expect(p).toContain("Substantial builder work consumes a counsel"); - expect(p).toContain("builder blocks if the plan is still missing"); expect(p).toContain("Tiny parent-DIY edits stay plan-optional"); expect(p).toContain("`/implement` does not steal planning from `/plan`"); }); @@ -321,12 +238,8 @@ describe("skywalkerPackage", () => { test("systemPrompt re-dispatches builder on blocking critic", () => { const p = skywalkerPackage.systemPrompt; expect(p).toContain("blocking"); - expect(p).toContain("re-dispatch **builder**"); - expect(p).toMatch(/narrowed or changed follow-up brief/i); - expect(p).toContain("ship → verify → fix → re-verify"); - // CL-6953: no runtime retry budget enforces a re-fix cap — keep the loop - // judgment, not the number. - expect(p).not.toContain("Cap re-fix rounds"); + expect(p).toContain("re-dispatch builder"); + expect(p).toMatch(/narrowed brief/i); }); test("systemPrompt Linear three-state: In Review at PR-open, never Done at PR-open", () => { @@ -340,20 +253,14 @@ describe("skywalkerPackage", () => { expect(p).not.toContain("gh pr review"); }); - test("systemPrompt routes mutation-checks to gauntlet (tiny verify-after-ship mention)", () => { + test("systemPrompt routes mutation-checks to gauntlet", () => { const p = skywalkerPackage.systemPrompt; - expect(p).toContain( - "gauntlet = mutation-check that tests can actually fail (tree clean)", - ); + expect(p).toContain("gauntlet = mutation-check"); }); - test("systemPrompt routes trust-path diffs to warden (tiny verify-after-ship mention)", () => { + test("systemPrompt routes trust-path diffs to warden", () => { const p = skywalkerPackage.systemPrompt; expect(p).toContain("warden = permission / provider-auth / plugin-loader"); - expect(p).toContain("only when the diff touches those paths"); - expect(p).toContain( - "When the diff touches permission, provider-auth, or plugin-loader paths, add a warden trust review alongside critic", - ); - expect(p).toContain("never ships fixes"); + expect(p).toContain("add warden when the diff touches permission"); }); }); diff --git a/src/agent/directors/skywalker/package.ts b/src/agent/directors/skywalker/package.ts index aced09925..bb6aab34b 100644 --- a/src/agent/directors/skywalker/package.ts +++ b/src/agent/directors/skywalker/package.ts @@ -1,150 +1,57 @@ -// Skywalker: primary orchestration director. Chains specialists into a workflow. +// Skywalker: primary dispatcher card. Idle/mailbox/poll live in the harness. import type { DirectorPackage } from "../types.js"; import { SKYWALKER_TOOLS } from "../tool-sets.js"; -const SKYWALKER_SYSTEM_PROMPT = `You are Skywalker — the primary orchestrator for Corbits Code. +const SKYWALKER_DISPATCHER_CARD = `You are Skywalker — the primary dispatcher for Corbits Code. When asked your name, answer: Skywalker. -Agent id: skywalker (primary session; not a spawned worker). Prefer spawn_agent for specialists (parallel OK), then idle. Mailbox mail arrives as inbound when workers finish — spawn then idle; do not poll. +Agent id: skywalker (primary session; not a spawned worker). Prefer spawn_agent for named specialists (parallel OK). -PRIMARY INTENT: run the workflow. DIY tiny/single-file/one-route product edits yourself with write_file/edit_file/delete_file; Delegate substantial work to specialists (spawn, then idle for mailbox mail). Answer questions yourself — COMMUNICATION first, never a fleet. You are the only surface that talks to the operator — give frequent short status updates while work is in flight. Do not become the reviewer or explorer by default. +PRIMARY INTENT: you are the operator surface. Classify every request. DIY tiny/single-file/one-route product edits yourself (Builder neighborhood). Delegate substantial work by spawning a named specialist. Answer COMMUNICATION yourself — never a fleet. Do not become the reviewer or explorer by default. -# Parent tools +# Classify -Do not run long-blocking jobs on the parent (evals, full test suites, long installs, long-running implementation). Dispatch intern (mechanical shell), tester (suite / repro), or builder (substantial code). Path tools (write_file/edit_file/delete_file) are the DIY surface; shell file-writes stay denied. +Every request is COMMUNICATION, IMPLEMENTATION, or ORCHESTRATION. +- COMMUNICATION (why/how/stalled, questions, screenshots): answer yourself; at most one explorer if a single unknown path blocks you. Do not reclassify as ORCHESTRATION to justify a fleet. +- IMPLEMENTATION: tiny/single-file/one-route → DIY; substantial/multi-file/parallel → spawn builder with the counsel / \`/plan\` plan (spawn counsel first if that plan is missing). Tiny parent-DIY edits stay plan-optional. \`/implement\` does not steal planning from \`/plan\`. Docs/design → shakespeare / bruckheimer / rand unless a one-line fix. +- ORCHESTRATION: spawn named, non-overlapping specialists (one lane per PR/path/ownership). Each spawned worker gets one focused task. Do not pack a multi-step workflow into one worker. No catch-all worker. If unsure, reclassify — do not spawn a blob agent. -Idle-orchestrator: fire one or more spawn_agent calls in a turn — each returns immediately with an agent_id and does not hold the parent. Then **reply to the operator** with who is running and **end the turn**. Workers keep running while you are idle; mailbox mail arrives as inbound when a worker finishes or fails — read it and decide the next action. Spawn then idle; do not poll. list_agents shows the fleet without blocking; do not poll list_agents. Enter mid-run delivers at the next parent tool.boundary — a long parent foreground run_shell holds those steers (start long commands with run_shell background:true instead). A bare spawn_agent does not. When the fleet goes dry the runtime re-enters with collected reports. +# Operator surface -# Operator updates (mandatory while fleet is live) +You are the only surface that talks to the operator. Give frequent short status updates while work is in flight. Workers cannot ask_operator; they ask_director — answer with send_input (target = that worker's session id). Escalate with ask_operator only when you cannot resolve it. After every spawn wave: short status (who, goal, what you are waiting on). On a finished report: short update — do not go silent. Operator text while a specialist is running: send_input (soft) to that agent_id, then a short ack. manage_tasks is the checklist; chat is the narrative. -You are the chat surface. Workers cannot ask_operator; they ask_director. A parked question arrives as an idle-send wake — answer with send_input using target = that worker's session id. Do not poll list_agents. Escalate with ask_operator only when you cannot resolve it. While any specialist is running: -- After every spawn wave: short status (who, goal, what you are waiting on) then end the turn. -- On mailbox mail or a finished report: short update — do not go silent. -- Operator text while a specialist is running: send_input (soft) to that agent_id, then a short ack. -- Keep updates short; no wall of task dumps. manage_tasks is the checklist; chat is the narrative. +# Tiny DIY -Example chains: -- tiny fix: DIY write_file/edit_file (do not spawn) -- feature: explorer → plan → implement → critic -- "why / how / is this stalled": answer yourself; at most one explorer if a single unknown blocks you +Tiny/single-file/one-route product edits: write_file/edit_file/delete_file yourself — same neighborhood as Builder tiny work. Skip spawn, skip explorer, skip plan, skip critic. Path tools are the DIY surface; shell file-writes stay denied. Do not run long-blocking jobs on the parent (evals, full suites, long installs, long implementation) — dispatch intern, tester, or builder. URLs: web_fetch is already mounted; do not curl/wget. -Closed directors (use search_agents / registry; each id is a spawn agent= target): builder, explorer, counsel, intern, critic, greybeard, neckbeard, bruckheimer, gaasbot, draper, emil, rand, shakespeare, testsmith, tester, gauntlet, prober, migrator, warden. -No catch-all worker. If unsure, reclassify — do not spawn a blob agent. +# Spawn + +Pass a typed brief and keep it tight: intent, success_criteria, do_not, report_focus, and agent. One job per spawn — do not stuff extra work into the prompt. Each spawned worker gets one focused task. Child starts blank — write a complete packet (Goal, contracts verbatim, Scope/do_not, Done-when, What to report). success_criteria is required for implement/review and their default directors. When the operator brief states a function signature or return shape, put that verbatim into implement success_criteria (including sync vs Promise). After every delegated builder landing, run critic in a fresh context (brief + diff + public API; clean-room, no fork); add greybeard when architecture is in play; add warden when the diff touches permission, provider-auth, or plugin-loader. Builder self-report is never sufficient to skip critic. If critic or tester reports blocking findings, re-dispatch builder with a narrowed brief. Use tester for independent suite evidence. Skip a new critic only for parent-DIY or when existing independent review already covers the diff and criteria. + +# Routing -Quick routing: - explorer = map/read codebase - counsel = ordered eng plan (no ship) - builder = ship product code + tests - critic = defects with evidence including hygiene the diff introduced (no fix) -- warden = permission / provider-auth / plugin-loader trust review (no fix; only when the diff touches those paths) -- greybeard = architecture judgment -- neckbeard = hygiene / pedantry with receipts -- tester = run the suite / repro -- testsmith = design permanent test cases -- gauntlet = mutation-check that tests can actually fail (tree clean) -- prober = measure-only latency/behavior probe per family/model -- migrator = reversible settings/config/session-state migrations -- shakespeare = PRODUCT/ARCHITECTURE/IMPLEMENTATION docs -- rand = DESIGN.md only -- draper = brand/design critique (visual, copy, interactive) -- emil = design-eng laws review +- warden = permission / provider-auth / plugin-loader trust review (no fix) +- greybeard = architecture; neckbeard = hygiene with receipts +- tester = suite / repro; testsmith = permanent cases; gauntlet = mutation-check (tree clean) +- prober = measure-only; migrator = reversible settings/config/session-state +- shakespeare = PRODUCT/ARCHITECTURE/IMPLEMENTATION docs; rand = DESIGN.md +- draper = brand/design; emil = design-eng laws; bruckheimer = product discovery - gaasbot = risk counsel -- bruckheimer = product discovery docs - intern = exact shell / mechanical ops -- After every delegated builder landing → run a critic on the diff/criteria in a fresh context; when architecture is in play, add greybeard for architecture judgment -- When the diff touches permission, provider-auth, or plugin-loader paths, add a warden trust review alongside critic; warden reports trust findings and never ships fixes. - -Pass intent, do_not, report_focus, and agent when specialist. -Parallelize independent lanes with spawn_agent, then idle. manage_tasks for your checklist. ask_operator when blocked or ambiguous — put long rationale in a normal transcript reply first, then call ask_operator with a short question and short option labels only. - -# Fetch URLs (primary-mounted) - -When the operator (or brief) gives an http(s) URL to read: -- Call **web_fetch** yourself on that URL — it is already mounted. Do not tool_search for it, do not shell curl/wget/fetch, do not thrash run_shell to download pages. -- After you have the content, DIY a tiny file write yourself; spawn builder only if the write is substantial. For pure Q&A from a URL, answer directly. -- Cap retries: if web_fetch fails once with a clear error, report the blocker — do not burn a long tool-only streak on shell workarounds. - -# Effort scaling (IMPLEMENTATION / ORCHESTRATION) - -Scale fan-out to the ask: -- Each spawned worker gets **one focused task**. Do not pack a multi-step workflow into one worker. -- Simple (answer, one-path lookup, tiny fix): 0–1 worker, few tools; often answer without fleet -- Tiny single-file / one-route asks: **DIY on the parent** with write_file/edit_file; skip spawn, skip explorer, skip plan, skip critic. Do not always explorer→plan→implement→critic for simple work — that burns wall clock. -- Multi-lane work: spawn only named, non-overlapping lanes (one lane per PR/path/ownership). Width follows independent lanes. Do not invent a numeric cap. - -# Anti-cascade (stall / dig / diagnose) - -Do **not** turn a "why is this stalled / why no thinking / spawn looks broken" dig into a fleet: -- Classify digs, screenshots of worker rows, and "why/how does X work" as COMMUNICATION first. Answer it yourself with parent read/search tools; one explorer worker only if a single unknown path blocks the answer. -- Never spawn parallel "parent UI / child UI / stream events / prompt guardrail / session dig" waves for the same question. -- When workers stall or loop: synthesize what returned, report Blockers, and change approach — do **not** re-fan-out another diagnostic wave on the same topic. -- Failed wait (\`status: failed\` plus \`error\`) or salvage \`incomplete-report\`: diagnose from the wait report or error; MAY \`spawn_agent\` **one** successor with a **changed** brief (new \`success_criteria\` / \`do_not\` / continuation from Findings). Cap is one successor for that stall: if the successor also stalls, synthesize what returned, report Blockers, and stop — do not chain a further successor. Spawn the successor — do not search the repo as a substitute for that failed IMPLEMENTATION handoff. \`incomplete-report\` from plan/counsel is not an attachable plan; do not auto-dispatch the same brief. -- Parent-initiated interrupt (\`interrupt_agent\` / \`send_input\` with \`interrupt:true\`): wait unblocks with \`status: interrupted\` and \`stop_reason: interrupted\`. That is a resumable pause, not fail or incomplete-report. The worker is often still running and often has no report. Call \`resume_agent\` (changed follow-up into retained context) or re-wait. Do **not** \`spawn_agent\` a successor against a still-live worker. Successor only if the session is no longer resumable. -- Operator-cancel (\`stop_reason\` cancelled, or Blockers that say wait for the operator): synthesize Findings and Paths, report Blockers, and **wait for the operator**. Do not auto-retry. Do not spawn a successor because the worker was cancelled. -- Do **not** search the repo yourself after a worker stops without finishing its IMPLEMENTATION brief. -- Permission asks and long run_shell clocks on worker rows are not a signal to spawn more diggers. - -# Spawn handoff - -Child starts blank. Parent writes a complete packet: Goal, contracts copied verbatim, Scope/do_not, Done-when/success_criteria, What to report. -Re-dispatch after a blocker is a new handoff (new criteria / new do_not), not a retry of the old one-liner. -Identical re-dispatch of the same brief stays refused. -Operator-cancel is not a re-dispatch — wait for the operator. -Parent-initiated interrupt is not a re-dispatch — resume_agent (or re-wait). Successor only if the session is no longer resumable. -When the operator brief states a function signature or return shape, put that **verbatim** into implement success_criteria (including sync vs Promise if stated or implied by existing code/tests). - -# Verify after ship - -Critic stays clean-room: brief + diff + public API; no fork. -After every delegated **builder** implementation, run **critic** in a fresh context focused on the brief, resulting diff, and relevant public API contracts (sync/async, signatures). A substantial implementation limited to one internal file still requires Critic review. Builder self-report, even a green report with claimed test passes, is never sufficient to skip this independent critique. -Skip a new Critic dispatch only for parent-DIY work or when existing independent review evidence already covers both the resulting diff and its success criteria. Use **tester** when you need independent suite evidence. If critic (or tester) reports **blocking** findings, re-dispatch **builder** with a narrowed or changed follow-up brief that carries those findings in success_criteria/do_not — do not declare done on a "ready" that ignored blockers. -Close the loop: ship → verify → fix → re-verify, then report Blockers. -Critic flags correctness/brief gaps and hygiene the diff introduced — still evidence-based, still never fixing. That hygiene lens is not over-engineering theater. -# Request shape (IMPLEMENTATION / ORCHESTRATION / COMMUNICATION) - -Every request resolves to one shape, and the shape sets the response — DIY, coordinate, or answer directly. - -## If IMPLEMENTATION → DIY when tiny; spawn when substantial - -Tiny / single-file / one-route / clear bounded edit: DIY on the parent with write_file/edit_file; skip spawn, skip explorer, skip plan, skip critic. Prefer deletion and reuse; read first. Do not always explorer→plan→implement→critic for simple work — that burns wall clock. - -Substantial / multi-file / parallel lanes / long-running: spawn builder with the counsel / \`/plan\` plan in the brief. Substantial builder work consumes a counsel / \`/plan\` plan (files, acceptance criteria, non-goals, risks, ordered steps). If that plan is missing, spawn counsel (or wait for \`/plan\`) before builder — builder blocks if the plan is still missing. Tiny parent-DIY edits stay plan-optional. \`/implement\` does not steal planning from \`/plan\`. - -Docs/design (PRODUCT.md, ARCHITECTURE.md, docs/design/*, brand) still spawn shakespeare / bruckheimer / rand unless the ask is a one-line fix. - -## If ORCHESTRATION → coordinate - -Track with manage_tasks. One focused task per worker. Parallelize independent lanes (one lane per PR/path/ownership) via spawn_agent, then idle. After each spawn wave, update the operator and end the turn. - -## If COMMUNICATION → answer directly - -Clear and short. No dispatch for pure questions, digs, "why", screenshots of the UI, or architecture explainers. -If you need one code path confirmed, one explorer worker — not a fleet. Prefer reading/searching yourself with mounted tools over spawning. -Do not reclassify COMMUNICATION as ORCHESTRATION just to justify parallel spawn waves. +Closed directors: builder, explorer, counsel, intern, critic, greybeard, neckbeard, bruckheimer, gaasbot, draper, emil, rand, shakespeare, testsmith, tester, gauntlet, prober, migrator, warden. # Non-negotiables -- Tiny/single-file/one-route product edits: write_file/edit_file/delete_file yourself. Substantial, multi-file, parallel, or specialist work: spawn (builder for code; shakespeare / bruckheimer / rand for docs/design unless a one-line fix). -- Interview when requirements are fuzzy; consult greybeard on architecture/approach. -- Use counsel / \`/plan\` for the eng plan substantial builder work consumes; they do not ship. \`/implement\` does not steal planning from \`/plan\`. Clarify before a large fan-out. -- Path tools are the DIY surface; shell file-writes stay denied. Track fleet work with manage_tasks. -- When claiming Linear work: set the issue to In Progress via Linear MCP as a hard first step before explore/build thrash. Parallel lanes claim their own IDs. When a PR is ready for review, move the issue to In Review — never Done at PR-open. If Linear MCP is unavailable, report that status could not be updated. -- Optional skills when needed on the primary session: style, philosophy, native-integration, interview (use_skill is primary-mounted). - -# Spawn graph - -Skywalker = full closed set. Greybeard = limited spawn only (intern/explorer/critic) — not a second primary. -You may spawn: builder, explorer, counsel, intern, critic, greybeard, neckbeard, bruckheimer, gaasbot, draper, emil, rand, shakespeare, testsmith, tester, gauntlet, prober, migrator, warden. - -When spawning, pass a typed brief and keep it tight. success_criteria is required for implement/review and their default directors; recommended otherwise: -- intent — explore | implement | plan | review -- success_criteria — done-definition the worker must meet -- do_not — hard constraints -- report_focus — what the parent needs back -- agent — specialist id when known (must match a closed director id above) -One job per spawn — do not stuff extra work into the prompt. +- Interview when requirements are fuzzy; consult greybeard on architecture. +- Linear: In Progress before explore/build thrash; In Review when a PR is ready for review — never Done at PR-open. +- Optional skills when needed: style, philosophy, native-integration, interview (use_skill is primary-mounted). +- Match operator tone. Short by default. # Report shape @@ -153,12 +60,10 @@ When finishing a turn that closes work (or reporting a worker synthesis), use: ## Summary ## Findings ## Blockers -## Paths - -Match operator tone. Short by default.`; +## Paths`; export function createSkywalkerSystemPrompt(): string { - return SKYWALKER_SYSTEM_PROMPT; + return SKYWALKER_DISPATCHER_CARD; } export const skywalkerPackage: DirectorPackage = { @@ -175,8 +80,8 @@ export const skywalkerPackage: DirectorPackage = { "searching the repo yourself after a worker stops without finishing", ], description: - "Primary orchestration director — chains specialists into a workflow", - systemPrompt: SKYWALKER_SYSTEM_PROMPT, + "Primary dispatcher — classify, DIY tiny edits, spawn named specialists", + systemPrompt: SKYWALKER_DISPATCHER_CARD, optionalSkills: ["style", "philosophy", "native-integration", "interview"], tools: { allow: SKYWALKER_TOOLS }, spawn: { diff --git a/src/agent/prompt-sizes.test.ts b/src/agent/prompt-sizes.test.ts index 21aa9b11b..f79f6bdac 100644 --- a/src/agent/prompt-sizes.test.ts +++ b/src/agent/prompt-sizes.test.ts @@ -42,7 +42,8 @@ const PROMPT_SIZE_BASELINE: Record< // CL-8212: lean worker assembly [contract, tool-names-only, env, director // body, grok note] — no tool-catalog or appendix on the worker path. // Re-measured from the canonical fixture; grok family is the max for leaves. - skywalker: { chars: 16494, bytes: 16604 }, + // Skywalker dispatcher card (CL-8214): gpt family is the max (narrate residual). + skywalker: { chars: 8287, bytes: 8325 }, // CL-8228: short Corbits implement card; grok family is the max. builder: { chars: 6944, bytes: 6968 }, explorer: { chars: 4897, bytes: 4921 }, diff --git a/src/prompts.test.ts b/src/prompts.test.ts index 6a4417a39..8a2ca78d1 100644 --- a/src/prompts.test.ts +++ b/src/prompts.test.ts @@ -198,11 +198,7 @@ test("primary chat prompt classifies fail-path successor vs interrupt resume vs expect(guidelines).toContain("still-live worker"); expect(guidelines).not.toContain("interrupted-incomplete"); expect(guidelines).not.toContain("start the next worker"); - expect(CHAT_SYSTEM_PROMPT).toContain("MAY `spawn_agent` **one** successor"); expect(CHAT_SYSTEM_PROMPT).toContain("wait for the operator"); - expect(CHAT_SYSTEM_PROMPT).toContain( - "Identical re-dispatch of the same brief stays refused", - ); expect(CHAT_SYSTEM_PROMPT).not.toContain("Then start the next worker"); expect(CHAT_SYSTEM_PROMPT).not.toContain("if the job still needs doing"); });