From 7e490b628f8327781272320eef32ac0199aaa9a8 Mon Sep 17 00:00:00 2001 From: Poytr1 Date: Mon, 6 Jul 2026 10:26:20 +0800 Subject: [PATCH] feat(bench): aj bench --gen to generate+run repetition fixtures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the generate-and-run flow so `aj bench` can sweep the break-even curve without hand-authoring task files. - Fixture registry: FixtureByShape / FixtureShapes (currently "nullcheck"). - `aj bench --gen --n 1,2,4 [--workdir DIR]`: materializes the fixture at each repeat count (own subdir per count so they never collide) and runs it, composing with --arm and --compare. Verified with a real `claude` sweep: nullcheck n=1,2 both solved and dual-gate-verified (guards actually added), reporting real T2S per count. --dry-run correctly shows 0% / no-T2S on the unmodified fixtures. Refs: benchmark sandbox design §3.2, Phase 3. Co-Authored-By: Claude Opus 4.8 --- cmd/aj/bench_cmd.go | 65 ++++++++++++++++++++++++++++++---- internal/bench/fixture.go | 22 ++++++++++++ internal/bench/fixture_test.go | 17 +++++++++ 3 files changed, 98 insertions(+), 6 deletions(-) diff --git a/cmd/aj/bench_cmd.go b/cmd/aj/bench_cmd.go index 8fbde44..1c70027 100644 --- a/cmd/aj/bench_cmd.go +++ b/cmd/aj/bench_cmd.go @@ -5,6 +5,7 @@ import ( "encoding/json" "fmt" "os" + "path/filepath" "github.com/agent-jit/agentjit/internal/bench" "github.com/agent-jit/agentjit/internal/config" @@ -20,17 +21,24 @@ var ( benchDryRun bool benchCompare bool benchCompileCost int + benchGen string + benchN []int + benchWorkdir string ) var benchCmd = &cobra.Command{ Use: "bench", Short: "Benchmark AgentJIT skill ROI (baseline vs JIT) on a task suite", - Long: "Runs each task in a JSONL suite for N rollouts under an arm and reports\n" + - "tokens-to-success at iso-accuracy — only verified rollouts count.\n" + + Long: "Runs tasks for N rollouts under an arm and reports tokens-to-success at\n" + + "iso-accuracy — only verified rollouts count. Load tasks from a JSONL suite\n" + + "(--tasks) or generate a repetition fixture (--gen --n 1,2,4).\n" + "Point AJ at an isolated sandbox with AJ_HOME so real data is untouched.", RunE: func(cmd *cobra.Command, args []string) error { - if benchTasksFile == "" { - return fmt.Errorf("--tasks is required (a JSONL task file)") + if benchTasksFile == "" && benchGen == "" { + return fmt.Errorf("provide --tasks or --gen ") + } + if benchTasksFile != "" && benchGen != "" { + return fmt.Errorf("--tasks and --gen are mutually exclusive") } arm := bench.Arm(benchArm) if arm != bench.ArmBaseline && arm != bench.ArmJIT { @@ -40,9 +48,15 @@ var benchCmd = &cobra.Command{ return fmt.Errorf("--rollouts must be >= 1") } - tasks, err := bench.LoadTasks(benchTasksFile) + var tasks []bench.Task + var err error + if benchGen != "" { + tasks, err = generateTasks(benchGen, benchN, benchWorkdir) + } else { + tasks, err = bench.LoadTasks(benchTasksFile) + } if err != nil { - return fmt.Errorf("loading tasks: %w", err) + return err } if len(tasks) == 0 { fmt.Println("[AJ] No tasks in suite.") @@ -140,6 +154,42 @@ func printComparison(c bench.Comparison, compileCost int) { fmt.Println() } +// generateTasks materializes a repetition fixture at each requested repeat count +// under workdir (a temp dir when empty), returning one Task per count so a +// benchmark can sweep the break-even curve. +func generateTasks(shape string, counts []int, workdir string) ([]bench.Task, error) { + fixture, ok := bench.FixtureByShape(shape) + if !ok { + return nil, fmt.Errorf("unknown --gen shape %q (have: %v)", shape, bench.FixtureShapes()) + } + if len(counts) == 0 { + counts = []int{3} + } + if workdir == "" { + dir, err := os.MkdirTemp("", "aj-bench-") + if err != nil { + return nil, err + } + workdir = dir + fmt.Printf("[AJ] Generated fixtures under %s\n", workdir) + } + + tasks := make([]bench.Task, 0, len(counts)) + for _, n := range counts { + // Each count gets its own subdir so fixtures never collide. + dir := filepath.Join(workdir, fmt.Sprintf("%s-%d", shape, n)) + if err := os.MkdirAll(dir, 0o755); err != nil { + return nil, err + } + task, err := fixture.Generate(dir, n) + if err != nil { + return nil, fmt.Errorf("generate %s n=%d: %w", shape, n, err) + } + tasks = append(tasks, task) + } + return tasks, nil +} + // modelArgs returns claude CLI args for a fixed model, or nil to use the default. func modelArgs(model string) []string { if model == "" { @@ -174,5 +224,8 @@ func init() { benchCmd.Flags().BoolVar(&benchDryRun, "dry-run", false, "Exercise the harness without invoking claude") benchCmd.Flags().BoolVar(&benchCompare, "compare", false, "Run both arms per task and report baseline-vs-JIT") benchCmd.Flags().IntVar(&benchCompileCost, "compile-cost", 0, "Skill compile cost (tokens) for break-even; auto-read from AJ_HOME stats if unset") + benchCmd.Flags().StringVar(&benchGen, "gen", "", "Generate a repetition fixture by shape instead of --tasks (e.g. nullcheck)") + benchCmd.Flags().IntSliceVar(&benchN, "n", nil, "Repeat counts for --gen (e.g. --n 1,2,4 sweeps the curve)") + benchCmd.Flags().StringVar(&benchWorkdir, "workdir", "", "Where --gen writes fixtures (default: a temp dir)") rootCmd.AddCommand(benchCmd) } diff --git a/internal/bench/fixture.go b/internal/bench/fixture.go index c93241f..5f96b65 100644 --- a/internal/bench/fixture.go +++ b/internal/bench/fixture.go @@ -4,8 +4,30 @@ import ( "fmt" "os" "path/filepath" + "sort" ) +// fixtures is the registry of built-in repetition fixtures, keyed by shape name. +var fixtures = map[string]Fixture{ + "nullcheck": NullCheckFixture{}, +} + +// FixtureByShape returns the built-in fixture for a shape name. +func FixtureByShape(shape string) (Fixture, bool) { + f, ok := fixtures[shape] + return f, ok +} + +// FixtureShapes returns the sorted list of registered fixture shape names. +func FixtureShapes() []string { + names := make([]string, 0, len(fixtures)) + for k := range fixtures { + names = append(names, k) + } + sort.Strings(names) + return names +} + // Fixture materializes a self-contained, reproducible workspace for a // repetition-parameterized task (same shape repeated N times), and reports the // Task (prompt + verifier) that runs against it. Generating the workspace makes diff --git a/internal/bench/fixture_test.go b/internal/bench/fixture_test.go index de89464..b6b8ecb 100644 --- a/internal/bench/fixture_test.go +++ b/internal/bench/fixture_test.go @@ -84,3 +84,20 @@ func TestNullCheckVerifierPassesAfterFix(t *testing.T) { t.Error("verifier failed after adding all guards; expected pass") } } + +func TestFixtureRegistry(t *testing.T) { + f, ok := FixtureByShape("nullcheck") + if !ok { + t.Fatal("nullcheck fixture not registered") + } + if f.Shape() != "nullcheck" { + t.Errorf("Shape() = %q, want nullcheck", f.Shape()) + } + if _, ok := FixtureByShape("does-not-exist"); ok { + t.Error("unknown shape reported as registered") + } + shapes := FixtureShapes() + if len(shapes) == 0 { + t.Error("FixtureShapes() is empty") + } +}