From 68501d0de5c7e5de0daa057aab32bb61cc26d447 Mon Sep 17 00:00:00 2001 From: dkn16 Date: Fri, 2 Oct 2026 15:29:15 -0700 Subject: [PATCH 1/4] CI: dispatch lightcone-bench on /benchmark comments and on merges to main Two sender workflows, copied from lightcone-bench's senders/: - benchmark-on-comment: a maintainer comment "/benchmark" on a PR (author association OWNER/MEMBER/COLLABORATOR) dispatches lightcone-bench's benchmark workflow with stack_ref = the PR head sha, task lightcone-cli/snae-build, and acks with a rocket reaction. Expensive runs happen only on human judgment. - benchmark-baseline: every push to main dispatches the same with save_baseline, so the stored "current main" numbers in lightcone-bench track this repo. Both need one repository secret, BENCH_DISPATCH_TOKEN: a fine-grained PAT (or App token) with Actions read/write on LightconeResearch/lightcone-bench. The built-in GITHUB_TOKEN cannot dispatch into another repo. snae-build is lightcone-cli's own eval (evals/prompt.md + evals/tasks/snae) run through Harbor: oracle-gated, agent-agnostic, with the skills axis measurable. Results, traces and baselines live in lightcone-bench. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01VcKmGNUek7oWEPv8zLqo9Q Signed-off-by: dkn16 --- .github/workflows/benchmark-baseline.yml | 17 ++++++++++ .github/workflows/benchmark-on-comment.yml | 39 ++++++++++++++++++++++ 2 files changed, 56 insertions(+) create mode 100644 .github/workflows/benchmark-baseline.yml create mode 100644 .github/workflows/benchmark-on-comment.yml diff --git a/.github/workflows/benchmark-baseline.yml b/.github/workflows/benchmark-baseline.yml new file mode 100644 index 00000000..7623cd2d --- /dev/null +++ b/.github/workflows/benchmark-baseline.yml @@ -0,0 +1,17 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Every merge to main refreshes lightcone-bench's stored stack baseline. +name: benchmark-baseline +on: + push: + branches: [main] +jobs: + dispatch: + runs-on: ubuntu-latest + steps: + - env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","stack_ref":"${{ github.sha }}","save_baseline":true}}' diff --git a/.github/workflows/benchmark-on-comment.yml b/.github/workflows/benchmark-on-comment.yml new file mode 100644 index 00000000..1e377ff3 --- /dev/null +++ b/.github/workflows/benchmark-on-comment.yml @@ -0,0 +1,39 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack. +name: benchmark-on-comment +on: + issue_comment: + types: [created] +jobs: + dispatch: + if: > + github.event.issue.pull_request && + startsWith(github.event.comment.body, '/benchmark') && + contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association) + runs-on: ubuntu-latest + permissions: + pull-requests: write # for the reaction ack + steps: + - name: Resolve PR head sha + id: pr + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + sha=$(curl -sf -H "Authorization: Bearer $GH_TOKEN" \ + "${{ github.event.issue.pull_request.url }}" | jq -r .head.sha) + echo "sha=$sha" >> "$GITHUB_OUTPUT" + - name: Dispatch lightcone-bench (stack under test = PR head) + env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","stack_ref":"${{ steps.pr.outputs.sha }}"}}' + - name: Ack with a rocket + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + "https://api.github.com/repos/${{ github.repository }}/issues/comments/${{ github.event.comment.id }}/reactions" \ + -d '{"content":"rocket"}' From 65128b95c66c2f2d8de7839c35222a5e74ee7825 Mon Sep 17 00:00:00 2001 From: dkn16 Date: Fri, 2 Oct 2026 15:39:36 -0700 Subject: [PATCH 2/4] CI senders: install the skills from agent-skills main on stack runs The skills axis is held fixed (agent-skills main) on both the /benchmark runs and the baseline refresh, so the comparison stays matched; varying the skills is the agent-skills repo's own sender's job. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01VcKmGNUek7oWEPv8zLqo9Q Signed-off-by: dkn16 --- .github/workflows/benchmark-baseline.yml | 5 +++-- .github/workflows/benchmark-on-comment.yml | 6 ++++-- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/.github/workflows/benchmark-baseline.yml b/.github/workflows/benchmark-baseline.yml index 7623cd2d..7a97612e 100644 --- a/.github/workflows/benchmark-baseline.yml +++ b/.github/workflows/benchmark-baseline.yml @@ -1,5 +1,6 @@ # Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). -# Every merge to main refreshes lightcone-bench's stored stack baseline. +# Every merge to main refreshes lightcone-bench's stored stack baseline (skills from +# agent-skills main installed, matching the /benchmark runs it is compared with). name: benchmark-baseline on: push: @@ -14,4 +15,4 @@ jobs: curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ -H "Accept: application/vnd.github+json" \ https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ - -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","stack_ref":"${{ github.sha }}","save_baseline":true}}' + -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","use_skills":"true","stack_ref":"${{ github.sha }}","save_baseline":true}}' diff --git a/.github/workflows/benchmark-on-comment.yml b/.github/workflows/benchmark-on-comment.yml index 1e377ff3..b9c33031 100644 --- a/.github/workflows/benchmark-on-comment.yml +++ b/.github/workflows/benchmark-on-comment.yml @@ -1,5 +1,7 @@ # Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). -# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack. +# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack, +# with the skills from agent-skills main installed (the skills axis is held fixed +# here; varying it is the agent-skills senders' job). name: benchmark-on-comment on: issue_comment: @@ -29,7 +31,7 @@ jobs: curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ -H "Accept: application/vnd.github+json" \ https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ - -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","stack_ref":"${{ steps.pr.outputs.sha }}"}}' + -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","use_skills":"true","stack_ref":"${{ steps.pr.outputs.sha }}"}}' - name: Ack with a rocket env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} From 2df3e7c335d7bfdb6ce57dc7ba8426c43f771b95 Mon Sep 17 00:00:00 2001 From: dkn16 Date: Fri, 2 Oct 2026 15:56:29 -0700 Subject: [PATCH 3/4] CI: benchmark every non-draft PR automatically; /benchmark stays for drafts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit benchmark-on-pr.yml dispatches lightcone-bench for every non-draft PR — opened, pushed to, reopened, or marked ready for review — against its head sha, with the skills from agent-skills main installed. Fork PRs are skipped (they do not receive this repo's secrets). benchmark-on-comment.yml remains the on-demand path: draft PRs, or a re-run. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01VcKmGNUek7oWEPv8zLqo9Q Signed-off-by: dkn16 --- .github/workflows/benchmark-on-pr.yml | 28 +++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 .github/workflows/benchmark-on-pr.yml diff --git a/.github/workflows/benchmark-on-pr.yml b/.github/workflows/benchmark-on-pr.yml new file mode 100644 index 00000000..438fac8a --- /dev/null +++ b/.github/workflows/benchmark-on-pr.yml @@ -0,0 +1,28 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Every non-draft PR — opened, pushed to, reopened, or marked ready for review — is +# benchmarked against its head commit, with the skills from agent-skills main +# installed (the skills axis is held fixed here; varying it is the agent-skills +# senders' job). Draft PRs are not run automatically: comment "/benchmark" on them +# (benchmark-on-comment.yml) when a run is wanted. Fork PRs are skipped — they do +# not receive this repo's secrets. +name: benchmark-on-pr +on: + pull_request: + types: [opened, synchronize, reopened, ready_for_review] +jobs: + dispatch: + if: > + github.event.pull_request.draft == false && + github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + steps: + - name: Dispatch lightcone-bench (stack under test = PR head) + env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + SHA: ${{ github.event.pull_request.head.sha }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}" + echo "Dispatched benchmark for $SHA — results in lightcone-bench's Actions tab" From 36a0be29418a686f4979fe1da0da69555476194f Mon Sep 17 00:00:00 2001 From: dkn16 Date: Fri, 2 Oct 2026 16:05:46 -0700 Subject: [PATCH 4/4] CI: link the benchmark run on the PR and report its outcome Both PR workflows now run .github/scripts/bench_watch.sh after dispatching: it finds the lightcone-bench run (by the PR sha in the run's title), posts one comment on the PR with the link, then checks every 10 minutes and edits that comment with the outcome when the run ends. A new push replaces the watcher, so a PR never accumulates comments. Uses the dispatch token to read runs and the job's GITHUB_TOKEN (pull-requests: write) to comment. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01VcKmGNUek7oWEPv8zLqo9Q Signed-off-by: dkn16 --- .github/scripts/bench_watch.sh | 64 ++++++++++++++++++++++ .github/workflows/benchmark-on-comment.yml | 27 +++++++-- .github/workflows/benchmark-on-pr.yml | 22 +++++++- 3 files changed, 106 insertions(+), 7 deletions(-) create mode 100755 .github/scripts/bench_watch.sh diff --git a/.github/scripts/bench_watch.sh b/.github/scripts/bench_watch.sh new file mode 100755 index 00000000..dc1414d3 --- /dev/null +++ b/.github/scripts/bench_watch.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Source: senders/lightcone-cli.bench_watch.sh in LightconeResearch/lightcone-bench. +# Watch the lightcone-bench run dispatched for a PR commit and keep ONE PR comment +# (found by the marker below) up to date: the run link while it runs, the outcome +# when it ends. A dispatch returns no run id, so the run is found by the sha in its +# title — the bench workflow's run-name carries stack_ref for exactly this reason. +# +# Env: BENCH_TOKEN reads runs in lightcone-bench (the dispatch token) +# GH_TOKEN comments on this repo (the job's GITHUB_TOKEN, pull-requests: write) +# REPO owner/name of this repo PR pull request number +# SHA head sha being benchmarked POLL_SECONDS (default 600) +# MAX_POLLS give up after this many polls (default 18 → 3 h) +set -uo pipefail +BENCH=LightconeResearch/lightcone-bench +API=https://api.github.com +MARKER="" +POLL_SECONDS="${POLL_SECONDS:-600}" +MAX_POLLS="${MAX_POLLS:-18}" +short="${SHA:0:7}" + +bench_api() { curl -sf -H "Authorization: Bearer $BENCH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; } +repo_api() { curl -sf -H "Authorization: Bearer $GH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; } + +upsert_comment() { # $1 = markdown body; creates the marker comment or edits it in place + local payload id + payload=$(jq -n --arg m "$MARKER" --arg b "$1" '{body: ($m + "\n" + $b)}') + id=$(repo_api "$API/repos/$REPO/issues/$PR/comments?per_page=100" \ + | jq -r --arg m "$MARKER" '[.[] | select(.body | startswith($m))] | first | .id // empty') + if [ -n "$id" ]; then + repo_api -X PATCH "$API/repos/$REPO/issues/comments/$id" -d "$payload" >/dev/null + else + repo_api -X POST "$API/repos/$REPO/issues/$PR/comments" -d "$payload" >/dev/null + fi +} + +# 1. Find the run: it shows up a few seconds after the dispatch. +run_id=""; run_url="" +for _ in $(seq 1 12); do + sleep 5 + read -r run_id run_url < <(bench_api "$API/repos/$BENCH/actions/workflows/benchmark.yml/runs?event=workflow_dispatch&per_page=30" \ + | jq -r --arg sha "$SHA" '[.workflow_runs[] | select(.display_title | contains($sha))] + | sort_by(.created_at) | last | select(. != null) | "\(.id) \(.html_url)"') + [ -n "$run_id" ] && break +done +if [ -z "$run_id" ]; then + upsert_comment "**lightcone-bench** — dispatched a run for \`$short\` but could not find it in the queue; look under [lightcone-bench → Actions](https://github.com/$BENCH/actions/workflows/benchmark.yml)." + exit 0 +fi +upsert_comment "**lightcone-bench** — benchmarking \`$short\`: [run $run_id]($run_url) ⏳ in progress (checked every $((POLL_SECONDS / 60)) min)." + +# 2. Wait for it, editing the same comment when it ends. +for _ in $(seq 1 "$MAX_POLLS"); do + sleep "$POLL_SECONDS" + read -r status conclusion < <(bench_api "$API/repos/$BENCH/actions/runs/$run_id" | jq -r '"\(.status) \(.conclusion)"') + [ "${status:-}" = "completed" ] || continue + case "$conclusion" in + success) icon="✅"; word="passed" ;; + cancelled) icon="⚪"; word="was cancelled" ;; + *) icon="❌"; word="failed ($conclusion)" ;; + esac + upsert_comment "**lightcone-bench** — benchmark of \`$short\` $icon $word: [run $run_id]($run_url) — the matrix, baseline deltas and trace report are in its job summary." + exit 0 +done +upsert_comment "**lightcone-bench** — benchmark of \`$short\` is still running after $((MAX_POLLS * POLL_SECONDS / 3600)) h: [run $run_id]($run_url)." diff --git a/.github/workflows/benchmark-on-comment.yml b/.github/workflows/benchmark-on-comment.yml index b9c33031..af07fad1 100644 --- a/.github/workflows/benchmark-on-comment.yml +++ b/.github/workflows/benchmark-on-comment.yml @@ -1,11 +1,16 @@ # Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). -# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack, -# with the skills from agent-skills main installed (the skills axis is held fixed -# here; varying it is the agent-skills senders' job). +# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack on +# demand — a draft PR, or a re-run — with the skills from agent-skills main +# installed (the skills axis is held fixed here; varying it is the agent-skills +# senders' job). One comment on the PR links the run and is edited with the +# outcome when it ends (checked every 10 minutes). name: benchmark-on-comment on: issue_comment: types: [created] +concurrency: + group: benchmark-watch-${{ github.event.issue.number }} + cancel-in-progress: true jobs: dispatch: if: > @@ -13,9 +18,12 @@ jobs: startsWith(github.event.comment.body, '/benchmark') && contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association) runs-on: ubuntu-latest + timeout-minutes: 200 permissions: - pull-requests: write # for the reaction ack + contents: read + pull-requests: write steps: + - uses: actions/checkout@v4 - name: Resolve PR head sha id: pr env: @@ -27,11 +35,12 @@ jobs: - name: Dispatch lightcone-bench (stack under test = PR head) env: GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + SHA: ${{ steps.pr.outputs.sha }} run: | curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ -H "Accept: application/vnd.github+json" \ https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ - -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","use_skills":"true","stack_ref":"${{ steps.pr.outputs.sha }}"}}' + -d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}" - name: Ack with a rocket env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} @@ -39,3 +48,11 @@ jobs: curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ "https://api.github.com/repos/${{ github.repository }}/issues/comments/${{ github.event.comment.id }}/reactions" \ -d '{"content":"rocket"}' + - name: Link the run on the PR and report its outcome + env: + BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REPO: ${{ github.repository }} + PR: ${{ github.event.issue.number }} + SHA: ${{ steps.pr.outputs.sha }} + run: bash .github/scripts/bench_watch.sh diff --git a/.github/workflows/benchmark-on-pr.yml b/.github/workflows/benchmark-on-pr.yml index 438fac8a..9772fbc9 100644 --- a/.github/workflows/benchmark-on-pr.yml +++ b/.github/workflows/benchmark-on-pr.yml @@ -4,18 +4,29 @@ # installed (the skills axis is held fixed here; varying it is the agent-skills # senders' job). Draft PRs are not run automatically: comment "/benchmark" on them # (benchmark-on-comment.yml) when a run is wanted. Fork PRs are skipped — they do -# not receive this repo's secrets. +# not receive this repo's secrets. One comment on the PR links the run and is +# edited with the outcome when it ends (checked every 10 minutes). name: benchmark-on-pr on: pull_request: types: [opened, synchronize, reopened, ready_for_review] +concurrency: + # A new push replaces the watcher for this PR (the earlier bench run itself keeps + # going; its result just stops being reported here). + group: benchmark-watch-${{ github.event.pull_request.number }} + cancel-in-progress: true jobs: dispatch: if: > github.event.pull_request.draft == false && github.event.pull_request.head.repo.full_name == github.repository runs-on: ubuntu-latest + timeout-minutes: 200 + permissions: + contents: read + pull-requests: write steps: + - uses: actions/checkout@v4 - name: Dispatch lightcone-bench (stack under test = PR head) env: GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} @@ -25,4 +36,11 @@ jobs: -H "Accept: application/vnd.github+json" \ https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ -d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}" - echo "Dispatched benchmark for $SHA — results in lightcone-bench's Actions tab" + - name: Link the run on the PR and report its outcome + env: + BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REPO: ${{ github.repository }} + PR: ${{ github.event.pull_request.number }} + SHA: ${{ github.event.pull_request.head.sha }} + run: bash .github/scripts/bench_watch.sh