diff --git a/.github/scripts/bench_watch.sh b/.github/scripts/bench_watch.sh new file mode 100755 index 00000000..dc1414d3 --- /dev/null +++ b/.github/scripts/bench_watch.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Source: senders/lightcone-cli.bench_watch.sh in LightconeResearch/lightcone-bench. +# Watch the lightcone-bench run dispatched for a PR commit and keep ONE PR comment +# (found by the marker below) up to date: the run link while it runs, the outcome +# when it ends. A dispatch returns no run id, so the run is found by the sha in its +# title — the bench workflow's run-name carries stack_ref for exactly this reason. +# +# Env: BENCH_TOKEN reads runs in lightcone-bench (the dispatch token) +# GH_TOKEN comments on this repo (the job's GITHUB_TOKEN, pull-requests: write) +# REPO owner/name of this repo PR pull request number +# SHA head sha being benchmarked POLL_SECONDS (default 600) +# MAX_POLLS give up after this many polls (default 18 → 3 h) +set -uo pipefail +BENCH=LightconeResearch/lightcone-bench +API=https://api.github.com +MARKER="" +POLL_SECONDS="${POLL_SECONDS:-600}" +MAX_POLLS="${MAX_POLLS:-18}" +short="${SHA:0:7}" + +bench_api() { curl -sf -H "Authorization: Bearer $BENCH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; } +repo_api() { curl -sf -H "Authorization: Bearer $GH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; } + +upsert_comment() { # $1 = markdown body; creates the marker comment or edits it in place + local payload id + payload=$(jq -n --arg m "$MARKER" --arg b "$1" '{body: ($m + "\n" + $b)}') + id=$(repo_api "$API/repos/$REPO/issues/$PR/comments?per_page=100" \ + | jq -r --arg m "$MARKER" '[.[] | select(.body | startswith($m))] | first | .id // empty') + if [ -n "$id" ]; then + repo_api -X PATCH "$API/repos/$REPO/issues/comments/$id" -d "$payload" >/dev/null + else + repo_api -X POST "$API/repos/$REPO/issues/$PR/comments" -d "$payload" >/dev/null + fi +} + +# 1. Find the run: it shows up a few seconds after the dispatch. +run_id=""; run_url="" +for _ in $(seq 1 12); do + sleep 5 + read -r run_id run_url < <(bench_api "$API/repos/$BENCH/actions/workflows/benchmark.yml/runs?event=workflow_dispatch&per_page=30" \ + | jq -r --arg sha "$SHA" '[.workflow_runs[] | select(.display_title | contains($sha))] + | sort_by(.created_at) | last | select(. != null) | "\(.id) \(.html_url)"') + [ -n "$run_id" ] && break +done +if [ -z "$run_id" ]; then + upsert_comment "**lightcone-bench** — dispatched a run for \`$short\` but could not find it in the queue; look under [lightcone-bench → Actions](https://github.com/$BENCH/actions/workflows/benchmark.yml)." + exit 0 +fi +upsert_comment "**lightcone-bench** — benchmarking \`$short\`: [run $run_id]($run_url) ⏳ in progress (checked every $((POLL_SECONDS / 60)) min)." + +# 2. Wait for it, editing the same comment when it ends. +for _ in $(seq 1 "$MAX_POLLS"); do + sleep "$POLL_SECONDS" + read -r status conclusion < <(bench_api "$API/repos/$BENCH/actions/runs/$run_id" | jq -r '"\(.status) \(.conclusion)"') + [ "${status:-}" = "completed" ] || continue + case "$conclusion" in + success) icon="✅"; word="passed" ;; + cancelled) icon="⚪"; word="was cancelled" ;; + *) icon="❌"; word="failed ($conclusion)" ;; + esac + upsert_comment "**lightcone-bench** — benchmark of \`$short\` $icon $word: [run $run_id]($run_url) — the matrix, baseline deltas and trace report are in its job summary." + exit 0 +done +upsert_comment "**lightcone-bench** — benchmark of \`$short\` is still running after $((MAX_POLLS * POLL_SECONDS / 3600)) h: [run $run_id]($run_url)." diff --git a/.github/workflows/benchmark-baseline.yml b/.github/workflows/benchmark-baseline.yml new file mode 100644 index 00000000..7a97612e --- /dev/null +++ b/.github/workflows/benchmark-baseline.yml @@ -0,0 +1,18 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Every merge to main refreshes lightcone-bench's stored stack baseline (skills from +# agent-skills main installed, matching the /benchmark runs it is compared with). +name: benchmark-baseline +on: + push: + branches: [main] +jobs: + dispatch: + runs-on: ubuntu-latest + steps: + - env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","use_skills":"true","stack_ref":"${{ github.sha }}","save_baseline":true}}' diff --git a/.github/workflows/benchmark-on-comment.yml b/.github/workflows/benchmark-on-comment.yml new file mode 100644 index 00000000..af07fad1 --- /dev/null +++ b/.github/workflows/benchmark-on-comment.yml @@ -0,0 +1,58 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack on +# demand — a draft PR, or a re-run — with the skills from agent-skills main +# installed (the skills axis is held fixed here; varying it is the agent-skills +# senders' job). One comment on the PR links the run and is edited with the +# outcome when it ends (checked every 10 minutes). +name: benchmark-on-comment +on: + issue_comment: + types: [created] +concurrency: + group: benchmark-watch-${{ github.event.issue.number }} + cancel-in-progress: true +jobs: + dispatch: + if: > + github.event.issue.pull_request && + startsWith(github.event.comment.body, '/benchmark') && + contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association) + runs-on: ubuntu-latest + timeout-minutes: 200 + permissions: + contents: read + pull-requests: write + steps: + - uses: actions/checkout@v4 + - name: Resolve PR head sha + id: pr + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + sha=$(curl -sf -H "Authorization: Bearer $GH_TOKEN" \ + "${{ github.event.issue.pull_request.url }}" | jq -r .head.sha) + echo "sha=$sha" >> "$GITHUB_OUTPUT" + - name: Dispatch lightcone-bench (stack under test = PR head) + env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + SHA: ${{ steps.pr.outputs.sha }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}" + - name: Ack with a rocket + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + "https://api.github.com/repos/${{ github.repository }}/issues/comments/${{ github.event.comment.id }}/reactions" \ + -d '{"content":"rocket"}' + - name: Link the run on the PR and report its outcome + env: + BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REPO: ${{ github.repository }} + PR: ${{ github.event.issue.number }} + SHA: ${{ steps.pr.outputs.sha }} + run: bash .github/scripts/bench_watch.sh diff --git a/.github/workflows/benchmark-on-pr.yml b/.github/workflows/benchmark-on-pr.yml new file mode 100644 index 00000000..9772fbc9 --- /dev/null +++ b/.github/workflows/benchmark-on-pr.yml @@ -0,0 +1,46 @@ +# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md). +# Every non-draft PR — opened, pushed to, reopened, or marked ready for review — is +# benchmarked against its head commit, with the skills from agent-skills main +# installed (the skills axis is held fixed here; varying it is the agent-skills +# senders' job). Draft PRs are not run automatically: comment "/benchmark" on them +# (benchmark-on-comment.yml) when a run is wanted. Fork PRs are skipped — they do +# not receive this repo's secrets. One comment on the PR links the run and is +# edited with the outcome when it ends (checked every 10 minutes). +name: benchmark-on-pr +on: + pull_request: + types: [opened, synchronize, reopened, ready_for_review] +concurrency: + # A new push replaces the watcher for this PR (the earlier bench run itself keeps + # going; its result just stops being reported here). + group: benchmark-watch-${{ github.event.pull_request.number }} + cancel-in-progress: true +jobs: + dispatch: + if: > + github.event.pull_request.draft == false && + github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + timeout-minutes: 200 + permissions: + contents: read + pull-requests: write + steps: + - uses: actions/checkout@v4 + - name: Dispatch lightcone-bench (stack under test = PR head) + env: + GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + SHA: ${{ github.event.pull_request.head.sha }} + run: | + curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \ + -d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}" + - name: Link the run on the PR and report its outcome + env: + BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + REPO: ${{ github.repository }} + PR: ${{ github.event.pull_request.number }} + SHA: ${{ github.event.pull_request.head.sha }} + run: bash .github/scripts/bench_watch.sh