Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
64 changes: 64 additions & 0 deletions .github/scripts/bench_watch.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
#!/usr/bin/env bash
# Source: senders/lightcone-cli.bench_watch.sh in LightconeResearch/lightcone-bench.
# Watch the lightcone-bench run dispatched for a PR commit and keep ONE PR comment
# (found by the marker below) up to date: the run link while it runs, the outcome
# when it ends. A dispatch returns no run id, so the run is found by the sha in its
# title — the bench workflow's run-name carries stack_ref for exactly this reason.
#
# Env: BENCH_TOKEN reads runs in lightcone-bench (the dispatch token)
# GH_TOKEN comments on this repo (the job's GITHUB_TOKEN, pull-requests: write)
# REPO owner/name of this repo PR pull request number
# SHA head sha being benchmarked POLL_SECONDS (default 600)
# MAX_POLLS give up after this many polls (default 18 → 3 h)
set -uo pipefail
BENCH=LightconeResearch/lightcone-bench
API=https://api.github.com
MARKER="<!-- lightcone-bench-run -->"
POLL_SECONDS="${POLL_SECONDS:-600}"
MAX_POLLS="${MAX_POLLS:-18}"
short="${SHA:0:7}"

bench_api() { curl -sf -H "Authorization: Bearer $BENCH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; }
repo_api() { curl -sf -H "Authorization: Bearer $GH_TOKEN" -H "Accept: application/vnd.github+json" "$@"; }

upsert_comment() { # $1 = markdown body; creates the marker comment or edits it in place
local payload id
payload=$(jq -n --arg m "$MARKER" --arg b "$1" '{body: ($m + "\n" + $b)}')
id=$(repo_api "$API/repos/$REPO/issues/$PR/comments?per_page=100" \
| jq -r --arg m "$MARKER" '[.[] | select(.body | startswith($m))] | first | .id // empty')
if [ -n "$id" ]; then
repo_api -X PATCH "$API/repos/$REPO/issues/comments/$id" -d "$payload" >/dev/null
else
repo_api -X POST "$API/repos/$REPO/issues/$PR/comments" -d "$payload" >/dev/null
fi
}

# 1. Find the run: it shows up a few seconds after the dispatch.
run_id=""; run_url=""
for _ in $(seq 1 12); do
sleep 5
read -r run_id run_url < <(bench_api "$API/repos/$BENCH/actions/workflows/benchmark.yml/runs?event=workflow_dispatch&per_page=30" \
| jq -r --arg sha "$SHA" '[.workflow_runs[] | select(.display_title | contains($sha))]
| sort_by(.created_at) | last | select(. != null) | "\(.id) \(.html_url)"')
[ -n "$run_id" ] && break
done
if [ -z "$run_id" ]; then
upsert_comment "**lightcone-bench** — dispatched a run for \`$short\` but could not find it in the queue; look under [lightcone-bench → Actions](https://github.com/$BENCH/actions/workflows/benchmark.yml)."
exit 0
fi
upsert_comment "**lightcone-bench** — benchmarking \`$short\`: [run $run_id]($run_url) ⏳ in progress (checked every $((POLL_SECONDS / 60)) min)."

# 2. Wait for it, editing the same comment when it ends.
for _ in $(seq 1 "$MAX_POLLS"); do
sleep "$POLL_SECONDS"
read -r status conclusion < <(bench_api "$API/repos/$BENCH/actions/runs/$run_id" | jq -r '"\(.status) \(.conclusion)"')
[ "${status:-}" = "completed" ] || continue
case "$conclusion" in
success) icon="✅"; word="passed" ;;
cancelled) icon="⚪"; word="was cancelled" ;;
*) icon="❌"; word="failed ($conclusion)" ;;
esac
upsert_comment "**lightcone-bench** — benchmark of \`$short\` $icon $word: [run $run_id]($run_url) — the matrix, baseline deltas and trace report are in its job summary."
exit 0
done
upsert_comment "**lightcone-bench** — benchmark of \`$short\` is still running after $((MAX_POLLS * POLL_SECONDS / 3600)) h: [run $run_id]($run_url)."
18 changes: 18 additions & 0 deletions .github/workflows/benchmark-baseline.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md).
# Every merge to main refreshes lightcone-bench's stored stack baseline (skills from
# agent-skills main installed, matching the /benchmark runs it is compared with).
name: benchmark-baseline
on:
push:
branches: [main]
jobs:
dispatch:
runs-on: ubuntu-latest
steps:
- env:
GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }}
run: |
curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \
-H "Accept: application/vnd.github+json" \
https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \
-d '{"ref":"main","inputs":{"task":"lightcone-cli/snae-build","use_skills":"true","stack_ref":"${{ github.sha }}","save_baseline":true}}'
58 changes: 58 additions & 0 deletions .github/workflows/benchmark-on-comment.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md).
# Comment "/benchmark" on a PR (maintainers only) to benchmark that PR's stack on
# demand — a draft PR, or a re-run — with the skills from agent-skills main
# installed (the skills axis is held fixed here; varying it is the agent-skills
# senders' job). One comment on the PR links the run and is edited with the
# outcome when it ends (checked every 10 minutes).
name: benchmark-on-comment
on:
issue_comment:
types: [created]
concurrency:
group: benchmark-watch-${{ github.event.issue.number }}
cancel-in-progress: true
jobs:
dispatch:
if: >
github.event.issue.pull_request &&
startsWith(github.event.comment.body, '/benchmark') &&
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association)
runs-on: ubuntu-latest
timeout-minutes: 200
permissions:
contents: read
pull-requests: write
steps:
- uses: actions/checkout@v4
- name: Resolve PR head sha
id: pr
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
sha=$(curl -sf -H "Authorization: Bearer $GH_TOKEN" \
"${{ github.event.issue.pull_request.url }}" | jq -r .head.sha)
echo "sha=$sha" >> "$GITHUB_OUTPUT"
- name: Dispatch lightcone-bench (stack under test = PR head)
env:
GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }}
SHA: ${{ steps.pr.outputs.sha }}
run: |
curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \
-H "Accept: application/vnd.github+json" \
https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \
-d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}"
- name: Ack with a rocket
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \
"https://api.github.com/repos/${{ github.repository }}/issues/comments/${{ github.event.comment.id }}/reactions" \
-d '{"content":"rocket"}'
- name: Link the run on the PR and report its outcome
env:
BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REPO: ${{ github.repository }}
PR: ${{ github.event.issue.number }}
SHA: ${{ steps.pr.outputs.sha }}
run: bash .github/scripts/bench_watch.sh
46 changes: 46 additions & 0 deletions .github/workflows/benchmark-on-pr.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
# Sender for LightconeResearch/lightcone-bench (source: senders/ in that repo; setup in its docs/pr-triggered-benchmarks.md).
# Every non-draft PR — opened, pushed to, reopened, or marked ready for review — is
# benchmarked against its head commit, with the skills from agent-skills main
# installed (the skills axis is held fixed here; varying it is the agent-skills
# senders' job). Draft PRs are not run automatically: comment "/benchmark" on them
# (benchmark-on-comment.yml) when a run is wanted. Fork PRs are skipped — they do
# not receive this repo's secrets. One comment on the PR links the run and is
# edited with the outcome when it ends (checked every 10 minutes).
name: benchmark-on-pr
on:
pull_request:
types: [opened, synchronize, reopened, ready_for_review]
concurrency:
# A new push replaces the watcher for this PR (the earlier bench run itself keeps
# going; its result just stops being reported here).
group: benchmark-watch-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
dispatch:
if: >
github.event.pull_request.draft == false &&
github.event.pull_request.head.repo.full_name == github.repository
runs-on: ubuntu-latest
timeout-minutes: 200
permissions:
contents: read
pull-requests: write
steps:
- uses: actions/checkout@v4
- name: Dispatch lightcone-bench (stack under test = PR head)
env:
GH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }}
SHA: ${{ github.event.pull_request.head.sha }}
run: |
curl -sf -X POST -H "Authorization: Bearer $GH_TOKEN" \
-H "Accept: application/vnd.github+json" \
https://api.github.com/repos/LightconeResearch/lightcone-bench/actions/workflows/benchmark.yml/dispatches \
-d "{\"ref\":\"main\",\"inputs\":{\"task\":\"lightcone-cli/snae-build\",\"use_skills\":\"true\",\"stack_ref\":\"$SHA\"}}"
- name: Link the run on the PR and report its outcome
env:
BENCH_TOKEN: ${{ secrets.BENCH_DISPATCH_TOKEN }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REPO: ${{ github.repository }}
PR: ${{ github.event.pull_request.number }}
SHA: ${{ github.event.pull_request.head.sha }}
run: bash .github/scripts/bench_watch.sh
Loading