From 0113528809bdd92b239cd48bfcd7b7d66eb37b30 Mon Sep 17 00:00:00 2001 From: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> Date: Tue, 15 Sep 2026 13:01:33 -0700 Subject: [PATCH 1/4] Benchmark Hyperlight host 0.17.0, JS 0.4.0, Wasm 0.15.0 Additional runtime versions: * Wasmtime: 36.0.14 * jco: 1.16.0 * componentize-js: 0.19.3 * componentize-qjs-cli: 0.3.0 Build and load tools: * cargo-hyperlight: 0.1.14 * wasm-tools: 1.243.0 * oha: 1.9.0 Signed-off-by: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> --- README.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/README.md b/README.md index e7d0e3c..1ece193 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,7 @@ # Hyperlight HTTP Benchmarks +[View the dashboard](https://hyperlight-dev.github.io/hyperlight-bench/). + HTTP benchmarks and a results dashboard for Hyperlight. Compare throughput, latency, and memory use across JavaScript and WebAssembly runtimes on Linux KVM and MSHV. From 0ad9e6485dca690c8b8e55247bdf58997c95fbb3 Mon Sep 17 00:00:00 2001 From: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> Date: Tue, 15 Sep 2026 13:27:40 -0700 Subject: [PATCH 2/4] Cancel superseded PR benchmark runs Signed-off-by: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> --- .github/workflows/benchmark.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index 8e27196..4c0edb6 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -12,7 +12,7 @@ permissions: concurrency: group: benchmark-${{ github.event.pull_request.number }} - cancel-in-progress: false + cancel-in-progress: true env: CARGO_TERM_COLOR: always From 893a2e8c20508c957220f412adcb31a40265b8d2 Mon Sep 17 00:00:00 2001 From: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> Date: Tue, 15 Sep 2026 14:36:32 -0700 Subject: [PATCH 3/4] Support partial benchmark retries and increase HTTP timeout Signed-off-by: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> --- .github/workflows/benchmark.yml | 7 ++- docs/README.md | 8 +++- docs/results.md | 7 ++- scripts/benchmark.ts | 55 +++++++++++++++------ scripts/build-harness.ts | 5 +- scripts/merge-results.ts | 48 ++++++++++++++----- scripts/publication.test.ts | 85 ++++++++++++++++++++++++++++++++- scripts/publish-ci.ts | 21 +++++--- shared/catalog.ts | 2 +- 9 files changed, 198 insertions(+), 40 deletions(-) diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index 4c0edb6..38989da 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -67,6 +67,7 @@ jobs: name: run-identity path: artifacts/run.json if-no-files-found: error + overwrite: true # Build guest inputs once on KVM for all measurement configurations. producer: @@ -101,6 +102,7 @@ jobs: - uses: actions/upload-artifact@v7 with: name: guest-inputs + overwrite: true path: | js/src/wit/handler_wit.wasm js/default/handler.wasm @@ -191,6 +193,7 @@ jobs: if: always() with: name: result-${{ matrix.platform }}-${{ matrix.runtime }}-${{ matrix.strategy }} + overwrite: true path: results/ if-no-files-found: error retention-days: 14 @@ -216,8 +219,8 @@ jobs: with: pattern: result-* path: results/ - merge-multiple: true - - run: node scripts/merge-results.ts --output run.json results/*/bundle.json + - name: Merge measurements + run: node scripts/merge-results.ts --output run.json results/*/*/bundle.json - uses: actions/upload-artifact@v7 with: name: run-${{ github.run_id }}-${{ github.run_attempt }} diff --git a/docs/README.md b/docs/README.md index 41aed9f..e5a3037 100644 --- a/docs/README.md +++ b/docs/README.md @@ -272,7 +272,13 @@ complete benchmark run for the current candidate. `Benchmark Publication` runs trusted `main` scripts on hosted runners. Successful PR attempts are checked against GitHub workflow and commit metadata. All jobs -must succeed in one attempt. Use **Re-run all jobs** for publishable retries. +must have a successful latest result through the selected attempt. Use +**Re-run failed jobs** to retain successful configurations, or rerun an individual +job and its dependents. Each retry replaces its configuration's result artifact. +Collection combines the artifacts from that workflow run. +A failed retry blocks publication even if an earlier +attempt succeeded. Source revisions and benchmark definitions must match. +Use **Re-run all jobs** when required artifacts have expired. The executed Benchmark workflow must match the trusted workflow. A workflow change can require archival recovery after maintainer review and merge. diff --git a/docs/results.md b/docs/results.md index b1c393d..e38f540 100644 --- a/docs/results.md +++ b/docs/results.md @@ -9,6 +9,11 @@ parser when loading bundles. Store one immutable JSON bundle per workflow run and attempt. The identity is `run.id` plus `run.attempt`. A retry has its own bundle. Commit SHAs may repeat. +`run.attempt` identifies the collection attempt. Partial retries retain earlier +successful measurements from that workflow run. Source revisions and benchmark +definitions must match across all shards. Each configuration has a stable artifact +name. A retry replaces that artifact. Collection rejects duplicate configurations. + | Field | Contents | | --- | --- | | `schemaVersion` | Storage format version, currently `1` | @@ -20,7 +25,7 @@ Store one immutable JSON bundle per workflow run and attempt. The identity is | `catalog.platforms` | Stable platform IDs and display labels | | `catalog.metrics` | Metric IDs, units, direction and measurement method versions | | `benchmark` | Workload ID, version, settings, required metrics and expected matrix | -| `runners` | Runner records scoped to this workflow attempt | +| `runners` | Runner records for the selected measurements | | `measurements` | Results linked to a runner and runtime | Runtime and metric IDs are extensible strings. Lifecycle IDs are `reload`, diff --git a/scripts/benchmark.ts b/scripts/benchmark.ts index 877bb87..0eb178a 100644 --- a/scripts/benchmark.ts +++ b/scripts/benchmark.ts @@ -1,9 +1,9 @@ import { spawn, type ChildProcess } from 'node:child_process' -import { createWriteStream, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createWriteStream, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { resolve } from 'node:path' import { createInterface } from 'node:readline' import { fileURLToPath } from 'node:url' -import { parseArgs } from 'node:util' +import { inspect, parseArgs } from 'node:util' import { z } from 'zod' import { benchmark, catalog, runnerPools, runtimes, serverFlavor } from '../shared/catalog.ts' import { parseRunBundle, strategySchema, type RunBundle } from '../shared/results.ts' @@ -26,6 +26,11 @@ mkdirSync(directory, { recursive: true }) const local = values['local-bundle'] ? parseRunBundle(JSON.parse(readFileSync(values['local-bundle'], 'utf8'))) : undefined if (local && local.source !== 'local') throw new Error('Local collection requires a local bundle') const run = local?.run ?? JSON.parse(readFileSync(values.run, 'utf8')) as RunBundle['run'] +if (!local && process.env.GITHUB_ACTIONS) { + if (run.id !== process.env.GITHUB_RUN_ID || run.commit.sha !== process.env.GITHUB_SHA) throw new Error('Run identity differs from the current workflow') + run.attempt = z.coerce.number().int().positive().parse(process.env.GITHUB_RUN_ATTEMPT) + run.workflow.url = `https://github.com/${run.pullRequest!.repository}/actions/runs/${run.id}/attempts/${run.attempt}` +} const executable = resolve('artifacts/bin', serverFlavor(runtime.id), 'http-bench') const oha = 'oha' const runner = await runnerMetadata(platform, !!local) @@ -45,7 +50,7 @@ async function bounded(promise: Promise, milliseconds: number, mes let timer: ReturnType try { return await Promise.race([promise, new Promise((_, reject) => { - timer = setTimeout(() => reject(new Error(message)), milliseconds) + timer = setTimeout(() => reject(new Error(`${message} after ${milliseconds}ms`)), milliseconds) })]) } finally { clearTimeout(timer!) @@ -55,7 +60,10 @@ async function bounded(promise: Promise, milliseconds: number, mes function closed(child: ChildProcess) { return new Promise((resolve, reject) => { child.once('error', reject) - child.once('close', code => resolve(code)) + child.once('close', (code, signal) => { + if (signal) reject(new Error(`${child.spawnfile} terminated by ${signal}`)) + else resolve(code) + }) }) } @@ -114,31 +122,37 @@ try { const exitCode = await bounded(Promise.race([loadClosed, unexpectedExit, aborted]), ((settings.requestCount ? settings.clientTimeoutSeconds : settings.durationSeconds) + 60) * 1000, 'Load generator timed out') if (exitCode !== 0) throw new Error(`oha exited with ${exitCode}`) server.stdin!.end('shutdown\n') - if (await bounded(serverClosed, 10000, 'Server shutdown timed out') !== 0) throw new Error('Server shutdown failed') + const shutdownCode = await bounded(serverClosed, 10000, 'Server shutdown timed out') + if (shutdownCode !== 0) throw new Error(`Server shutdown failed with exit code ${shutdownCode}`) const performance = JSON.parse(readFileSync(performancePath, 'utf8')) - const memory = JSON.parse(readFileSync(memoryPath, 'utf8')) - rawOutputs.push({ tool: 'oha', format: 'json', content: performance }, { tool: 'sysinfo', format: 'json', content: memory }) + rawOutputs.push({ tool: 'oha', format: 'json', content: performance }) + const requestDetails = `HTTP statuses: ${JSON.stringify(performance?.statusCodeDistribution)}. Request errors: ${JSON.stringify(performance?.errorDistribution)}. Client timeout: ${settings.clientTimeoutSeconds}s. Concurrency: ${settings.concurrency}.` const nonnegative = z.number().finite().nonnegative() - const report = z.object({ + const parsedReport = z.object({ summary: z.object({ requestsPerSec: nonnegative.positive() }), latencyPercentiles: z.object({ p50: nonnegative, p95: nonnegative, 'p99.99': nonnegative }), statusCodeDistribution: z.record(z.string(), nonnegative.int()), errorDistribution: z.record(z.string(), nonnegative.int()), - }).parse(performance) + }).safeParse(performance) + if (!parsedReport.success) throw new Error(`Invalid oha report. ${requestDetails}`, { cause: parsedReport.error }) + const report = parsedReport.data + if (!report.statusCodeDistribution['200'] || Object.entries(report.statusCodeDistribution).some(([status, count]) => status !== '200' && count > 0)) throw new Error(`Unexpected HTTP status distribution. ${requestDetails}`) + if (Object.entries(report.errorDistribution).some(([error, count]) => count > 0 && error !== 'aborted due to deadline')) throw new Error(`Request errors occurred. ${requestDetails}`) + if ((report.errorDistribution['aborted due to deadline'] ?? 0) > settings.concurrency) throw new Error(`Deadline cancellations exceed concurrency. ${requestDetails}`) + const memory = JSON.parse(readFileSync(memoryPath, 'utf8')) + rawOutputs.push({ tool: 'sysinfo', format: 'json', content: memory }) const peak = z.array(z.object({ name: z.literal('Peak Resident Memory Usage'), unit: z.literal('bytes'), value: nonnegative.positive() })).length(1).parse(memory)[0]! - if (!report.statusCodeDistribution['200'] || Object.entries(report.statusCodeDistribution).some(([status, count]) => status !== '200' && count > 0)) throw new Error('Unexpected HTTP status distribution') - if (Object.entries(report.errorDistribution).some(([error, count]) => count > 0 && error !== 'aborted due to deadline')) throw new Error('Request errors occurred') - if ((report.errorDistribution['aborted due to deadline'] ?? 0) > settings.concurrency) throw new Error('Unexpected deadline cancellation count') bundle.measurements.push({ id: configuration, runnerId: runner.id, runtimeId: runtime.id, strategy, status: 'success', rawOutputs, values: { rps: report.summary.requestsPerSec, memory: peak.value / 1024 ** 2, p50: report.latencyPercentiles.p50 * 1000, p95: report.latencyPercentiles.p95 * 1000, p9999: report.latencyPercentiles['p99.99'] * 1000 }, }) } catch (error) { console.error(`Benchmark failed: ${configuration}`, error) + const reason = error instanceof Error && error.cause === undefined ? String(error) : inspect(error, { depth: 5, colors: false }) if (process.env.GITHUB_ACTIONS) { - const message = String(error).replaceAll('%', '%25').replaceAll('\r', '%0D').replaceAll('\n', '%0A') + const message = reason.replaceAll('%', '%25').replaceAll('\r', '%0D').replaceAll('\n', '%0A') console.error(`::error::${configuration}: ${message}`) } - bundle.measurements.push({ id: configuration, runnerId: runner.id, runtimeId: runtime.id, strategy, status: 'failed', reason: String(error), rawOutputs }) + bundle.measurements.push({ id: configuration, runnerId: runner.id, runtimeId: runtime.id, strategy, status: 'failed', reason, rawOutputs }) process.exitCode = 1 } finally { if (load && load.exitCode === null && load.signalCode === null) load.kill() @@ -159,5 +173,18 @@ try { log.end() process.removeListener('SIGINT', interrupt) process.removeListener('SIGTERM', interrupt) + for (const [tool, path] of [['oha', performancePath], ['sysinfo', memoryPath]] as const) { + if (rawOutputs.some(output => output.tool === tool) || !existsSync(path)) continue + try { + const content = readFileSync(path, 'utf8') + try { + rawOutputs.push({ tool, format: 'json', content: JSON.parse(content) }) + } catch { + rawOutputs.push({ tool, format: 'text', content }) + } + } catch (error) { + console.error(`Failed to retain ${tool} output at ${path}`, error) + } + } writeFileSync(resolve(directory, 'bundle.json'), `${JSON.stringify(parseRunBundle(bundle), null, 2)}\n`) } \ No newline at end of file diff --git a/scripts/build-harness.ts b/scripts/build-harness.ts index 8a125a7..75cce8d 100644 --- a/scripts/build-harness.ts +++ b/scripts/build-harness.ts @@ -1,5 +1,5 @@ import { spawnSync } from 'node:child_process' -import { copyFileSync, mkdirSync, readFileSync, rmSync, statSync, writeFileSync } from 'node:fs' +import { copyFileSync, existsSync, mkdirSync, readFileSync, rmSync, statSync, writeFileSync } from 'node:fs' import { dirname, resolve } from 'node:path' import { fileURLToPath } from 'node:url' @@ -78,6 +78,9 @@ if (['dummy', 'inputs', 'all'].includes(phase)) { if (['servers', 'all'].includes(phase)) { const executable = 'http-bench' + if (!existsSync('target/hyperlight-js-runtime/x86_64-hyperlight-none/release/hyperlight-js-runtime')) { + run('cargo', ['clean', '-p', 'hyperlight-js', '--release']) + } for (const serverFlavor of flavor === 'all' ? ['native', 'pulley'] : [flavor]) { const features = [serverFlavor === 'pulley' ? 'pulley' : '', phase === 'servers' ? process.argv[4] ?? '' : ''].filter(Boolean).join(',') run('cargo', ['clean', '-p', 'hyperlight-wasm', '--release']) diff --git a/scripts/merge-results.ts b/scripts/merge-results.ts index 3e9cdfe..de32145 100644 --- a/scripts/merge-results.ts +++ b/scripts/merge-results.ts @@ -1,16 +1,42 @@ import { readFileSync, writeFileSync } from 'node:fs' +import { resolve } from 'node:path' +import { pathToFileURL } from 'node:url' import { parseArgs } from 'node:util' -import { canonicalJson, parseRunBundle, validatePublishableRun } from '../shared/results.ts' +import { canonicalJson, parseRunBundle, validatePublishableRun, type RunBundle } from '../shared/results.ts' import { publicationPolicy } from '../shared/catalog.ts' -const { values, positionals } = parseArgs({ allowPositionals: true, options: { output: { type: 'string', default: 'run.json' } } }) -if (!positionals.length) throw new Error('Provide bundle files to merge') -const bundles = positionals.map(path => parseRunBundle(JSON.parse(readFileSync(path, 'utf8')))) -const merged = structuredClone(bundles[0]!) -for (const bundle of bundles.slice(1)) { - if (canonicalJson({ run: bundle.run, catalog: bundle.catalog, benchmark: bundle.benchmark, source: bundle.source }) !== canonicalJson({ run: merged.run, catalog: merged.catalog, benchmark: merged.benchmark, source: merged.source })) throw new Error('Shard definitions differ') - merged.runners.push(...bundle.runners) - merged.measurements.push(...bundle.measurements) +export function mergeBundles(inputs: unknown[], attempt?: number, runId?: string): RunBundle { + if (!inputs.length) throw new Error('Provide bundle files to merge') + const bundles = inputs.map(parseRunBundle) + const merged = structuredClone(bundles[0]!) + const collectionAttempt = attempt ?? Math.max(...bundles.map(bundle => bundle.run.attempt)) + if (!Number.isSafeInteger(collectionAttempt) || collectionAttempt < 1) throw new Error('Invalid collection attempt') + if (runId !== undefined && merged.run.id !== runId) throw new Error('Run identity differs from the current workflow') + const definition = (bundle: RunBundle) => { + const { attempt, createdAt, workflow, ...run } = bundle.run + const { url, ...workflowDefinition } = workflow + return canonicalJson({ run, workflow: workflowDefinition, catalog: bundle.catalog, benchmark: bundle.benchmark, source: bundle.source }) + } + const expected = definition(merged) + merged.runners = [] + merged.measurements = [] + for (const bundle of bundles) { + if (definition(bundle) !== expected) throw new Error('Shard definitions differ') + if (bundle.run.attempt > collectionAttempt) throw new Error('Shard attempt exceeds collection attempt') + if (bundle.source === 'published' && bundle.run.workflow.url !== `https://github.com/${bundle.run.pullRequest?.repository}/actions/runs/${bundle.run.id}/attempts/${bundle.run.attempt}`) throw new Error('Shard workflow URL differs') + if (bundle.run.createdAt < merged.run.createdAt) merged.run.createdAt = bundle.run.createdAt + merged.runners.push(...bundle.runners) + merged.measurements.push(...bundle.measurements) + } + merged.run.attempt = collectionAttempt + if (merged.source === 'published') merged.run.workflow.url = `https://github.com/${merged.run.pullRequest!.repository}/actions/runs/${merged.run.id}/attempts/${collectionAttempt}` + return validatePublishableRun(merged, publicationPolicy) } -writeFileSync(values.output, `${JSON.stringify(validatePublishableRun(merged, publicationPolicy), null, 2)}\n`, { flag: 'wx' }) -console.log(`Merged ${merged.measurements.length} successful measurements`) \ No newline at end of file + +if (process.argv[1] && import.meta.url === pathToFileURL(resolve(process.argv[1])).href) { + const { values, positionals } = parseArgs({ allowPositionals: true, options: { output: { type: 'string', default: 'run.json' } } }) + const attempt = process.env.GITHUB_ACTIONS ? Number(process.env.GITHUB_RUN_ATTEMPT) : undefined + const merged = mergeBundles(positionals.map(path => JSON.parse(readFileSync(path, 'utf8'))), attempt, process.env.GITHUB_ACTIONS ? process.env.GITHUB_RUN_ID : undefined) + writeFileSync(values.output, `${JSON.stringify(merged, null, 2)}\n`, { flag: 'wx' }) + console.log(`Merged ${merged.measurements.length} successful measurements`) +} \ No newline at end of file diff --git a/scripts/publication.test.ts b/scripts/publication.test.ts index 449ccc8..65ec43e 100644 --- a/scripts/publication.test.ts +++ b/scripts/publication.test.ts @@ -13,10 +13,59 @@ import { readJson, writeJson, withStoreLock } from './result-store.ts' import { assembleSite } from './assemble-site.ts' import { loadHistory } from '../src/data.ts' import { promoteRun, storeRun } from './publish-results.ts' +import { mergeBundles } from './merge-results.ts' const repository = 'example/benchmarks' const root = resolve(import.meta.dirname, '..') +test('partial retries retain successful measurements from the same source revision', () => { + const bundle = fixture() + const shards = bundle.measurements.map(measurement => { + const runner = structuredClone(bundle.runners.find(entry => entry.id === measurement.runnerId)!) + runner.id = measurement.id + return { ...structuredClone(bundle), runners: [runner], measurements: [{ ...structuredClone(measurement), runnerId: runner.id }] } + }) + const retry = structuredClone(shards[0]!) + retry.run.attempt = 2 + retry.run.workflow.url = `https://github.com/${repository}/actions/runs/${retry.run.id}/attempts/2` + retry.runners[0]!.name = 'retry-runner' + const measurement = shards[0]!.measurements[0]! + shards[0]!.measurements = [{ id: measurement.id, runnerId: measurement.runnerId, runtimeId: measurement.runtimeId, strategy: measurement.strategy, status: 'failed', reason: 'Transient failure', rawOutputs: [] }] + const retained = shards.slice(1) + const merged = mergeBundles([retry, ...retained], 2, bundle.run.id) + assert.equal(merged.run.attempt, 2) + assert.equal(merged.run.workflow.url, retry.run.workflow.url) + assert.equal(merged.measurements.length, 108) + assert.deepEqual(merged.measurements[0], retry.measurements[0]) + assert.deepEqual(merged.measurements[1], shards[1]!.measurements[0]) + assert.equal(merged.runners.find(runner => runner.id === retry.runners[0]!.id)!.name, 'retry-runner') + assert.throws(() => mergeBundles([retry, ...retained, structuredClone(retry)], 2), /Duplicate/) + assert.equal(mergeBundles([retry, ...retained], 3).run.attempt, 3) + + const failedRetry = structuredClone(retry) + failedRetry.measurements = structuredClone(shards[0]!.measurements) + assert.throws(() => mergeBundles([failedRetry, ...retained], 2), /Missing successful configurations|Only successful/) + assert.throws(() => mergeBundles(shards, 2), /Missing successful configurations|Only successful/) + assert.throws(() => mergeBundles([...shards.slice(1, -1), retry], 2), /Missing successful configurations/) + assert.throws(() => mergeBundles([...shards, retry], 2), /Duplicate/) + assert.throws(() => mergeBundles([retry, ...retained], 1), /exceeds collection attempt/) + assert.throws(() => mergeBundles([retry, ...retained], 2, 'another-run'), /Run identity differs/) + for (const mutate of [ + (input: typeof retry) => { input.run.id = 'another-run' }, + (input: typeof retry) => { input.run.commit.sha = 'f'.repeat(40) }, + (input: typeof retry) => { input.run.commit.tree = 'f'.repeat(40) }, + (input: typeof retry) => { input.run.pullRequest!.base = 'f'.repeat(40) }, + (input: typeof retry) => { input.benchmark.settings.concurrency = 1 }, + ]) { + const invalid = structuredClone(retry) + mutate(invalid) + assert.throws(() => mergeBundles([...retained, invalid], 2), /Shard definitions differ/) + } + const invalidUrl = structuredClone(retry) + invalidUrl.run.workflow.url = bundle.run.workflow.url + assert.throws(() => mergeBundles([...retained, invalidUrl], 2), /Shard workflow URL differs/) +}) + test('local history loads measured results but cannot be published', async context => { const bundle = fixture() bundle.source = 'local' @@ -327,13 +376,14 @@ test('CI archival and promotion with simulated GitHub and Git', async context => ...Array.from({ length: 2 }, (_, index) => `prepare (${index})`), ...Array.from({ length: 108 }, (_, index) => `measure (${index})`), ].map(name => ({ name, conclusion: 'success' })) + let retryJobs: typeof jobs = [] context.mock.method(globalThis, 'fetch', async (url: string) => { const parsed = new URL(url) const path = parsed.pathname.replace(`/repos/${repository}/`, '') if (path.endsWith('/jobs')) { const page = Number(parsed.searchParams.get('page')) - const response = structuredClone(jobs) - if (failedJob) response[0]!.conclusion = 'failure' + const response = structuredClone(path.includes('/attempts/2/') ? retryJobs : jobs) + if (failedJob && !path.includes('/attempts/2/')) response[0]!.conclusion = 'failure' return Response.json({ jobs: response.slice((page - 1) * 100, page * 100), total_count: response.length }) } const responses: Record = { @@ -344,6 +394,7 @@ test('CI archival and promotion with simulated GitHub and Git', async context => 'issues/7/events': [{ event: 'labeled', label: { name: 'benchmarks: skip' }, actor: { login: 'maintainer' } }], 'collaborators/maintainer/permission': { permission: 'write' }, 'actions/runs/12345/attempts/1': { id: 12345, event: 'pull_request', conclusion: 'success', path: '.github/workflows/benchmark.yml', head_sha: workflowHead, created_at: bundle.run.createdAt }, + 'actions/runs/12345/attempts/2': { id: 12345, event: 'pull_request', conclusion: 'success', path: '.github/workflows/benchmark.yml', head_sha: workflowHead, created_at: bundle.run.createdAt }, 'contents/.github/workflows/benchmark.yml': { encoding: 'base64', content: Buffer.from(workflow).toString('base64') }, [`git/commits/${bundle.run.commit.sha}`]: { tree: { sha: bundle.run.commit.tree }, parents: [{ sha: bundle.run.pullRequest!.base }, { sha: bundle.run.pullRequest!.head }] }, [`git/commits/${pr.merge_commit_sha}`]: { tree: { sha: mergedTree } }, @@ -424,6 +475,36 @@ test('CI archival and promotion with simulated GitHub and Git', async context => assert.equal(downloads, 0) assert.equal(pushes, 0) }) + await context.test('partial retry archives and promotes retained successful jobs', async () => { + clearStore() + const originalUrl = bundle.run.workflow.url + bundle.run.attempt = 2 + bundle.run.workflow.url = `https://github.com/${repository}/actions/runs/12345/attempts/2` + retryJobs = jobs.filter(job => ['eligibility', 'measure (0)', 'collect', 'Benchmark Status'].includes(job.name)) + failedJob = true + writeJson(eventPath, { workflow_run: { ...notification.workflow_run, run_attempt: 2 } }, false) + try { + await main() + assert.equal(downloads, 1) + assert.equal(pushes, 2) + assert.deepEqual(readJson(resolve(temporary, 'data-store/index.json')), { schemaVersion: 1, runs: [{ id: '12345', attempt: 2 }] }) + const stored = validateCompleteRun(readJson(resolve(temporary, 'data-store/runs/12345/2.json'))) + assert.deepEqual(stored.measurements, bundle.measurements) + retryJobs = retryJobs.map(job => ({ ...job, conclusion: job.name === 'measure (0)' ? 'failure' : 'success' })) + await assert.rejects(main(), /successful measure/) + assert.equal(downloads, 1, 'A failed retry must not use an earlier successful job') + retryJobs = retryJobs.filter(job => job.name !== 'eligibility') + await assert.rejects(main(), /successful eligibility/) + assert.equal(downloads, 1) + } finally { + bundle.run.attempt = 1 + bundle.run.workflow.url = originalUrl + retryJobs = [] + failedJob = false + writeJson(eventPath, notification, false) + clearStore() + } + }) await context.test('label changes prevent archival and promotion', async () => { pr.labels = [{ name: 'benchmarks: skip' }] await main() diff --git a/scripts/publish-ci.ts b/scripts/publish-ci.ts index 8188cc0..bafaeb4 100644 --- a/scripts/publish-ci.ts +++ b/scripts/publish-ci.ts @@ -48,13 +48,20 @@ async function verifiedRun(runId: number, attempt: number) { if (workflow.encoding !== 'base64' || Buffer.from(workflow.content, 'base64').toString('utf8') !== readFileSync('.github/workflows/benchmark.yml', 'utf8')) { throw new Error('Benchmark workflow differs from the trusted workflow. Maintainer review and a fresh run are required.') } - const jobs: any[] = [] - for (let page = 1; ; page++) { - const response = await github(`actions/runs/${runId}/attempts/${attempt}/jobs?per_page=100&page=${page}`) - jobs.push(...response.jobs) - if (jobs.length >= response.total_count) break - if (!response.jobs.length) throw new Error('Incomplete workflow job list') + const latestJobs = new Map() + for (let currentAttempt = attempt; currentAttempt >= 1; currentAttempt--) { + let fetched = 0 + for (let page = 1; ; page++) { + const response = await github(`actions/runs/${runId}/attempts/${currentAttempt}/jobs?per_page=100&page=${page}`) + for (const job of response.jobs) { + if (!latestJobs.has(job.name)) latestJobs.set(job.name, job) + } + fetched += response.jobs.length + if (fetched >= response.total_count) break + if (!response.jobs.length) throw new Error('Incomplete workflow job list') + } } + const jobs = [...latestJobs.values()] const expected = new Map([ ['eligibility', 1], ['configure', 1], ['producer', 1], ['prepare', 2], ['measure', publicationPolicy.benchmark.expectedConfigurations.length], ['collect', 1], ['Benchmark Status', 1], @@ -62,7 +69,7 @@ async function verifiedRun(runId: number, attempt: number) { for (const [name, count] of expected) { const matching = jobs.filter(job => job.name === name || job.name.startsWith(`${name} (`)) if (matching.length !== count || matching.some(job => job.conclusion !== 'success')) { - throw new Error(`Attempt ${runId}.${attempt} requires ${count} successful ${name} jobs. Rerun all jobs.`) + throw new Error(`Run ${runId} through attempt ${attempt} requires ${count} successful ${name} jobs. Rerun the failed jobs.`) } } return run diff --git a/shared/catalog.ts b/shared/catalog.ts index 77ef0ac..9ad9189 100644 --- a/shared/catalog.ts +++ b/shared/catalog.ts @@ -50,7 +50,7 @@ export const benchmark = { id: 'http-redirect', version: 2, settings: { durationSeconds: 60, concurrency: 50, poolSize: 4, workerThreads: 'available-parallelism', workerReadiness: 'all-workers', - sandboxTimeoutMs: 1000, timeoutCheckIntervalMs: 10, clientTimeoutSeconds: 10, + sandboxTimeoutMs: 1000, timeoutCheckIntervalMs: 10, clientTimeoutSeconds: 30, memorySampleIntervalMs: 500, memorySampleAtShutdown: true, readinessRequests: 1, warmupRequests: 0, waitOngoingRequests: false, repetitions: 1, hyperlightAotMapping: 'linux-file-mapping-per-worker', }, From e6d27d02874a693f7ca8eecf4b8ff84c5f3f43f4 Mon Sep 17 00:00:00 2001 From: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> Date: Tue, 15 Sep 2026 15:39:19 -0700 Subject: [PATCH 4/4] Restore timed benchmark behavior and accept newer wasm-tools Signed-off-by: Ludvig Liljenberg <4257730+ludfjig@users.noreply.github.com> --- .github/actions/setup/action.yml | 13 ++++++------- docs/README.md | 4 +++- scripts/benchmark-local.ts | 2 +- scripts/benchmark.ts | 10 ++++++---- shared/catalog.ts | 2 +- 5 files changed, 17 insertions(+), 14 deletions(-) diff --git a/.github/actions/setup/action.yml b/.github/actions/setup/action.yml index 07da04a..a9681d3 100644 --- a/.github/actions/setup/action.yml +++ b/.github/actions/setup/action.yml @@ -112,7 +112,7 @@ runs: echo "✅ Cargo is available: $(cargo --version)" # Required tool versions (crates.io) - WASM_TOOLS_REQUIRED_VERSION="1.243.0" + WASM_TOOLS_MIN_VERSION="1.259.0" CARGO_HYPERLIGHT_REQUIRED_VERSION="0.1.14" OHA_REQUIRED_VERSION="1.9.0" @@ -127,19 +127,18 @@ runs: COMPONENTIZE_QJS_REQUIRED_VERSION="0.3.0" if [[ "${{ inputs.BENCH_ONLY }}" != "true" ]]; then - # Install wasm-tools if not cached or wrong version if command -v wasm-tools &> /dev/null; then WASM_TOOLS_CURRENT_VERSION=$(wasm-tools --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1) WASM_TOOLS_CURRENT_VERSION=${WASM_TOOLS_CURRENT_VERSION:-"unknown"} - if [ "$WASM_TOOLS_CURRENT_VERSION" = "$WASM_TOOLS_REQUIRED_VERSION" ]; then + if [[ "$WASM_TOOLS_CURRENT_VERSION" != "unknown" ]] && printf '%s\n' "$WASM_TOOLS_MIN_VERSION" "$WASM_TOOLS_CURRENT_VERSION" | sort -V -C; then echo "wasm-tools $WASM_TOOLS_CURRENT_VERSION already available (cached)" else - echo "wasm-tools version mismatch: found $WASM_TOOLS_CURRENT_VERSION, need $WASM_TOOLS_REQUIRED_VERSION. Reinstalling..." - cargo install wasm-tools@$WASM_TOOLS_REQUIRED_VERSION --locked --force + echo "wasm-tools $WASM_TOOLS_CURRENT_VERSION does not meet minimum $WASM_TOOLS_MIN_VERSION. Installing..." + cargo install wasm-tools@$WASM_TOOLS_MIN_VERSION --locked --force fi else - echo "Installing wasm-tools $WASM_TOOLS_REQUIRED_VERSION..." - cargo install wasm-tools@$WASM_TOOLS_REQUIRED_VERSION --locked + echo "Installing wasm-tools $WASM_TOOLS_MIN_VERSION..." + cargo install wasm-tools@$WASM_TOOLS_MIN_VERSION --locked fi # Install hyperlight-wasm-aot if not cached or wrong version. diff --git a/docs/README.md b/docs/README.md index e5a3037..ddb93f3 100644 --- a/docs/README.md +++ b/docs/README.md @@ -40,7 +40,7 @@ summaries are not loaded by this interface. ### Local End To End Requires Linux x86_64, Node.js 24+, Rust, just, and read/write access to -`/dev/kvm` or `/dev/mshv`. Install wasm-tools 1.243.0, +`/dev/kvm` or `/dev/mshv`. Install wasm-tools 1.259.0 or newer, hyperlight-wasm-aot 0.15.0, cargo-hyperlight 0.1.14, componentize-qjs-cli 0.3.0 with `--no-default-features`, and oha 1.9.0. Use `cargo install @ --locked` for each tool. @@ -74,6 +74,8 @@ not performance stability. The bundle records the request count and load setting JIT-heavy configurations can still take several minutes across the full matrix. Timed runs accept `--duration` and `--client-timeout` in seconds. +The default client timeout is unlimited. Requests still in flight are cancelled +when the load duration ends. `--client-timeout` enables a per-request limit. `--concurrency` sets the connection count. For slower local hosts, use: ```sh diff --git a/scripts/benchmark-local.ts b/scripts/benchmark-local.ts index a0f4a8c..e732f36 100644 --- a/scripts/benchmark-local.ts +++ b/scripts/benchmark-local.ts @@ -37,7 +37,7 @@ async function main() { ...benchmark.settings, durationSeconds: values.smoke ? 0 : positive(values.duration, benchmark.settings.durationSeconds, 'Duration'), concurrency: values.smoke ? 1 : positive(values.concurrency, benchmark.settings.concurrency, 'Concurrency'), - clientTimeoutSeconds: values.smoke ? 120 : positive(values['client-timeout'], benchmark.settings.clientTimeoutSeconds, 'Client timeout'), + clientTimeoutSeconds: values.smoke ? 120 : values['client-timeout'] === undefined ? benchmark.settings.clientTimeoutSeconds : positive(values['client-timeout'], 120, 'Client timeout'), ...(values.smoke ? { requestCount: 1 } : {}), } if (selected.some(runtime => runtime.id.startsWith('hyperlight-'))) { diff --git a/scripts/benchmark.ts b/scripts/benchmark.ts index 0eb178a..ec16144 100644 --- a/scripts/benchmark.ts +++ b/scripts/benchmark.ts @@ -77,7 +77,7 @@ try { ...z.object({ durationSeconds: z.number().int().nonnegative(), concurrency: z.number().int().positive(), - clientTimeoutSeconds: z.number().int().positive(), + clientTimeoutSeconds: z.number().int().positive().nullable(), requestCount: z.literal(1).optional(), }).parse(bundle.benchmark.settings), } @@ -116,17 +116,19 @@ try { if (response.status !== 200) throw new Error(`Readiness request returned ${response.status}`) z.object({ uri: z.literal('/redirected.html') }).parse(await response.json()) const limit = settings.requestCount ? ['-n', String(settings.requestCount)] : ['-z', `${settings.durationSeconds}s`] - load = spawn(oha, [url, ...limit, '-c', String(settings.concurrency), '-t', `${settings.clientTimeoutSeconds}s`, '--no-tui', '--output-format', 'json', '--output', performancePath], { stdio: ['ignore', 'inherit', 'inherit'] }) + if (settings.requestCount && settings.clientTimeoutSeconds === null) throw new Error('Request-count runs require a client timeout') + const clientTimeout = settings.clientTimeoutSeconds === null ? [] : ['-t', `${settings.clientTimeoutSeconds}s`] + load = spawn(oha, [url, ...limit, '-c', String(settings.concurrency), ...clientTimeout, '--no-tui', '--output-format', 'json', '--output', performancePath], { stdio: ['ignore', 'inherit', 'inherit'] }) loadClosed = closed(load) loadClosed.catch(() => {}) - const exitCode = await bounded(Promise.race([loadClosed, unexpectedExit, aborted]), ((settings.requestCount ? settings.clientTimeoutSeconds : settings.durationSeconds) + 60) * 1000, 'Load generator timed out') + const exitCode = await bounded(Promise.race([loadClosed, unexpectedExit, aborted]), ((settings.requestCount ? settings.clientTimeoutSeconds! : settings.durationSeconds) + 60) * 1000, 'Load generator timed out') if (exitCode !== 0) throw new Error(`oha exited with ${exitCode}`) server.stdin!.end('shutdown\n') const shutdownCode = await bounded(serverClosed, 10000, 'Server shutdown timed out') if (shutdownCode !== 0) throw new Error(`Server shutdown failed with exit code ${shutdownCode}`) const performance = JSON.parse(readFileSync(performancePath, 'utf8')) rawOutputs.push({ tool: 'oha', format: 'json', content: performance }) - const requestDetails = `HTTP statuses: ${JSON.stringify(performance?.statusCodeDistribution)}. Request errors: ${JSON.stringify(performance?.errorDistribution)}. Client timeout: ${settings.clientTimeoutSeconds}s. Concurrency: ${settings.concurrency}.` + const requestDetails = `HTTP statuses: ${JSON.stringify(performance?.statusCodeDistribution)}. Request errors: ${JSON.stringify(performance?.errorDistribution)}. Client timeout: ${settings.clientTimeoutSeconds === null ? 'unlimited' : `${settings.clientTimeoutSeconds}s`}. Concurrency: ${settings.concurrency}.` const nonnegative = z.number().finite().nonnegative() const parsedReport = z.object({ summary: z.object({ requestsPerSec: nonnegative.positive() }), diff --git a/shared/catalog.ts b/shared/catalog.ts index 9ad9189..32025d5 100644 --- a/shared/catalog.ts +++ b/shared/catalog.ts @@ -50,7 +50,7 @@ export const benchmark = { id: 'http-redirect', version: 2, settings: { durationSeconds: 60, concurrency: 50, poolSize: 4, workerThreads: 'available-parallelism', workerReadiness: 'all-workers', - sandboxTimeoutMs: 1000, timeoutCheckIntervalMs: 10, clientTimeoutSeconds: 30, + sandboxTimeoutMs: 1000, timeoutCheckIntervalMs: 10, clientTimeoutSeconds: null, memorySampleIntervalMs: 500, memorySampleAtShutdown: true, readinessRequests: 1, warmupRequests: 0, waitOngoingRequests: false, repetitions: 1, hyperlightAotMapping: 'linux-file-mapping-per-worker', },