From c2a7e2ce8362129b996605ced92f92c22cf8ab1d Mon Sep 17 00:00:00 2001 From: Sawyer Date: Tue, 29 Sep 2026 20:54:21 -0700 Subject: [PATCH] feat(web): bench-scoped Insights stat set, outcome chart, and coming-later row (CL-9553) --- apps/web/src/app.css | 206 ++++++++++++++++ apps/web/src/bench-insights-stats.test.ts | 46 ++++ apps/web/src/insights-stats.ts | 127 ++++++++++ apps/web/src/pages/bench-insights.tsx | 287 ++++++++++++++++++++++ apps/web/src/pages/insights-page.tsx | 18 +- 5 files changed, 678 insertions(+), 6 deletions(-) create mode 100644 apps/web/src/bench-insights-stats.test.ts create mode 100644 apps/web/src/pages/bench-insights.tsx diff --git a/apps/web/src/app.css b/apps/web/src/app.css index 7a8a245b68..026d04a683 100644 --- a/apps/web/src/app.css +++ b/apps/web/src/app.css @@ -3942,3 +3942,209 @@ tr.insights-row-clickable:hover { font-size: 0.95rem; font-weight: 600; } + +/* Bench Insights (pages/bench-insights.tsx) */ +.bi-seg { + display: inline-flex; + gap: 2px; + padding: 2px; + border-radius: 8px; + background: var(--muted); + align-self: flex-start; +} +.bi-seg button { + padding: 0.25rem 0.7rem; + font-size: 0.78rem; + font-weight: 600; + border-radius: 6px; + color: var(--muted-foreground); +} +.bi-seg button.active { + background: var(--card, var(--background)); + color: var(--foreground); +} +.bi-stats { + display: grid; + grid-template-columns: repeat(auto-fit, minmax(140px, 1fr)); + gap: 0.75rem; +} +.bi-stat { + padding: 1rem; + border: 1px solid var(--border); + border-radius: 8px; + background: var(--card, var(--background)); +} +.bi-stat-v { + font-size: 1.6rem; + font-weight: 700; + font-variant-numeric: tabular-nums; +} +.bi-stat-k { + font-size: 0.75rem; + color: var(--muted-foreground); +} +.bi-chart { + position: relative; +} +.bi-legend { + display: flex; + gap: 1rem; + font-size: 0.78rem; + margin-bottom: 0.5rem; +} +.bi-legend span { + display: inline-flex; + align-items: center; + gap: 0.35rem; +} +.bi-legend i { + width: 10px; + height: 10px; + border-radius: 3px; + display: inline-block; +} +.bi-bars { + display: flex; + align-items: flex-end; + height: 160px; + padding-top: 8px; + border-bottom: 1px solid var(--border); +} +.bi-col { + flex: 1; + height: 100%; + display: flex; + flex-direction: column-reverse; + align-items: center; + gap: 2px; +} +.bi-col i { + display: block; + width: 100%; + max-width: 28px; +} +.bi-col:hover i, +.bi-col:focus-visible i { + filter: brightness(1.12); +} +.bi-col:focus-visible { + outline: 2px solid var(--ring, currentColor); + outline-offset: 1px; +} +.bi-x { + display: flex; + margin-top: 6px; + font-size: 0.7rem; + color: var(--muted-foreground); + font-variant-numeric: tabular-nums; +} +.bi-x span { + flex: 1; + text-align: center; +} +.bi-tip { + position: absolute; + top: 20px; + transform: translateX(-50%); + pointer-events: none; + z-index: 5; + padding: 0.4rem 0.6rem; + font-size: 0.78rem; + white-space: nowrap; + background: var(--popover, var(--card)); + border: 1px solid var(--border); + border-radius: 6px; +} +.bi-tip b { + display: block; +} +.bi-scroll { + overflow-x: auto; +} +.bi-table { + width: 100%; + min-width: 560px; + border-collapse: collapse; + font-size: 0.85rem; +} +.bi-table th { + text-align: left; + font-size: 0.7rem; + text-transform: uppercase; + letter-spacing: 0.08em; + color: var(--muted-foreground); + padding: 0.4rem 0.5rem; +} +.bi-table td { + padding: 0.55rem 0.5rem; + border-top: 1px solid var(--border); + font-variant-numeric: tabular-nums; +} +.bi-meter { + display: inline-flex; + gap: 2px; + width: 72px; + height: 8px; + margin-right: 8px; + vertical-align: middle; + border-radius: 4px; + overflow: hidden; + background: var(--muted); +} +.bi-meter i { + display: block; + height: 100%; +} +.bi-fails { + list-style: none; + margin: 0; + padding: 0; +} +.bi-fails button { + display: flex; + align-items: center; + gap: 0.75rem; + width: 100%; + padding: 0.6rem 0; + text-align: left; + border-top: 1px solid var(--border); +} +.bi-fails li:first-child button { + border-top: 0; +} +.bi-fails span { + margin-left: auto; + font-size: 0.78rem; + color: var(--muted-foreground); +} +.bi-later-h { + margin: 0.5rem 0 0.25rem; + font-size: 0.85rem; + font-weight: 700; +} +.bi-soon { + display: grid; + grid-template-columns: repeat(3, minmax(0, 1fr)); + gap: 0.75rem; + margin-top: 0.75rem; +} +.bi-soon > div { + padding: 1rem; + border: 1px solid var(--border); + border-radius: 8px; + background: var(--muted); +} +.bi-soon b { + display: block; + font-size: 0.85rem; + margin-bottom: 0.25rem; +} +.bi-soon span { + font-size: 0.78rem; + color: var(--muted-foreground); +} +@media (max-width: 960px) { + .bi-soon { + grid-template-columns: 1fr; + } +} diff --git a/apps/web/src/bench-insights-stats.test.ts b/apps/web/src/bench-insights-stats.test.ts new file mode 100644 index 0000000000..8ed98c560b --- /dev/null +++ b/apps/web/src/bench-insights-stats.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, test } from "bun:test"; + +import type { InsightsRun } from "./insights-api"; +import { computeBenchInsights, medianMs } from "./insights-stats"; + +const NOW = Date.parse("2026-09-10T12:00:00"); + +function run(id: string, createdAt: string, status: string, endedAt?: string): InsightsRun { + return { + id, + definitionId: "def_a", + definitionName: "brief", + createdAt, + status, + ...(endedAt === undefined ? {} : { endedAt }), + } as unknown as InsightsRun; +} + +describe("medianMs", () => { + test("averages the middle pair of an even set", () => { + expect(medianMs([4, 1, 3, 2])).toBe(2.5); + expect(medianMs([])).toBeNull(); + }); +}); + +describe("computeBenchInsights", () => { + test("windows, buckets by outcome and day, and takes median duration", () => { + const stats = computeBenchInsights( + [ + run("a", "2026-09-10T09:00:00", "completed", "2026-09-10T09:01:00"), + run("b", "2026-09-10T10:00:00", "failed", "2026-09-10T10:03:00"), + run("c", "2026-09-09T10:00:00", "completed", "2026-09-09T10:02:00"), + run("old", "2026-08-01T10:00:00", "completed", "2026-08-01T10:02:00"), + ], + 7, + NOW, + ); + expect(stats.total).toBe(3); + expect(stats.ok).toBe(2); + expect(stats.fail).toBe(1); + expect(stats.medianMs).toBe(120_000); + expect(stats.days.at(-1)).toMatchObject({ ok: 1, fail: 1 }); + expect(stats.days.at(-2)).toMatchObject({ ok: 1, fail: 0 }); + expect(stats.failures.map((r) => r.id)).toEqual(["b"]); + }); +}); diff --git a/apps/web/src/insights-stats.ts b/apps/web/src/insights-stats.ts index 662f761749..8b7423e037 100644 --- a/apps/web/src/insights-stats.ts +++ b/apps/web/src/insights-stats.ts @@ -135,3 +135,130 @@ export function computeInsightsStats( recentRuns, }; } + +export const BENCH_RANGES = [7, 30, 90] as const; +export type BenchRange = (typeof BENCH_RANGES)[number]; + +const DAY_MS = 86_400_000; + +export type BenchDay = { + readonly date: Date; + readonly ok: number; + readonly fail: number; +}; +export type BenchWorkflowRow = { + readonly key: string; + readonly name: string; + readonly runs: number; + readonly ok: number; + readonly fail: number; + readonly medianMs: number | null; + readonly lastRun: string; +}; +export type BenchInsights = { + readonly total: number; + readonly ok: number; + readonly fail: number; + readonly medianMs: number | null; + readonly days: readonly BenchDay[]; + readonly workflows: readonly BenchWorkflowRow[]; + readonly failures: readonly InsightsRun[]; +}; + +export function medianMs(values: readonly number[]): number | null { + if (values.length === 0) return null; + const sorted = [...values].sort((a, b) => a - b); + const mid = Math.floor(sorted.length / 2); + const hi = sorted[mid] ?? 0; + return sorted.length % 2 === 1 ? hi : ((sorted[mid - 1] ?? 0) + hi) / 2; +} + +function runDurationMs(run: InsightsRun): number | null { + if (run.endedAt === undefined || run.endedAt === null) return null; + const ms = Date.parse(run.endedAt) - Date.parse(run.createdAt); + return Number.isNaN(ms) || ms < 0 ? null : ms; +} + +type Bucket = "ok" | "fail" | "other"; +function bucketOf(run: InsightsRun, now: number): Bucket { + const outcome = runOutcomeStatus(withListingAbandoned(run, now), now) ?? run.status; + if (outcome === "completed") return "ok"; + if (outcome === "failed" || outcome === "error") return "fail"; + return "other"; +} + +/** Rollups over the runs created in the last `range` local days (today + * included). "Other" outcomes (running, stopped) count as runs but are + * neither succeeded nor failed. */ +export function computeBenchInsights( + runs: readonly InsightsRun[], + range: BenchRange, + now: number = Date.now(), +): BenchInsights { + const today = new Date(now); + today.setHours(0, 0, 0, 0); + const days: { date: Date; ok: number; fail: number }[] = []; + for (let i = range - 1; i >= 0; i--) { + const date = new Date(today); + date.setDate(today.getDate() - i); + days.push({ date, ok: 0, fail: 0 }); + } + const start = days[0]?.date.getTime() ?? 0; + + const inRange = runs.filter((run) => { + const t = Date.parse(run.createdAt); + return !Number.isNaN(t) && t >= start && t <= now; + }); + let ok = 0; + let fail = 0; + const groups = new Map(); + for (const run of inRange) { + const bucket = bucketOf(run, now); + const midnight = new Date(run.createdAt); + midnight.setHours(0, 0, 0, 0); + // round, not floor: DST days are 23h or 25h long. + const day = days[Math.round((midnight.getTime() - start) / DAY_MS)]; + if (bucket === "ok") { + ok += 1; + if (day !== undefined) day.ok += 1; + } else if (bucket === "fail") { + fail += 1; + if (day !== undefined) day.fail += 1; + } + const key = run.routineId ?? run.definitionId; + groups.set(key, [...(groups.get(key) ?? []), run]); + } + + const durations = (rs: readonly InsightsRun[]) => + rs.flatMap((r) => { + const d = runDurationMs(r); + return d === null ? [] : [d]; + }); + const workflows = [...groups.entries()] + .map(([key, rs]) => { + const newest = [...rs].sort((a, b) => b.createdAt.localeCompare(a.createdAt)); + return { + key, + name: newest[0] !== undefined ? runDisplayName(newest[0]) : key, + runs: rs.length, + ok: rs.filter((r) => bucketOf(r, now) === "ok").length, + fail: rs.filter((r) => bucketOf(r, now) === "fail").length, + medianMs: medianMs(durations(rs)), + lastRun: newest[0]?.createdAt ?? "", + }; + }) + .sort((a, b) => b.lastRun.localeCompare(a.lastRun)); + + return { + total: inRange.length, + ok, + fail, + medianMs: medianMs(durations(inRange)), + days, + workflows, + failures: inRange + .filter((r) => bucketOf(r, now) === "fail") + .sort((a, b) => b.createdAt.localeCompare(a.createdAt)) + .slice(0, 5), + }; +} diff --git a/apps/web/src/pages/bench-insights.tsx b/apps/web/src/pages/bench-insights.tsx new file mode 100644 index 0000000000..e9385da68e --- /dev/null +++ b/apps/web/src/pages/bench-insights.tsx @@ -0,0 +1,287 @@ +// The bench-scoped Insights dashboard: every number is computed from the +// stock `GET /workflows/runs` listing of the bench's own tenant. + +import { Badge, RichEmptyState, Skeleton, RUN_STATUS_TONE } from "@corbits/react-ui"; +import { useState } from "react"; + +import { useAPIQuery } from "../api"; +import { insightsTopLevelRunsPath, TopLevelRunsSchema } from "../insights-api"; +import { + BENCH_RANGES, + computeBenchInsights, + durationLabel, + formatCount, + runDisplayName, + type BenchDay, + type BenchRange, +} from "../insights-stats"; +import { formatWhen } from "./insights-page"; + +const RANGE_LABEL: Readonly> = { + 7: "Last 7 days", + 30: "Last 30 days", + 90: "Last 90 days", +}; + +const COMING_LATER = [ + ["Tokens and cost", "Per worker, per workflow, per model"], + ["Tool calls", "Which tools ran, how often, and which failed"], + ["Latency", "Time to first reply and time per step"], +] as const; + +function dayLabel(date: Date): string { + return date.toLocaleDateString(undefined, { + weekday: "short", + month: "short", + day: "numeric", + }); +} + +function OutcomeChart({ days }: { readonly days: readonly BenchDay[] }) { + const [active, setActive] = useState(null); + const max = Math.max(...days.map((d) => d.ok + d.fail), 1); + const every = days.length > 30 ? 14 : days.length > 7 ? 5 : 1; + const shown = active === null ? undefined : days[active]; + return ( +
+
+ + + Succeeded + + + + Failed + +
+
30 ? 1 : 4 }}> + {days.map((d, i) => ( +
setActive(i)} + onMouseLeave={() => setActive(null)} + onFocus={() => setActive(i)} + onBlur={() => setActive(null)} + > + {d.ok > 0 ? ( + + ) : null} + {d.fail > 0 ? ( + + ) : null} +
+ ))} +
+ + {shown !== undefined && active !== null ? ( +
+ {dayLabel(shown.date)} + {shown.ok} succeeded · {shown.fail} failed +
+ ) : null} +
+ ); +} + +function Meter({ + ok, + fail, + total, +}: { + readonly ok: number; + readonly fail: number; + readonly total: number; +}) { + return ( + + ); +} + +export function BenchInsights({ + tenantId, + onOpenRun, +}: { + readonly tenantId: string; + readonly onOpenRun: (id: string) => void; +}) { + const [range, setRange] = useState(7); + const runs = useAPIQuery(insightsTopLevelRunsPath(tenantId), TopLevelRunsSchema); + + if (runs.kind === "loading") return ; + if (runs.kind !== "ready") { + return ( + + ); + } + + const stats = computeBenchInsights(runs.data.data, range); + const finished = stats.ok + stats.fail; + const tiles: readonly (readonly [string, string])[] = [ + ["Runs", formatCount(stats.total)], + ["Succeeded", finished === 0 ? "—" : `${Math.round((stats.ok / finished) * 100)}%`], + ["Median run", stats.medianMs === null ? "—" : durationLabel(stats.medianMs)], + ]; + + return ( +
+
+ {BENCH_RANGES.map((r) => ( + + ))} +
+ {runs.data.nextCursor !== null ? ( +

Figures reflect the 100 most recent runs — more exist.

+ ) : null} + +
+ {tiles.map(([label, value]) => ( +
+
{value}
+
{label}
+
+ ))} +
+ +
+

Runs per day

+

Every run in this workbench, by how it ended.

+ +
+ +
+

By workflow

+ {stats.workflows.length === 0 ? ( + + ) : ( +
+ + + + + + + + + + + + {stats.workflows.map((w) => { + const done = w.ok + w.fail; + return ( + + + + + + + + ); + })} + +
WorkflowRunsSucceededMedianLast run
+ {w.name} + {w.runs} + {done === 0 ? ( + "—" + ) : ( + <> + + {Math.round((w.ok / done) * 100)}% + + )} + {w.medianMs === null ? "—" : durationLabel(w.medianMs)}{formatWhen(w.lastRun)}
+
+ )} +
+ +
+

Recent failures

+ {stats.failures.length === 0 ? ( +

No failures in the last {range} days.

+ ) : ( +
    + {stats.failures.map((run) => ( +
  • + +
  • + ))} +
+ )} +
+ +
+

Coming later

+

These need usage data the platform doesn't report yet.

+
+ {COMING_LATER.map(([title, detail]) => ( +
+ {title} + {detail} +
+ ))} +
+
+
+ ); +} diff --git a/apps/web/src/pages/insights-page.tsx b/apps/web/src/pages/insights-page.tsx index 99479ff40d..341ab7baaa 100644 --- a/apps/web/src/pages/insights-page.tsx +++ b/apps/web/src/pages/insights-page.tsx @@ -30,6 +30,7 @@ import { SignedOutNotice, type APIQuery } from "@/lib/api-query"; import { workbenchesQueryKey, listWorkbenches } from "@/chat/workbench-tenants"; import { useBench } from "../bench-context"; +import { BenchInsights } from "./bench-insights"; import { resolveWorkbenchInsightsScope } from "../insights-workbench-scope"; import { parseInsightsPath } from "../insights-path"; import { @@ -682,7 +683,9 @@ export function InsightsPage({ function InsightsWorkbenchPage({ workbenchesLoading, resolution, + onOpenRun, }: { + readonly onOpenRun: (id: string) => void; readonly workbenchesLoading: boolean; readonly resolution: ReturnType; }) { @@ -731,11 +734,7 @@ function InsightsWorkbenchPage({ />
- } - title="This workbench's activity lives in its own conversation" - description="Open the workbench to read its conversation." - /> +
@@ -795,6 +794,7 @@ export function InsightsRoute({ path }: { readonly path?: string }) { function InsightsWorkbenchPageRoute({ workbenchId, benchTenantId, + onOpenRun, }: { readonly workbenchId: string; readonly benchTenantId: string | null; @@ -802,7 +802,13 @@ function InsightsWorkbenchPageRoute({ }) { const { workbenches, isLoading } = useWorkbenchList(benchTenantId); const resolution = resolveWorkbenchInsightsScope(workbenches, workbenchId); - return ; + return ( + + ); } function useWorkbenchList(tenantId: string | null) {