Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 2 additions & 3 deletions web/lib/groups.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -321,10 +321,9 @@ describe.skipIf(!dockerAvailable())(
expect(res.headers.get('cache-control')).toBeNull();
});

it('treats legacy random-access rows without open_mode as hot access', async () => {
it('treats cached-only random-access rows as hot access', async () => {
const pool = getPool();
await pool.query("DELETE FROM random_access_times WHERE open_mode = 'reopen'");
await pool.query('ALTER TABLE random_access_times DROP COLUMN open_mode');

const summary = await collectGroupSummary({ k: 'RandomAccessGroup' });
if (summary === null || summary.type !== 'randomAccess') {
Expand All @@ -339,7 +338,7 @@ describe.skipIf(!dockerAvailable())(

const group = expectDefined(
await collectGroupCharts({ k: 'RandomAccessGroup' }, parseCommitWindow(null)),
'legacy random access group',
'cached-only random access group',
);
expect(Object.keys(group.charts[0].series).sort()).toEqual([
'arrow-ipc:hot',
Expand Down
2 changes: 1 addition & 1 deletion web/lib/queries.ts
Original file line number Diff line number Diff line change
Expand Up @@ -534,7 +534,7 @@ async function collectRandomAccessChart(
const text = `
SELECT r.commit_sha,
r.format,
COALESCE(to_jsonb(r) ->> 'open_mode', 'cached') AS open_mode,
r.open_mode AS open_mode,
r.value_ns::float8 AS value
FROM random_access_times r
JOIN commits c USING (commit_sha)
Expand Down
14 changes: 10 additions & 4 deletions web/lib/summary.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -177,8 +177,10 @@ describe('compression summaries', () => {
}
expect(calls[0][0]).toContain('latest_uncompressed_sizes');
for (const [text] of calls) {
expect(text).toContain("to_jsonb(s) ->> 'uncompressed_bytes'");
expect(text).not.toMatch(/\bs\.uncompressed_bytes\b/);
// Read the column directly: `to_jsonb(s)` serialized every row (a
// whole-table CPU cost on every summary) just to reach one field.
expect(text).toMatch(/\bs\.uncompressed_bytes\b/);
expect(text).not.toContain('to_jsonb(');
}
expect(calls[0][0]).toContain('LEFT JOIN latest_uncompressed_sizes');
expect(calls[1][0]).toContain('latest_uncompressed_sizes');
Expand Down Expand Up @@ -268,8 +270,12 @@ describe('timing summaries (shared ranking model)', () => {
const [text] = query.mock.calls[0] as [string, unknown[] | undefined];
// Per-series freshness, not one global latest commit: a format that skipped
// the newest commit stays on the card at its own last run.
expect(text).toContain("COALESCE(to_jsonb(r) ->> 'open_mode', 'cached')");
expect(text).not.toContain('r.open_mode');
// `open_mode` is read as a plain column. The previous `to_jsonb(r)` lookup
// serialized every row's `all_runtimes_ns` (tens of thousands of samples
// per random-access row), which is what pushed this statement past the
// server-side statement timeout in production.
expect(text).toContain('r.open_mode');
expect(text).not.toContain('to_jsonb(');
expect(text).toContain('c.timestamp DESC');
expect(text).not.toContain('MAX(c2.timestamp)');
});
Expand Down
33 changes: 15 additions & 18 deletions web/lib/summary.ts
Original file line number Diff line number Diff line change
Expand Up @@ -410,28 +410,25 @@ function groupRandomAccessSamples(
* `lance`: a format that skipped the newest commit is compared as of when it
* last ran instead of vanishing from the card.
*
* After migration 010, `random_access_times` holds one row per
* `(commit_sha, dataset, format, open_mode)`. Before that migration, the JSON
* lookup returns NULL and the query treats every historical row as `cached`.
* The table is small, so the per-series `DISTINCT ON` descent is cheap.
* Since migration 010, `random_access_times` holds one row per
* `(commit_sha, dataset, format, open_mode)`, with historical rows backfilled
* as `cached`. `open_mode` must be read as a plain column: an earlier
* `to_jsonb(r) ->> 'open_mode'` lookup serialized every row, including its
* `all_runtimes_ns` array of tens of thousands of samples, which made this
* statement exceed the server-side statement timeout in production. The
* table is small, so the per-series `DISTINCT ON` descent is cheap.
*/
async function collectRandomAccessSummary(): Promise<Summary | null> {
const text = `
SELECT DISTINCT ON (
r.dataset,
r.format,
COALESCE(to_jsonb(r) ->> 'open_mode', 'cached')
)
SELECT DISTINCT ON (r.dataset, r.format, r.open_mode)
r.dataset AS bucket,
r.format AS series,
COALESCE(to_jsonb(r) ->> 'open_mode', 'cached') AS open_mode,
r.open_mode AS open_mode,
r.value_ns::float8 AS value
FROM random_access_times r
JOIN commits c USING (commit_sha)
WHERE r.value_ns > 0
ORDER BY r.dataset,
r.format,
COALESCE(to_jsonb(r) ->> 'open_mode', 'cached'),
ORDER BY r.dataset, r.format, r.open_mode,
c.timestamp DESC,
r.commit_sha DESC
`;
Expand Down Expand Up @@ -645,11 +642,11 @@ async function compressionSamples(): Promise<
), latest_uncompressed_sizes AS (
SELECT DISTINCT ON (s.dataset, s.dataset_variant)
s.dataset, s.dataset_variant,
(to_jsonb(s) ->> 'uncompressed_bytes')::float8 AS uncompressed_bytes
s.uncompressed_bytes::float8 AS uncompressed_bytes
FROM compression_sizes s
JOIN commits c ON c.commit_sha = s.commit_sha
WHERE s.format = $4
AND (to_jsonb(s) ->> 'uncompressed_bytes')::float8 > 0
AND s.uncompressed_bytes::float8 > 0
AND lower(s.dataset) NOT LIKE '%wide table%'
ORDER BY s.dataset, s.dataset_variant NULLS FIRST,
c.timestamp DESC, s.commit_sha DESC
Expand Down Expand Up @@ -765,7 +762,7 @@ async function compressionSizeSamples(): Promise<
c.timestamp AS ts,
s.commit_sha AS commit_sha,
s.value_bytes::float8 AS value_bytes,
(to_jsonb(s) ->> 'uncompressed_bytes')::float8 AS uncompressed_bytes,
s.uncompressed_bytes::float8 AS uncompressed_bytes,
p.value_bytes::float8 AS parquet_bytes,
s.dataset AS dataset,
s.dataset_variant AS dataset_variant
Expand Down Expand Up @@ -818,11 +815,11 @@ async function compressionSizeSamples(): Promise<
), latest_uncompressed_sizes AS (
SELECT DISTINCT ON (s.dataset, s.dataset_variant)
s.dataset, s.dataset_variant,
(to_jsonb(s) ->> 'uncompressed_bytes')::float8 AS uncompressed_bytes
s.uncompressed_bytes::float8 AS uncompressed_bytes
FROM compression_sizes s
JOIN commits c ON c.commit_sha = s.commit_sha
WHERE s.format = $4
AND (to_jsonb(s) ->> 'uncompressed_bytes')::float8 > 0
AND s.uncompressed_bytes::float8 > 0
AND lower(s.dataset) NOT LIKE '%wide table%'
ORDER BY s.dataset, s.dataset_variant NULLS FIRST,
c.timestamp DESC, s.commit_sha DESC
Expand Down
Loading