diff --git a/package.json b/package.json index 89c94775..5d8abf65 100644 --- a/package.json +++ b/package.json @@ -35,7 +35,7 @@ "build": "npm run build:clean; npm run test:types; pkgroll", "build:clean": "rm -rf dist", "build:watch": "npm run build -- --watch", - "build:collections": "tsx ./scripts/update.collection.patternFlyApi.ts && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections", + "build:collections": "UPDATE_COLLECTIONS=true tsx ./scripts/update.collection.patternFlyApi.ts && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections", "container:build": "bash ./scripts/container.build.sh", "container:start": "bash ./scripts/container.run.sh", "release": "changelog --non-cc --link-url https://github.com/patternfly/patternfly-mcp.git", @@ -45,6 +45,7 @@ "test:audit": "NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects audit", "test:audit-container": "npm run container:build && jest --selectProjects audit:container", "test:ci": "npm test -- --coverage", + "test:collections": "NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections", "test:dev": "npm test -- --watchAll", "test:integration": "npm run build && jest --selectProjects package && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects e2e", "test:integration-dev": "npm run test:integration -- --watchAll", diff --git a/scripts/update.collection.patternFlyApi.ts b/scripts/update.collection.patternFlyApi.ts index 609e3866..ab9f5add 100644 --- a/scripts/update.collection.patternFlyApi.ts +++ b/scripts/update.collection.patternFlyApi.ts @@ -1,32 +1,238 @@ -import { readFile, writeFile } from 'node:fs/promises'; -import { resolve } from 'node:path'; +import { readFile, writeFile, mkdir } from 'node:fs/promises'; +import { dirname, resolve } from 'node:path'; import { fileURLToPath } from 'node:url'; import { apiSpider, contentMetadata, + type ApiContent, type ApiCrawler, type ApiEmbedded, - type ApiEmbeddedCollection + type ApiEmbeddedCollection, + MIN_API_QUALITY_THRESHOLD } from '../src/collection.patternFlyApi'; -import { getOptions, runWithOptions } from '../src/options.context'; +import { getSessionOptions, getOptions, runWithOptions } from '../src/options.context'; +import { createLogger } from '../src/logger'; +import { type LoggingSession } from '../src/options.defaults'; /** - * Create a light diff report between old and new collections. + * Reason classification for omitted or removed API records. + */ +type RemovalReason = + 'lacks quality' | + 'empty response' | + 'deferred category' | + 'upstream removed' | + 'other'; + +/** + * Entry describing a removed record with its determined reason and details. + */ +interface RemovedRecordReport { + record: ApiEmbedded; + reason: RemovalReason; + details?: string; +} + +/** + * Entry describing a modified record and the changed fields. + */ +interface ModifiedRecordReport { + record: ApiEmbedded; + reasons: string[]; +} + +/** + * Options for generating CSV report output. + */ +interface GenerateCsvReportOptions { + diff: ReturnType; + oldRecords: ApiEmbedded[]; + newRecords: ApiEmbedded[]; + crawledMap: Map; +} + +/** + * Safely escape and format a field for standard RFC 4180 CSV output. + * + * @note **CSV / Formula Injection:** By default (`sanitizeFormulas = true`), leading formula + * trigger characters are prefixed with a single quote to prevent spreadsheet execution. Pass + * `false` to preserve strict raw string fidelity for automated downstream parsers. + * + * @param field - Value to format for CSV + * @param [sanitizeFormulas=true] - Whether to prefix formula trigger characters with a single quote + * @returns RFC 4180 compliant CSV cell string + */ +const escapeCsvField = (field: unknown, sanitizeFormulas = true): string => { + if (field === null || field === undefined) { + return ''; + } + + let str = String(field); + + if (sanitizeFormulas && /^[=+\-@\t\r]/.test(str)) { + str = `'${str}`; + } + + if (str.includes(',') || str.includes('"') || str.includes('\n') || str.includes('\r')) { + return `"${str.replace(/"/g, '""')}"`; + } + + return str; +}; + +/** + * Format rows and headers into standard CSV string. + * + * @param headers - Column headers + * @param rows - Table rows + */ +const formatCsv = (headers: string[], rows: (string | number | undefined | null)[][]): string => { + const headerLine = headers.map(field => escapeCsvField(field)).join(','); + const rowLines = rows.map(row => row.map(cell => escapeCsvField(cell)).join(',')); + + return [headerLine, ...rowLines].join('\n') + '\n'; +}; + +/** + * Generate a complete, non-truncated CSV report for additions, removals, modifications, and unchanged records. + * + * @param options - Generation options + * @param options.diff - Diff calculation between old and new records + * @param options.oldRecords - Previous collection records + * @param options.newRecords - Current collection records + * @param options.crawledMap - Map of crawled entries and metadata + */ +const generateReportCsv = ({ + diff, + oldRecords, + newRecords, + crawledMap +}: GenerateCsvReportOptions): string => { + const oldMap = new Map(oldRecords.map(record => [record.p, record])); + const headers = ['status', 'path', 'name', 'previousQualityScore', 'newQualityScore', 'contentType', 'reason', 'details']; + const rows: (string | number | undefined | null)[][] = []; + + for (const record of diff.added) { + rows.push(['ADDED', record.p, record.n, '', record.q, record.c, '', '']); + } + + for (const { record, reason, details } of diff.removed) { + const newQualityScore = crawledMap.get(record.p)?.entry.qualityScore ?? ''; + + rows.push(['REMOVED', record.p, record.n, record.q, newQualityScore, record.c, reason, details || '']); + } + + for (const { record, reasons } of diff.modified) { + const previousQualityScore = oldMap.get(record.p)?.q ?? ''; + + rows.push(['MODIFIED', record.p, record.n, previousQualityScore, record.q, record.c, 'property changes', reasons.join('; ')]); + } + + const changedPaths = new Set([ + ...diff.added.map(record => record.p), + ...diff.removed.map(removedItem => removedItem.record.p), + ...diff.modified.map(modifiedItem => modifiedItem.record.p) + ]); + + for (const record of newRecords) { + if (!changedPaths.has(record.p)) { + rows.push(['UNCHANGED', record.p, record.n, record.q, record.q, record.c, '', '']); + } + } + + return formatCsv(headers, rows); +}; + +/** + * Create a diff report with annotated reasons between old and new collections. * * @param oldRecords - Previous collection * @param newRecords - Updated collection + * @param crawledMap - Map of all crawled entries and evaluated metadata */ -const diffCollections = (oldRecords: ApiEmbedded[], newRecords: ApiEmbedded[]) => { +const diffCollections = ( + oldRecords: ApiEmbedded[], + newRecords: ApiEmbedded[], + crawledMap: Map +) => { const oldMap = new Map(oldRecords.map(record => [record.p, record])); const newMap = new Map(newRecords.map(record => [record.p, record])); const added = newRecords.filter(record => !oldMap.has(record.p)); - const removed = oldRecords.filter(record => !newMap.has(record.p)); - const modified = newRecords.filter(record => { + + const removed: RemovedRecordReport[] = []; + + for (const oldRecord of oldRecords) { + if (newMap.has(oldRecord.p)) { + continue; + } + + const crawled = crawledMap.get(oldRecord.p); + + if (!crawled) { + removed.push({ + record: oldRecord, + reason: 'upstream removed', + details: 'Endpoint no longer referenced upstream' + }); + } else if (!crawled.entry.content || crawled.entry.content.trim() === '' || crawled.entry.content === '{}' || crawled.entry.content === '[]') { + removed.push({ + record: oldRecord, + reason: 'empty response', + details: 'Empty payload returned' + }); + } else if (crawled.metadata.isDeferred) { + removed.push({ + record: oldRecord, + reason: 'deferred category', + details: `Category '${crawled.metadata.category}' is deferred` + }); + } else if (crawled.metadata.isLowQuality || crawled.entry.qualityScore < MIN_API_QUALITY_THRESHOLD) { + removed.push({ + record: oldRecord, + reason: 'lacks quality', + details: `Evaluated Q: ${crawled.entry.qualityScore} < ${MIN_API_QUALITY_THRESHOLD} threshold` + }); + } else { + removed.push({ + record: oldRecord, + reason: 'other', + details: `Excluded during crawl processing (Q: ${crawled.entry.qualityScore})` + }); + } + } + + const modified: ModifiedRecordReport[] = []; + + for (const record of newRecords) { const prev = oldMap.get(record.p); - return prev && (prev.q !== record.q || prev.n !== record.n || prev.d !== record.d || prev.c !== record.c); - }); + if (!prev) { + continue; + } + + const reasons: string[] = []; + + if (prev.q !== record.q) { + reasons.push(`quality score (${prev.q} -> ${record.q})`); + } + + if (prev.n !== record.n) { + reasons.push(`name ("${prev.n}" -> "${record.n}")`); + } + + if (prev.d !== record.d) { + reasons.push('description updated'); + } + + if (prev.c !== record.c) { + reasons.push(`content-type (${prev.c} -> ${record.c})`); + } + + if (reasons.length > 0) { + modified.push({ record, reasons }); + } + } return { added, removed, modified }; }; @@ -59,16 +265,20 @@ const diffReport = (diff: ReturnType) => { if (removed.length > 0) { console.log(` ➖ Removed (${removed.length}):`); - removed.slice(0, 10).forEach(record => console.log(` - ${record.p}`)); + removed.slice(0, 15).forEach(({ record, reason, details }) => { + console.log(` - ${record.p} (Previous Q: ${record.q}) [Reason: ${reason}${details ? ` — ${details}` : ''}]`); + }); - if (removed.length > 10) { - console.log(` ... and ${removed.length - 10} more`); + if (removed.length > 15) { + console.log(` ... and ${removed.length - 15} more`); } } if (modified.length > 0) { console.log(` 🔄 Modified (${modified.length}):`); - modified.slice(0, 10).forEach(record => console.log(` ~ ${record.p} (Q: ${record.q})`)); + modified.slice(0, 10).forEach(({ record, reasons }) => { + console.log(` ~ ${record.p} [${reasons.join(', ')}]`); + }); if (modified.length > 10) { console.log(` ... and ${modified.length - 10} more`); @@ -82,13 +292,24 @@ const diffReport = (diff: ReturnType) => { * @param [options] - Optional configuration options. * @param [options.isPrettyPrint=true] - Whether to pretty-print the JSON output. * @param [options.filterLowQualityRecords=false] - Whether to filter low-quality records based on the collection's criteria. + * @param [options.outputCsv=true] - Whether to generate and save a full CSV diff report. + * @param [options.csvOutputPath] - Custom path to write CSV report. */ const run = async ( { isPrettyPrint = true, - filterLowQualityRecords = false - }: { isPrettyPrint?: boolean; filterLowQualityRecords?: boolean; } = {} + filterLowQualityRecords = false, + outputCsv = true, + csvOutputPath + }: { isPrettyPrint?: boolean; filterLowQualityRecords?: boolean; outputCsv?: boolean; csvOutputPath?: string; } = {} ) => { + // 1. Enable stderr logging so all diagnostics_channel logs (debug, info, warn, error) are printed + const unsubscribeLogger = createLogger({ + channelName: getSessionOptions().channelName, + stderr: true, + level: 'debug' + } as LoggingSession); + console.log('🚀 Generating PatternFly API embedded collection...'); const keepAlive = setTimeout(() => {}, 86_400_000); @@ -105,17 +326,19 @@ const run = async ( } const recordsMap = new Map(); + const crawledMap = new Map(); for (const entry of entries) { // Generate full metadata using the shared contentMetadata function const metadata = contentMetadata(entry, options); + const relativePath = metadata.path.replace(base, '').replace(/^\//, ''); + + crawledMap.set(relativePath, { entry, metadata }); if (filterLowQualityRecords && (metadata.isDeferred || metadata.isLowQuality)) { continue; } - const relativePath = metadata.path.replace(base, '').replace(/^\//, ''); - if (recordsMap.has(relativePath)) { continue; } @@ -162,16 +385,45 @@ const run = async ( console.log(` - File Size: ${sizeKb} KB`); console.log(` - Time Elapsed: ${durationSec}s`); - diffReport(diffCollections(oldRecords, records)); + const diff = diffCollections(oldRecords, records, crawledMap); + + diffReport(diff); + + if (outputCsv) { + const targetCsvPath = csvOutputPath || + process.env.CSV_REPORT_PATH || + resolve(fileURLToPath(new URL('../reports/collection.patternFlyApi.report.csv', import.meta.url))); + + await mkdir(dirname(targetCsvPath), { recursive: true }); + const csvContent = generateReportCsv({ diff, oldRecords, newRecords: records, crawledMap }); + + await writeFile(targetCsvPath, csvContent, 'utf-8'); + console.log(`📄 Exported full CSV report: ${targetCsvPath}`); + } } finally { clearTimeout(keepAlive); + unsubscribeLogger(); } }; /** * Configurable options for maintainers. + * Only execute when explicitly requested via UPDATE_COLLECTIONS=true */ -run({ isPrettyPrint: true, filterLowQualityRecords: true }).catch(error => { - console.error('❌ Failed to update API collection:', error); - process.exit(1); -}); +if (process.env.UPDATE_COLLECTIONS === 'true') { + run({ isPrettyPrint: true, filterLowQualityRecords: true }).catch(error => { + console.error('❌ Failed to update API collection:', error); + process.exit(1); + }); +} + +export { + diffCollections, + escapeCsvField, + formatCsv, + generateReportCsv, + run, + type ModifiedRecordReport, + type RemovalReason, + type RemovedRecordReport +}; diff --git a/src/__tests__/__snapshots__/options.defaults.test.ts.snap b/src/__tests__/__snapshots__/options.defaults.test.ts.snap index fea1f917..558fb52f 100644 --- a/src/__tests__/__snapshots__/options.defaults.test.ts.snap +++ b/src/__tests__/__snapshots__/options.defaults.test.ts.snap @@ -63,7 +63,7 @@ exports[`options defaults should return specific properties: defaults 1`] = ` "intervalMs": 604800000, "repeat": Infinity, }, - "timeoutMs": 300000, + "timeoutMs": 1200000, "traversalPaths": [ "examples", ], diff --git a/src/__tests__/collection.patternFlyApi.test.ts b/src/__tests__/collection.patternFlyApi.test.ts index 0f3a70a7..17dd74a9 100644 --- a/src/__tests__/collection.patternFlyApi.test.ts +++ b/src/__tests__/collection.patternFlyApi.test.ts @@ -445,7 +445,8 @@ describe('crawler', () => { // It should have called for sub-item AND default componentPaths (props, css) // but my mock returns 'leaf' for everything else expect(res.length).toBeGreaterThanOrEqual(1); - expect(mockedProcessDocsFunction).toHaveBeenCalledWith(['https://api.com/v1']); + expect(mockedProcessDocsFunction) + .toHaveBeenCalledWith(expect.arrayContaining(['https://api.com/v1']), expect.objectContaining({})); }); it('aborts crawling early when signal is aborted', async () => { diff --git a/src/collection.patternFlyApi.ts b/src/collection.patternFlyApi.ts index a0b78ae0..03f2d685 100644 --- a/src/collection.patternFlyApi.ts +++ b/src/collection.patternFlyApi.ts @@ -365,7 +365,10 @@ const crawler = async ( return []; } - const settled = await processDocsFunction(uniqueUrls) || []; + const settled = await processDocsFunction( + uniqueUrls, { loadLimit: 200, parallelLoadLimit: 15, parallelLoadThrottleMs: 75 } + ) || []; + const content: ApiCrawler[] = []; for (const res of settled) { @@ -695,6 +698,7 @@ const patternFlyApiCollection = (options = getOptions(), session = getSessionOpt }; export { + MIN_API_QUALITY_THRESHOLD, patternFlyApiCollection, collectionCallback, collectionInitialCallback, diff --git a/src/options.defaults.ts b/src/options.defaults.ts index 5aba2798..8c4f8078 100644 --- a/src/options.defaults.ts +++ b/src/options.defaults.ts @@ -522,9 +522,7 @@ const CHANNEL_BASENAME = 'pf-mcp'; * Default PatternFly-specific options. * * @note Current settings for time - * - `timeoutMs` is set to `5` minutes to accommodate the current average crawl time - * of `75` seconds and potential network issues. This value should be adjusted as - * the API grows. + * - `timeoutMs` This value should be adjusted as the API grows. * - `schedule.intervalMs` is set to `7` days. Most users, without persistence, will * never achieve this. * - `schedule.delayStartMs` AFTER persistence is set up will be `6` hours. Short term @@ -541,7 +539,7 @@ const PATTERNFLY_OPTIONS: PatternFlyOptions = { traversalPaths: [ 'examples' ], - timeoutMs: 300_000, // 5 minutes + timeoutMs: 1_200_000, // 20 minutes schedule: { continueOnError: true, intervalMs: 24 * 60 * 60 * 1000 * 7, // 7 days diff --git a/tests/scripts/update.collection.patternFlyApi.test.ts b/tests/scripts/update.collection.patternFlyApi.test.ts index 1e2b0cb8..3f5aaeda 100644 --- a/tests/scripts/update.collection.patternFlyApi.test.ts +++ b/tests/scripts/update.collection.patternFlyApi.test.ts @@ -1,16 +1,21 @@ import { readFileSync, existsSync } from 'node:fs'; import { resolve } from 'node:path'; import { expandApiEmbeddedCollection, type ApiEmbeddedCollection } from '../../src/collection.patternFlyApi'; +import { escapeCsvField, formatCsv, generateReportCsv, diffCollections, run } from '../../scripts/update.collection.patternFlyApi'; + +const COLLECTION_PATH = resolve(process.cwd(), 'src/collection.patternFlyApi.json'); describe('collection.patternFlyApi', () => { - const catalogPath = resolve(process.cwd(), 'src/collection.patternFlyApi.json'); + it('should export the run function for programmatic invocation', () => { + expect(typeof run).toBe('function'); + }); - it('should have a generated collection catalog file', () => { - expect(existsSync(catalogPath)).toBe(true); + it('should have a generated collection file', () => { + expect(existsSync(COLLECTION_PATH)).toBe(true); }); it('should have a consistent JSON schema', () => { - const raw = readFileSync(catalogPath, 'utf-8'); + const raw = readFileSync(COLLECTION_PATH, 'utf-8'); const parsed: ApiEmbeddedCollection = JSON.parse(raw); expect(parsed).toMatchObject({ @@ -23,10 +28,12 @@ describe('collection.patternFlyApi', () => { }); it('should have records that are compressed and have key properties', () => { - const raw = readFileSync(catalogPath, 'utf-8'); + const raw = readFileSync(COLLECTION_PATH, 'utf-8'); const parsed: ApiEmbeddedCollection = JSON.parse(raw); + const sample = parsed.records.slice(0, 50); - for (const record of parsed.records) { + expect(parsed.records.length).toBeGreaterThan(0); + for (const record of sample) { expect(typeof record.p).toBe('string'); // path expect(typeof record.n).toBe('string'); // display name expect(typeof record.d).toBe('string'); // description @@ -40,11 +47,122 @@ describe('collection.patternFlyApi', () => { }); it('should be able to expanded and hydrate properties without errors', () => { - const raw = readFileSync(catalogPath, 'utf-8'); + const raw = readFileSync(COLLECTION_PATH, 'utf-8'); const parsed: ApiEmbeddedCollection = JSON.parse(raw); - const expanded = expandApiEmbeddedCollection(parsed); + const sampleRecords = parsed.records.slice(0, 50); + const expanded = expandApiEmbeddedCollection({ ...parsed, records: sampleRecords }); - expect(expanded.length).toBe(parsed.records.length); + expect(expanded.length).toBe(sampleRecords.length); expect(expanded[0]?.path?.startsWith(parsed.base)).toBe(true); }); }); + +describe('collection.patternFlyApi CSV Report Generator', () => { + it('should correctly escape fields with commas, quotes, and newlines', () => { + expect(escapeCsvField('normal')).toBe('normal'); + expect(escapeCsvField('with,comma')).toBe('"with,comma"'); + expect(escapeCsvField('with "quotes"')).toBe('"with ""quotes"""'); + expect(escapeCsvField('with\nnewline')).toBe('"with\nnewline"'); + expect(escapeCsvField(null)).toBe(''); + expect(escapeCsvField(undefined)).toBe(''); + expect(escapeCsvField(123)).toBe('123'); + }); + + it('should sanitize formula injection characters by default', () => { + expect(escapeCsvField('=SUM(1+1)')).toBe("'=SUM(1+1)"); + expect(escapeCsvField('+123')).toBe("'+123"); + expect(escapeCsvField('-456')).toBe("'-456"); + expect(escapeCsvField('@lookup')).toBe("'@lookup"); + expect(escapeCsvField('\ttabPrefix')).toBe("'\ttabPrefix"); + expect(escapeCsvField('\rreturnPrefix')).toBe('"\'\rreturnPrefix"'); + }); + + it('should preserve raw formula characters when sanitizeFormulas is set to false', () => { + expect(escapeCsvField('=SUM(1+1)', false)).toBe('=SUM(1+1)'); + expect(escapeCsvField('+123', false)).toBe('+123'); + expect(escapeCsvField('-456', false)).toBe('-456'); + expect(escapeCsvField('@lookup', false)).toBe('@lookup'); + }); + + it('should format header and row lines into standard CSV', () => { + const headers = ['col1', 'col2']; + const rows = [ + ['val1', 'val2'], + ['val3,with,comma', 'val4 "quoted"'] + ]; + + const result = formatCsv(headers, rows); + + expect(result).toBe('col1,col2\nval1,val2\n"val3,with,comma","val4 ""quoted"""\n'); + }); + + it('should produce a full structured CSV report for added, removed, modified, and unchanged records', () => { + const oldRecords = [ + { p: 'endpoint/removed', n: 'Old Doc', d: 'Desc', c: 'text/html', q: 0.96 }, + { p: 'endpoint/modified', n: 'Mod Doc', d: 'Old Desc', c: 'text/html', q: 0.95 }, + { p: 'endpoint/unchanged', n: 'Unchanged Doc', d: 'Desc', c: 'text/html', q: 0.98 } + ]; + + const newRecords = [ + { p: 'endpoint/added', n: 'New Doc', d: 'Desc', c: 'text/html', q: 0.97 }, + { p: 'endpoint/modified', n: 'Mod Doc', d: 'New Desc', c: 'text/html', q: 0.99 }, + { p: 'endpoint/unchanged', n: 'Unchanged Doc', d: 'Desc', c: 'text/html', q: 0.98 } + ]; + + const crawledMap = new Map(); + + crawledMap.set('endpoint/removed', { + entry: { qualityScore: 0.8, content: 'some content' }, + metadata: { isDeferred: false, isLowQuality: true, category: 'components' } + }); + crawledMap.set('endpoint/modified', { + entry: { qualityScore: 0.99, content: 'some content' }, + metadata: { isDeferred: false, isLowQuality: false, category: 'components' } + }); + crawledMap.set('endpoint/unchanged', { + entry: { qualityScore: 0.98, content: 'some content' }, + metadata: { isDeferred: false, isLowQuality: false, category: 'components' } + }); + crawledMap.set('endpoint/added', { + entry: { qualityScore: 0.97, content: 'some content' }, + metadata: { isDeferred: false, isLowQuality: false, category: 'components' } + }); + + const diff = diffCollections(oldRecords, newRecords, crawledMap); + const csv = generateReportCsv({ diff, oldRecords, newRecords, crawledMap }); + const lines = csv.trim().split('\n'); + + expect(lines[0]).toBe('status,path,name,previousQualityScore,newQualityScore,contentType,reason,details'); + expect(lines.some(line => line.startsWith('ADDED,endpoint/added,New Doc,,0.97'))).toBe(true); + expect(lines.some(line => line.startsWith('REMOVED,endpoint/removed,Old Doc,0.96,0.8') && line.includes('lacks quality'))).toBe(true); + expect(lines.some(line => line.startsWith('MODIFIED,endpoint/modified,Mod Doc,0.95,0.99') && line.includes('quality score'))).toBe(true); + expect(lines.some(line => line.startsWith('UNCHANGED,endpoint/unchanged,Unchanged Doc,0.98,0.98'))).toBe(true); + }); + + it('should generate a CSV report from a sample of collection records', () => { + const raw = readFileSync(COLLECTION_PATH, 'utf-8'); + const parsed: ApiEmbeddedCollection = JSON.parse(raw); + const sampleRecords = parsed.records.slice(0, 5); + const crawledMap = new Map(); + + for (const rec of sampleRecords) { + crawledMap.set(rec.p, { + entry: { qualityScore: rec.q, content: 'sample content' }, + metadata: { isDeferred: false, isLowQuality: false, category: 'components' } + }); + } + + const diff = diffCollections(sampleRecords, sampleRecords, crawledMap); + const csv = generateReportCsv({ + diff, + oldRecords: sampleRecords, + newRecords: sampleRecords, + crawledMap + }); + const lines = csv.trim().split('\n'); + + expect(lines[0]).toBe('status,path,name,previousQualityScore,newQualityScore,contentType,reason,details'); + expect(lines.length).toBe(sampleRecords.length + 1); + expect(lines.slice(1).every(line => line.startsWith('UNCHANGED,'))).toBe(true); + }); +});