Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@
"build": "npm run build:clean; npm run test:types; pkgroll",
"build:clean": "rm -rf dist",
"build:watch": "npm run build -- --watch",
"build:collections": "tsx ./scripts/update.collection.patternFlyApi.ts && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections",
"build:collections": "UPDATE_COLLECTIONS=true tsx ./scripts/update.collection.patternFlyApi.ts && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections",
"container:build": "bash ./scripts/container.build.sh",
"container:start": "bash ./scripts/container.run.sh",
"release": "changelog --non-cc --link-url https://github.com/patternfly/patternfly-mcp.git",
Expand All @@ -45,6 +45,7 @@
"test:audit": "NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects audit",
"test:audit-container": "npm run container:build && jest --selectProjects audit:container",
"test:ci": "npm test -- --coverage",
"test:collections": "NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects collections",
"test:dev": "npm test -- --watchAll",
"test:integration": "npm run build && jest --selectProjects package && NODE_OPTIONS='--experimental-vm-modules' jest --selectProjects e2e",
"test:integration-dev": "npm run test:integration -- --watchAll",
Expand Down
298 changes: 275 additions & 23 deletions scripts/update.collection.patternFlyApi.ts
Original file line number Diff line number Diff line change
@@ -1,32 +1,238 @@
import { readFile, writeFile } from 'node:fs/promises';
import { resolve } from 'node:path';
import { readFile, writeFile, mkdir } from 'node:fs/promises';
import { dirname, resolve } from 'node:path';
import { fileURLToPath } from 'node:url';
import {
apiSpider,
contentMetadata,
type ApiContent,
type ApiCrawler,
type ApiEmbedded,
type ApiEmbeddedCollection
type ApiEmbeddedCollection,
MIN_API_QUALITY_THRESHOLD
} from '../src/collection.patternFlyApi';
import { getOptions, runWithOptions } from '../src/options.context';
import { getSessionOptions, getOptions, runWithOptions } from '../src/options.context';
import { createLogger } from '../src/logger';
import { type LoggingSession } from '../src/options.defaults';

/**
* Create a light diff report between old and new collections.
* Reason classification for omitted or removed API records.
*/
type RemovalReason =
'lacks quality' |
'empty response' |
'deferred category' |
'upstream removed' |
'other';

/**
* Entry describing a removed record with its determined reason and details.
*/
interface RemovedRecordReport {
record: ApiEmbedded;
reason: RemovalReason;
details?: string;
}

/**
* Entry describing a modified record and the changed fields.
*/
interface ModifiedRecordReport {
record: ApiEmbedded;
reasons: string[];
}

/**
* Options for generating CSV report output.
*/
interface GenerateCsvReportOptions {
diff: ReturnType<typeof diffCollections>;
oldRecords: ApiEmbedded[];
newRecords: ApiEmbedded[];
crawledMap: Map<string, { entry: ApiCrawler; metadata: ApiContent }>;
}

/**
* Safely escape and format a field for standard RFC 4180 CSV output.
*
* @note **CSV / Formula Injection:** By default (`sanitizeFormulas = true`), leading formula
* trigger characters are prefixed with a single quote to prevent spreadsheet execution. Pass
* `false` to preserve strict raw string fidelity for automated downstream parsers.
*
* @param field - Value to format for CSV
* @param [sanitizeFormulas=true] - Whether to prefix formula trigger characters with a single quote
* @returns RFC 4180 compliant CSV cell string
*/
const escapeCsvField = (field: unknown, sanitizeFormulas = true): string => {
if (field === null || field === undefined) {
return '';
}

let str = String(field);

if (sanitizeFormulas && /^[=+\-@\t\r]/.test(str)) {
str = `'${str}`;
}

if (str.includes(',') || str.includes('"') || str.includes('\n') || str.includes('\r')) {
return `"${str.replace(/"/g, '""')}"`;
}

return str;
};

/**
* Format rows and headers into standard CSV string.
*
* @param headers - Column headers
* @param rows - Table rows
*/
const formatCsv = (headers: string[], rows: (string | number | undefined | null)[][]): string => {
const headerLine = headers.map(field => escapeCsvField(field)).join(',');
const rowLines = rows.map(row => row.map(cell => escapeCsvField(cell)).join(','));

return [headerLine, ...rowLines].join('\n') + '\n';
};

/**
* Generate a complete, non-truncated CSV report for additions, removals, modifications, and unchanged records.
*
* @param options - Generation options
* @param options.diff - Diff calculation between old and new records
* @param options.oldRecords - Previous collection records
* @param options.newRecords - Current collection records
* @param options.crawledMap - Map of crawled entries and metadata
*/
const generateReportCsv = ({
diff,
oldRecords,
newRecords,
crawledMap
}: GenerateCsvReportOptions): string => {
const oldMap = new Map(oldRecords.map(record => [record.p, record]));
const headers = ['status', 'path', 'name', 'previousQualityScore', 'newQualityScore', 'contentType', 'reason', 'details'];
const rows: (string | number | undefined | null)[][] = [];

for (const record of diff.added) {
rows.push(['ADDED', record.p, record.n, '', record.q, record.c, '', '']);
}

for (const { record, reason, details } of diff.removed) {
const newQualityScore = crawledMap.get(record.p)?.entry.qualityScore ?? '';

rows.push(['REMOVED', record.p, record.n, record.q, newQualityScore, record.c, reason, details || '']);
}

for (const { record, reasons } of diff.modified) {
const previousQualityScore = oldMap.get(record.p)?.q ?? '';

rows.push(['MODIFIED', record.p, record.n, previousQualityScore, record.q, record.c, 'property changes', reasons.join('; ')]);
}

const changedPaths = new Set([
...diff.added.map(record => record.p),
...diff.removed.map(removedItem => removedItem.record.p),
...diff.modified.map(modifiedItem => modifiedItem.record.p)
]);

for (const record of newRecords) {
if (!changedPaths.has(record.p)) {
rows.push(['UNCHANGED', record.p, record.n, record.q, record.q, record.c, '', '']);
}
}

return formatCsv(headers, rows);
};

/**
* Create a diff report with annotated reasons between old and new collections.
*
* @param oldRecords - Previous collection
* @param newRecords - Updated collection
* @param crawledMap - Map of all crawled entries and evaluated metadata
*/
const diffCollections = (oldRecords: ApiEmbedded[], newRecords: ApiEmbedded[]) => {
const diffCollections = (
oldRecords: ApiEmbedded[],
newRecords: ApiEmbedded[],
crawledMap: Map<string, { entry: ApiCrawler; metadata: ApiContent }>
) => {
const oldMap = new Map(oldRecords.map(record => [record.p, record]));
const newMap = new Map(newRecords.map(record => [record.p, record]));

const added = newRecords.filter(record => !oldMap.has(record.p));
const removed = oldRecords.filter(record => !newMap.has(record.p));
const modified = newRecords.filter(record => {

const removed: RemovedRecordReport[] = [];

for (const oldRecord of oldRecords) {
if (newMap.has(oldRecord.p)) {
continue;
}

const crawled = crawledMap.get(oldRecord.p);

if (!crawled) {
removed.push({
record: oldRecord,
reason: 'upstream removed',
details: 'Endpoint no longer referenced upstream'
});
} else if (!crawled.entry.content || crawled.entry.content.trim() === '' || crawled.entry.content === '{}' || crawled.entry.content === '[]') {
removed.push({
record: oldRecord,
reason: 'empty response',
details: 'Empty payload returned'
});
} else if (crawled.metadata.isDeferred) {
removed.push({
record: oldRecord,
reason: 'deferred category',
details: `Category '${crawled.metadata.category}' is deferred`
});
} else if (crawled.metadata.isLowQuality || crawled.entry.qualityScore < MIN_API_QUALITY_THRESHOLD) {
removed.push({
record: oldRecord,
reason: 'lacks quality',
details: `Evaluated Q: ${crawled.entry.qualityScore} < ${MIN_API_QUALITY_THRESHOLD} threshold`
});
} else {
removed.push({
record: oldRecord,
reason: 'other',
details: `Excluded during crawl processing (Q: ${crawled.entry.qualityScore})`
});
}
}

const modified: ModifiedRecordReport[] = [];

for (const record of newRecords) {
const prev = oldMap.get(record.p);

return prev && (prev.q !== record.q || prev.n !== record.n || prev.d !== record.d || prev.c !== record.c);
});
if (!prev) {
continue;
}

const reasons: string[] = [];

if (prev.q !== record.q) {
reasons.push(`quality score (${prev.q} -> ${record.q})`);
}

if (prev.n !== record.n) {
reasons.push(`name ("${prev.n}" -> "${record.n}")`);
}

if (prev.d !== record.d) {
reasons.push('description updated');
}

if (prev.c !== record.c) {
reasons.push(`content-type (${prev.c} -> ${record.c})`);
}

if (reasons.length > 0) {
modified.push({ record, reasons });
}
}

return { added, removed, modified };
};
Expand Down Expand Up @@ -59,16 +265,20 @@ const diffReport = (diff: ReturnType<typeof diffCollections>) => {

if (removed.length > 0) {
console.log(` ➖ Removed (${removed.length}):`);
removed.slice(0, 10).forEach(record => console.log(` - ${record.p}`));
removed.slice(0, 15).forEach(({ record, reason, details }) => {
console.log(` - ${record.p} (Previous Q: ${record.q}) [Reason: ${reason}${details ? ` — ${details}` : ''}]`);
});

if (removed.length > 10) {
console.log(` ... and ${removed.length - 10} more`);
if (removed.length > 15) {
console.log(` ... and ${removed.length - 15} more`);
}
}

if (modified.length > 0) {
console.log(` 🔄 Modified (${modified.length}):`);
modified.slice(0, 10).forEach(record => console.log(` ~ ${record.p} (Q: ${record.q})`));
modified.slice(0, 10).forEach(({ record, reasons }) => {
console.log(` ~ ${record.p} [${reasons.join(', ')}]`);
});

if (modified.length > 10) {
console.log(` ... and ${modified.length - 10} more`);
Expand All @@ -82,13 +292,24 @@ const diffReport = (diff: ReturnType<typeof diffCollections>) => {
* @param [options] - Optional configuration options.
* @param [options.isPrettyPrint=true] - Whether to pretty-print the JSON output.
* @param [options.filterLowQualityRecords=false] - Whether to filter low-quality records based on the collection's criteria.
* @param [options.outputCsv=true] - Whether to generate and save a full CSV diff report.
* @param [options.csvOutputPath] - Custom path to write CSV report.
*/
const run = async (
{
isPrettyPrint = true,
filterLowQualityRecords = false
}: { isPrettyPrint?: boolean; filterLowQualityRecords?: boolean; } = {}
filterLowQualityRecords = false,
outputCsv = true,
csvOutputPath
}: { isPrettyPrint?: boolean; filterLowQualityRecords?: boolean; outputCsv?: boolean; csvOutputPath?: string; } = {}
) => {
// 1. Enable stderr logging so all diagnostics_channel logs (debug, info, warn, error) are printed
const unsubscribeLogger = createLogger({
channelName: getSessionOptions().channelName,
stderr: true,
level: 'debug'
} as LoggingSession);

console.log('🚀 Generating PatternFly API embedded collection...');
const keepAlive = setTimeout(() => {}, 86_400_000);

Expand All @@ -105,17 +326,19 @@ const run = async (
}

const recordsMap = new Map<string, ApiEmbedded>();
const crawledMap = new Map<string, { entry: ApiCrawler; metadata: ApiContent }>();

for (const entry of entries) {
// Generate full metadata using the shared contentMetadata function
const metadata = contentMetadata(entry, options);
const relativePath = metadata.path.replace(base, '').replace(/^\//, '');

crawledMap.set(relativePath, { entry, metadata });

if (filterLowQualityRecords && (metadata.isDeferred || metadata.isLowQuality)) {
continue;
}

const relativePath = metadata.path.replace(base, '').replace(/^\//, '');

if (recordsMap.has(relativePath)) {
continue;
}
Expand Down Expand Up @@ -162,16 +385,45 @@ const run = async (
console.log(` - File Size: ${sizeKb} KB`);
console.log(` - Time Elapsed: ${durationSec}s`);

diffReport(diffCollections(oldRecords, records));
const diff = diffCollections(oldRecords, records, crawledMap);

diffReport(diff);

if (outputCsv) {
const targetCsvPath = csvOutputPath ||
process.env.CSV_REPORT_PATH ||
resolve(fileURLToPath(new URL('../reports/collection.patternFlyApi.report.csv', import.meta.url)));

await mkdir(dirname(targetCsvPath), { recursive: true });
const csvContent = generateReportCsv({ diff, oldRecords, newRecords: records, crawledMap });

await writeFile(targetCsvPath, csvContent, 'utf-8');
console.log(`📄 Exported full CSV report: ${targetCsvPath}`);
}
} finally {
clearTimeout(keepAlive);
unsubscribeLogger();
}
};

/**
* Configurable options for maintainers.
* Only execute when explicitly requested via UPDATE_COLLECTIONS=true
*/
run({ isPrettyPrint: true, filterLowQualityRecords: true }).catch(error => {
console.error('❌ Failed to update API collection:', error);
process.exit(1);
});
if (process.env.UPDATE_COLLECTIONS === 'true') {
run({ isPrettyPrint: true, filterLowQualityRecords: true }).catch(error => {
console.error('❌ Failed to update API collection:', error);
process.exit(1);
});
}

export {
diffCollections,
escapeCsvField,
formatCsv,
generateReportCsv,
run,
type ModifiedRecordReport,
type RemovalReason,
type RemovedRecordReport
};
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ exports[`options defaults should return specific properties: defaults 1`] = `
"intervalMs": 604800000,
"repeat": Infinity,
},
"timeoutMs": 300000,
"timeoutMs": 1200000,
"traversalPaths": [
"examples",
],
Expand Down
Loading
Loading