Files

92 lines
3.4 KiB
JavaScript

import { execFile } from 'node:child_process';
import fs from 'node:fs/promises';
import path from 'node:path';
import { promisify } from 'node:util';
const execFileAsync = promisify(execFile);
function isTableLine(line) {
return /^\|.*\|\s*$/.test(line.trim());
}
function tableCells(line) {
return line.trim().slice(1, -1).split('|').map((cell) => cell.trim());
}
function isSeparator(cells) {
return cells.every((cell) => /^:?-{3,}:?$/.test(cell));
}
function extractDocumentContent(markdown) {
const titleMatch = markdown.match(/^#\s+.+\n+/m);
const afterTitle = markdown.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0));
const lines = afterTitle.split('\n');
for (let index = 0; index < lines.length; index += 1) {
if (!isTableLine(lines[index])) continue;
const table = [];
while (index < lines.length && isTableLine(lines[index])) table.push(lines[index++]);
const dataStart = isSeparator(tableCells(table[1] ?? '')) ? 2 : 1;
const fields = Object.fromEntries(table.slice(dataStart).map(tableCells).filter((row) => row.length >= 2));
if (fields.Document && fields.Project) return { body: lines.slice(index).join('\n'), fields };
index -= 1;
}
return { body: afterTitle, fields: {} };
}
function words(text) {
return new Set((text.normalize('NFC').toLocaleLowerCase().match(/[\p{L}\p{N}]+/gu) ?? []));
}
function compact(text) {
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
}
async function markdownFiles(directory) {
const entries = await fs.readdir(directory, { withFileTypes: true });
const results = await Promise.all(entries.map(async (entry) => {
const target = path.join(directory, entry.name);
if (entry.isDirectory()) return markdownFiles(target);
return entry.isFile() && entry.name.endsWith('.md') ? [target] : [];
}));
return results.flat();
}
async function pdfText(pdf) {
const { stdout } = await execFileAsync('pdftotext', [pdf, '-'], { maxBuffer: 32 * 1024 * 1024 });
return stdout;
}
async function main() {
const [sourceRoot, deliveryRoot] = process.argv.slice(2);
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY');
const sources = (await markdownFiles(sourceRoot)).sort();
const failures = [];
for (const source of sources) {
const relative = path.relative(sourceRoot, source);
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
const { body, fields } = extractDocumentContent(markdown);
const controlText = ['Document', 'Project', 'Project code', 'Project period', 'Release']
.map((field) => fields[field] ?? '')
.join('\n');
const expected = words(`${body}\n${controlText}`);
const actual = compact(extracted);
const missing = [...expected].filter((word) => !actual.includes(word));
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
}
if (failures.length) {
console.error(`Content coverage failed for ${failures.length}/${sources.length} PDFs:`);
failures.forEach((failure) => console.error(`- ${failure}`));
process.exit(1);
}
console.log(`Content coverage passed: ${sources.length}/${sources.length} PDFs.`);
}
main().catch((error) => { console.error(error); process.exit(1); });