91 lines
3.4 KiB
JavaScript
91 lines
3.4 KiB
JavaScript
import { execFile } from 'node:child_process';
|
|
import fs from 'node:fs/promises';
|
|
import path from 'node:path';
|
|
import { promisify } from 'node:util';
|
|
|
|
const execFileAsync = promisify(execFile);
|
|
|
|
function isTableLine(line) {
|
|
return /^\|.*\|\s*$/.test(line.trim());
|
|
}
|
|
|
|
function tableCells(line) {
|
|
return line.trim().slice(1, -1).split('|').map((cell) => cell.trim());
|
|
}
|
|
|
|
function isSeparator(cells) {
|
|
return cells.every((cell) => /^:?-{3,}:?$/.test(cell));
|
|
}
|
|
|
|
function extractDocumentContent(markdown) {
|
|
const titleMatch = markdown.match(/^#\s+.+\n+/m);
|
|
const body = markdown
|
|
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
|
|
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
|
|
.replace(/<br\s*\/?>/gi, ' '); // markup, not document text
|
|
return { body, fields: {} };
|
|
}
|
|
|
|
function words(text) {
|
|
return new Set((text.normalize('NFC').toLocaleLowerCase().match(/[\p{L}\p{N}]+/gu) ?? []));
|
|
}
|
|
|
|
function compact(text) {
|
|
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
|
|
}
|
|
|
|
async function markdownFiles(directory) {
|
|
const entries = await fs.readdir(directory, { withFileTypes: true });
|
|
const results = await Promise.all(entries.map(async (entry) => {
|
|
const target = path.join(directory, entry.name);
|
|
if (entry.isDirectory()) return markdownFiles(target);
|
|
return entry.isFile() && entry.name.endsWith('.md') ? [target] : [];
|
|
}));
|
|
return results.flat();
|
|
}
|
|
|
|
async function pdfText(pdf) {
|
|
const { stdout } = await execFileAsync('pdftotext', [pdf, '-'], { maxBuffer: 32 * 1024 * 1024 });
|
|
return stdout;
|
|
}
|
|
|
|
async function main() {
|
|
const [sourceRoot, deliveryRoot] = process.argv.slice(2);
|
|
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY');
|
|
|
|
const sources = (await markdownFiles(sourceRoot)).sort();
|
|
const failures = [];
|
|
|
|
for (const source of sources) {
|
|
const relative = path.relative(sourceRoot, source);
|
|
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
|
|
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
|
|
const { body } = extractDocumentContent(markdown);
|
|
// Thai has no word spaces, so a run can wrap several times inside a narrow
|
|
// table cell and reach the extracted text as separate fragments. Coverage is
|
|
// therefore checked by character count: every character of the source must
|
|
// appear in the PDF at least as many times. Clipped or dropped text loses
|
|
// glyphs and still fails; re-wrapping does not.
|
|
const tally = (text) => {
|
|
const counts = new Map();
|
|
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
|
return counts;
|
|
};
|
|
const rendered = tally(extracted);
|
|
const missing = [...tally(body)]
|
|
.filter(([character, count]) => (rendered.get(character) ?? 0) < count)
|
|
.map(([character, count]) => `${character}\u00d7${count - (rendered.get(character) ?? 0)}`);
|
|
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
|
|
}
|
|
|
|
if (failures.length) {
|
|
console.error(`Content coverage failed for ${failures.length}/${sources.length} PDFs:`);
|
|
failures.forEach((failure) => console.error(`- ${failure}`));
|
|
process.exit(1);
|
|
}
|
|
|
|
console.log(`Content coverage passed: ${sources.length}/${sources.length} PDFs.`);
|
|
}
|
|
|
|
main().catch((error) => { console.error(error); process.exit(1); });
|