140 lines
6.1 KiB
JavaScript
140 lines
6.1 KiB
JavaScript
import { execFile } from 'node:child_process';
|
||
import fs from 'node:fs/promises';
|
||
import path from 'node:path';
|
||
import { promisify } from 'node:util';
|
||
import { inflateRawSync } from 'node:zlib';
|
||
|
||
const execFileAsync = promisify(execFile);
|
||
|
||
function extractDocumentContent(markdown) {
|
||
const titleMatch = markdown.match(/^#\s+.+\n+/m);
|
||
const body = markdown
|
||
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
|
||
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
|
||
.replace(/<br\s*\/?>/gi, ' ') // markup, not document text
|
||
// A figure prints its caption; the image path is markup.
|
||
.replace(/^!\[([^\]]*)\]\([^)]*\)\s*$/gm, '$1')
|
||
// List markers are drawn by the renderer, not stored as text: the browser
|
||
// paints them from <ol>/<ul> and Word from numbering.xml, so they are
|
||
// markup here too and must not be expected in the extracted text.
|
||
.replace(/^[-*]\s+/gm, '')
|
||
.replace(/^\d+\.\s+/gm, '');
|
||
return { body, fields: {} };
|
||
}
|
||
|
||
function compact(text) {
|
||
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
|
||
}
|
||
|
||
// Work products sit in the process folders; top-level files in sdlc/ are notes.
|
||
async function markdownFiles(directory, depth = 0) {
|
||
const entries = await fs.readdir(directory, { withFileTypes: true });
|
||
const results = await Promise.all(entries.map(async (entry) => {
|
||
const target = path.join(directory, entry.name);
|
||
if (entry.isDirectory()) return markdownFiles(target, depth + 1);
|
||
return depth > 0 && entry.isFile() && entry.name.endsWith('.md') ? [target] : [];
|
||
}));
|
||
return results.flat();
|
||
}
|
||
|
||
async function pdfText(pdf) {
|
||
const { stdout } = await execFileAsync('pdftotext', [pdf, '-'], { maxBuffer: 32 * 1024 * 1024 });
|
||
return stdout;
|
||
}
|
||
|
||
// A .docx is a zip; only one member is needed, so it is read here rather than
|
||
// taking a dependency for it. Entries are located through the central
|
||
// directory, then inflated from their own local header.
|
||
function readZipEntry(buffer, name) {
|
||
const endOfDirectory = buffer.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
|
||
if (endOfDirectory < 0) throw new Error('not a zip archive');
|
||
const entries = buffer.readUInt16LE(endOfDirectory + 10);
|
||
let cursor = buffer.readUInt32LE(endOfDirectory + 16);
|
||
|
||
for (let index = 0; index < entries; index += 1) {
|
||
if (buffer.readUInt32LE(cursor) !== 0x02014b50) throw new Error('damaged central directory');
|
||
const method = buffer.readUInt16LE(cursor + 10);
|
||
const compressedSize = buffer.readUInt32LE(cursor + 20);
|
||
const nameLength = buffer.readUInt16LE(cursor + 28);
|
||
const extraLength = buffer.readUInt16LE(cursor + 30);
|
||
const commentLength = buffer.readUInt16LE(cursor + 32);
|
||
const localOffset = buffer.readUInt32LE(cursor + 42);
|
||
const entryName = buffer.toString('utf8', cursor + 46, cursor + 46 + nameLength);
|
||
|
||
if (entryName === name) {
|
||
const localName = buffer.readUInt16LE(localOffset + 26);
|
||
const localExtra = buffer.readUInt16LE(localOffset + 28);
|
||
const start = localOffset + 30 + localName + localExtra;
|
||
const data = buffer.subarray(start, start + compressedSize);
|
||
return method === 0 ? Buffer.from(data) : inflateRawSync(data);
|
||
}
|
||
cursor += 46 + nameLength + extraLength + commentLength;
|
||
}
|
||
throw new Error(`${name} not found in archive`);
|
||
}
|
||
|
||
async function docxText(docx) {
|
||
const xml = readZipEntry(await fs.readFile(docx), 'word/document.xml').toString('utf8');
|
||
// Paragraph and row ends are text boundaries; only <w:t> carries characters.
|
||
return xml
|
||
.replace(/<\/w:p>|<\/w:tr>|<w:br[^>]*\/>|<w:tab[^>]*\/>/g, '\n')
|
||
.replace(/<w:t[^>]*>([^<]*)<\/w:t>/g, '$1')
|
||
.replace(/<[^>]+>/g, '')
|
||
.replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>')
|
||
.replaceAll('"', '"').replaceAll(''', "'");
|
||
}
|
||
|
||
const READERS = { pdf: pdfText, docx: docxText };
|
||
|
||
function tally(text) {
|
||
const counts = new Map();
|
||
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
||
return counts;
|
||
}
|
||
|
||
async function main() {
|
||
const [sourceRoot, deliveryRoot, ...requested] = process.argv.slice(2);
|
||
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY [FORMAT…]');
|
||
const formats = requested.length ? requested : Object.keys(READERS);
|
||
const unknown = formats.filter((format) => !READERS[format]);
|
||
if (unknown.length) throw new Error(`Unknown format(s): ${unknown.join(', ')}`);
|
||
|
||
const sources = (await markdownFiles(sourceRoot)).sort();
|
||
let failed = false;
|
||
|
||
for (const format of formats) {
|
||
const failures = [];
|
||
for (const source of sources) {
|
||
const relative = path.relative(sourceRoot, source);
|
||
const rendered = path.join(deliveryRoot, relative.replace(/\.md$/, `.${format}`));
|
||
const [markdown, extracted] = await Promise.all([
|
||
fs.readFile(source, 'utf8'),
|
||
READERS[format](rendered)
|
||
]);
|
||
const { body } = extractDocumentContent(markdown);
|
||
// Thai has no word spaces, so a run can wrap several times inside a narrow
|
||
// table cell and reach the extracted text as separate fragments. Coverage is
|
||
// therefore checked by character count: every character of the source must
|
||
// appear in the rendering at least as many times. Clipped or dropped text
|
||
// loses glyphs and still fails; re-wrapping does not.
|
||
const counts = tally(extracted);
|
||
const missing = [...tally(body)]
|
||
.filter(([character, count]) => (counts.get(character) ?? 0) < count)
|
||
.map(([character, count]) => `${character}×${count - (counts.get(character) ?? 0)}`);
|
||
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
|
||
}
|
||
|
||
if (failures.length) {
|
||
failed = true;
|
||
console.error(`Content coverage failed for ${failures.length}/${sources.length} ${format.toUpperCase()}s:`);
|
||
failures.forEach((failure) => console.error(`- ${failure}`));
|
||
} else {
|
||
console.log(`Content coverage passed: ${sources.length}/${sources.length} ${format.toUpperCase()}s.`);
|
||
}
|
||
}
|
||
|
||
if (failed) process.exit(1);
|
||
}
|
||
|
||
main().catch((error) => { console.error(error); process.exit(1); });
|