Files
wms-app/scripts/sdlc-delivery/verify-content.mjs
T

140 lines
6.1 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { execFile } from 'node:child_process';
import fs from 'node:fs/promises';
import path from 'node:path';
import { promisify } from 'node:util';
import { inflateRawSync } from 'node:zlib';
const execFileAsync = promisify(execFile);
function extractDocumentContent(markdown) {
const titleMatch = markdown.match(/^#\s+.+\n+/m);
const body = markdown
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
.replace(/<br\s*\/?>/gi, ' ') // markup, not document text
// A figure prints its caption; the image path is markup.
.replace(/^!\[([^\]]*)\]\([^)]*\)\s*$/gm, '$1')
// List markers are drawn by the renderer, not stored as text: the browser
// paints them from <ol>/<ul> and Word from numbering.xml, so they are
// markup here too and must not be expected in the extracted text.
.replace(/^[-*]\s+/gm, '')
.replace(/^\d+\.\s+/gm, '');
return { body, fields: {} };
}
function compact(text) {
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
}
// Work products sit in the process folders; top-level files in sdlc/ are notes.
async function markdownFiles(directory, depth = 0) {
const entries = await fs.readdir(directory, { withFileTypes: true });
const results = await Promise.all(entries.map(async (entry) => {
const target = path.join(directory, entry.name);
if (entry.isDirectory()) return markdownFiles(target, depth + 1);
return depth > 0 && entry.isFile() && entry.name.endsWith('.md') ? [target] : [];
}));
return results.flat();
}
async function pdfText(pdf) {
const { stdout } = await execFileAsync('pdftotext', [pdf, '-'], { maxBuffer: 32 * 1024 * 1024 });
return stdout;
}
// A .docx is a zip; only one member is needed, so it is read here rather than
// taking a dependency for it. Entries are located through the central
// directory, then inflated from their own local header.
function readZipEntry(buffer, name) {
const endOfDirectory = buffer.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
if (endOfDirectory < 0) throw new Error('not a zip archive');
const entries = buffer.readUInt16LE(endOfDirectory + 10);
let cursor = buffer.readUInt32LE(endOfDirectory + 16);
for (let index = 0; index < entries; index += 1) {
if (buffer.readUInt32LE(cursor) !== 0x02014b50) throw new Error('damaged central directory');
const method = buffer.readUInt16LE(cursor + 10);
const compressedSize = buffer.readUInt32LE(cursor + 20);
const nameLength = buffer.readUInt16LE(cursor + 28);
const extraLength = buffer.readUInt16LE(cursor + 30);
const commentLength = buffer.readUInt16LE(cursor + 32);
const localOffset = buffer.readUInt32LE(cursor + 42);
const entryName = buffer.toString('utf8', cursor + 46, cursor + 46 + nameLength);
if (entryName === name) {
const localName = buffer.readUInt16LE(localOffset + 26);
const localExtra = buffer.readUInt16LE(localOffset + 28);
const start = localOffset + 30 + localName + localExtra;
const data = buffer.subarray(start, start + compressedSize);
return method === 0 ? Buffer.from(data) : inflateRawSync(data);
}
cursor += 46 + nameLength + extraLength + commentLength;
}
throw new Error(`${name} not found in archive`);
}
async function docxText(docx) {
const xml = readZipEntry(await fs.readFile(docx), 'word/document.xml').toString('utf8');
// Paragraph and row ends are text boundaries; only <w:t> carries characters.
return xml
.replace(/<\/w:p>|<\/w:tr>|<w:br[^>]*\/>|<w:tab[^>]*\/>/g, '\n')
.replace(/<w:t[^>]*>([^<]*)<\/w:t>/g, '$1')
.replace(/<[^>]+>/g, '')
.replaceAll('&amp;', '&').replaceAll('&lt;', '<').replaceAll('&gt;', '>')
.replaceAll('&quot;', '"').replaceAll('&apos;', "'");
}
const READERS = { pdf: pdfText, docx: docxText };
function tally(text) {
const counts = new Map();
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
return counts;
}
async function main() {
const [sourceRoot, deliveryRoot, ...requested] = process.argv.slice(2);
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY [FORMAT…]');
const formats = requested.length ? requested : Object.keys(READERS);
const unknown = formats.filter((format) => !READERS[format]);
if (unknown.length) throw new Error(`Unknown format(s): ${unknown.join(', ')}`);
const sources = (await markdownFiles(sourceRoot)).sort();
let failed = false;
for (const format of formats) {
const failures = [];
for (const source of sources) {
const relative = path.relative(sourceRoot, source);
const rendered = path.join(deliveryRoot, relative.replace(/\.md$/, `.${format}`));
const [markdown, extracted] = await Promise.all([
fs.readFile(source, 'utf8'),
READERS[format](rendered)
]);
const { body } = extractDocumentContent(markdown);
// Thai has no word spaces, so a run can wrap several times inside a narrow
// table cell and reach the extracted text as separate fragments. Coverage is
// therefore checked by character count: every character of the source must
// appear in the rendering at least as many times. Clipped or dropped text
// loses glyphs and still fails; re-wrapping does not.
const counts = tally(extracted);
const missing = [...tally(body)]
.filter(([character, count]) => (counts.get(character) ?? 0) < count)
.map(([character, count]) => `${character}×${count - (counts.get(character) ?? 0)}`);
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
}
if (failures.length) {
failed = true;
console.error(`Content coverage failed for ${failures.length}/${sources.length} ${format.toUpperCase()}s:`);
failures.forEach((failure) => console.error(`- ${failure}`));
} else {
console.log(`Content coverage passed: ${sources.length}/${sources.length} ${format.toUpperCase()}s.`);
}
}
if (failed) process.exit(1);
}
main().catch((error) => { console.error(error); process.exit(1); });