Add Word rendering of the SDLC delivery package

This commit is contained in:
Thanakorn
2026-09-11 16:47:46 +07:00
parent f1f805c3a0
commit 75ff043036
62 changed files with 151 additions and 47 deletions
+93 -47
View File
@@ -2,34 +2,24 @@ import { execFile } from 'node:child_process';
import fs from 'node:fs/promises';
import path from 'node:path';
import { promisify } from 'node:util';
import { inflateRawSync } from 'node:zlib';
const execFileAsync = promisify(execFile);
function isTableLine(line) {
return /^\|.*\|\s*$/.test(line.trim());
}
function tableCells(line) {
return line.trim().slice(1, -1).split('|').map((cell) => cell.trim());
}
function isSeparator(cells) {
return cells.every((cell) => /^:?-{3,}:?$/.test(cell));
}
function extractDocumentContent(markdown) {
const titleMatch = markdown.match(/^#\s+.+\n+/m);
const body = markdown
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
.replace(/<br\s*\/?>/gi, ' '); // markup, not document text
.replace(/<br\s*\/?>/gi, ' ') // markup, not document text
// List markers are drawn by the renderer, not stored as text: the browser
// paints them from <ol>/<ul> and Word from numbering.xml, so they are
// markup here too and must not be expected in the extracted text.
.replace(/^[-*]\s+/gm, '')
.replace(/^\d+\.\s+/gm, '');
return { body, fields: {} };
}
function words(text) {
return new Set((text.normalize('NFC').toLocaleLowerCase().match(/[\p{L}\p{N}]+/gu) ?? []));
}
function compact(text) {
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
}
@@ -49,42 +39,98 @@ async function pdfText(pdf) {
return stdout;
}
// A .docx is a zip; only one member is needed, so it is read here rather than
// taking a dependency for it. Entries are located through the central
// directory, then inflated from their own local header.
function readZipEntry(buffer, name) {
const endOfDirectory = buffer.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
if (endOfDirectory < 0) throw new Error('not a zip archive');
const entries = buffer.readUInt16LE(endOfDirectory + 10);
let cursor = buffer.readUInt32LE(endOfDirectory + 16);
for (let index = 0; index < entries; index += 1) {
if (buffer.readUInt32LE(cursor) !== 0x02014b50) throw new Error('damaged central directory');
const method = buffer.readUInt16LE(cursor + 10);
const compressedSize = buffer.readUInt32LE(cursor + 20);
const nameLength = buffer.readUInt16LE(cursor + 28);
const extraLength = buffer.readUInt16LE(cursor + 30);
const commentLength = buffer.readUInt16LE(cursor + 32);
const localOffset = buffer.readUInt32LE(cursor + 42);
const entryName = buffer.toString('utf8', cursor + 46, cursor + 46 + nameLength);
if (entryName === name) {
const localName = buffer.readUInt16LE(localOffset + 26);
const localExtra = buffer.readUInt16LE(localOffset + 28);
const start = localOffset + 30 + localName + localExtra;
const data = buffer.subarray(start, start + compressedSize);
return method === 0 ? Buffer.from(data) : inflateRawSync(data);
}
cursor += 46 + nameLength + extraLength + commentLength;
}
throw new Error(`${name} not found in archive`);
}
async function docxText(docx) {
const xml = readZipEntry(await fs.readFile(docx), 'word/document.xml').toString('utf8');
// Paragraph and row ends are text boundaries; only <w:t> carries characters.
return xml
.replace(/<\/w:p>|<\/w:tr>|<w:br[^>]*\/>|<w:tab[^>]*\/>/g, '\n')
.replace(/<w:t[^>]*>([^<]*)<\/w:t>/g, '$1')
.replace(/<[^>]+>/g, '')
.replaceAll('&amp;', '&').replaceAll('&lt;', '<').replaceAll('&gt;', '>')
.replaceAll('&quot;', '"').replaceAll('&apos;', "'");
}
const READERS = { pdf: pdfText, docx: docxText };
function tally(text) {
const counts = new Map();
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
return counts;
}
async function main() {
const [sourceRoot, deliveryRoot] = process.argv.slice(2);
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY');
const [sourceRoot, deliveryRoot, ...requested] = process.argv.slice(2);
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY [FORMAT…]');
const formats = requested.length ? requested : Object.keys(READERS);
const unknown = formats.filter((format) => !READERS[format]);
if (unknown.length) throw new Error(`Unknown format(s): ${unknown.join(', ')}`);
const sources = (await markdownFiles(sourceRoot)).sort();
const failures = [];
let failed = false;
for (const source of sources) {
const relative = path.relative(sourceRoot, source);
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
const { body } = extractDocumentContent(markdown);
// Thai has no word spaces, so a run can wrap several times inside a narrow
// table cell and reach the extracted text as separate fragments. Coverage is
// therefore checked by character count: every character of the source must
// appear in the PDF at least as many times. Clipped or dropped text loses
// glyphs and still fails; re-wrapping does not.
const tally = (text) => {
const counts = new Map();
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
return counts;
};
const rendered = tally(extracted);
const missing = [...tally(body)]
.filter(([character, count]) => (rendered.get(character) ?? 0) < count)
.map(([character, count]) => `${character}\u00d7${count - (rendered.get(character) ?? 0)}`);
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
for (const format of formats) {
const failures = [];
for (const source of sources) {
const relative = path.relative(sourceRoot, source);
const rendered = path.join(deliveryRoot, relative.replace(/\.md$/, `.${format}`));
const [markdown, extracted] = await Promise.all([
fs.readFile(source, 'utf8'),
READERS[format](rendered)
]);
const { body } = extractDocumentContent(markdown);
// Thai has no word spaces, so a run can wrap several times inside a narrow
// table cell and reach the extracted text as separate fragments. Coverage is
// therefore checked by character count: every character of the source must
// appear in the rendering at least as many times. Clipped or dropped text
// loses glyphs and still fails; re-wrapping does not.
const counts = tally(extracted);
const missing = [...tally(body)]
.filter(([character, count]) => (counts.get(character) ?? 0) < count)
.map(([character, count]) => `${character}×${count - (counts.get(character) ?? 0)}`);
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
}
if (failures.length) {
failed = true;
console.error(`Content coverage failed for ${failures.length}/${sources.length} ${format.toUpperCase()}s:`);
failures.forEach((failure) => console.error(`- ${failure}`));
} else {
console.log(`Content coverage passed: ${sources.length}/${sources.length} ${format.toUpperCase()}s.`);
}
}
if (failures.length) {
console.error(`Content coverage failed for ${failures.length}/${sources.length} PDFs:`);
failures.forEach((failure) => console.error(`- ${failure}`));
process.exit(1);
}
console.log(`Content coverage passed: ${sources.length}/${sources.length} PDFs.`);
if (failed) process.exit(1);
}
main().catch((error) => { console.error(error); process.exit(1); });