Add Word rendering of the SDLC delivery package
This commit is contained in:
@@ -0,0 +1,57 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert the rendered delivery package from PDF to Word.
|
||||
|
||||
The work products are authored in Markdown and rendered to PDF by
|
||||
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
|
||||
PDF, so the Word file keeps the page the reviewer signed off rather than being
|
||||
laid out a second time from the source.
|
||||
|
||||
Requires pdf2docx: pip install -r requirements.txt
|
||||
|
||||
Usage:
|
||||
convert-to-docx.py DELIVERY_DIRECTORY [PDF…]
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from pdf2docx import Converter
|
||||
|
||||
|
||||
def convert(pdf: Path) -> Path:
|
||||
docx = pdf.with_suffix('.docx')
|
||||
converter = Converter(str(pdf))
|
||||
try:
|
||||
converter.convert(str(docx))
|
||||
finally:
|
||||
converter.close()
|
||||
return docx
|
||||
|
||||
|
||||
def main(argv):
|
||||
if not argv:
|
||||
raise SystemExit(__doc__)
|
||||
|
||||
root = Path(argv[0])
|
||||
pdfs = [Path(p) for p in argv[1:]] or sorted(root.rglob('*.pdf'))
|
||||
if not pdfs:
|
||||
raise SystemExit(f'No PDFs found under {root}')
|
||||
|
||||
failures = []
|
||||
for index, pdf in enumerate(pdfs, start=1):
|
||||
try:
|
||||
docx = convert(pdf)
|
||||
print(f'[{index}/{len(pdfs)}] {docx.relative_to(root)}')
|
||||
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
||||
failures.append((pdf, error))
|
||||
print(f'[{index}/{len(pdfs)}] FAILED {pdf.relative_to(root)}: {error}', file=sys.stderr)
|
||||
|
||||
print(f'{len(pdfs) - len(failures)}/{len(pdfs)} documents converted')
|
||||
if failures:
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# pdf2docx is chatty; the per-document progress above is enough.
|
||||
os.environ.setdefault('PDF2DOCX_QUIET', '1')
|
||||
main(sys.argv[1:])
|
||||
@@ -0,0 +1 @@
|
||||
pdf2docx>=0.5.8
|
||||
@@ -2,34 +2,24 @@ import { execFile } from 'node:child_process';
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { promisify } from 'node:util';
|
||||
import { inflateRawSync } from 'node:zlib';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function isTableLine(line) {
|
||||
return /^\|.*\|\s*$/.test(line.trim());
|
||||
}
|
||||
|
||||
function tableCells(line) {
|
||||
return line.trim().slice(1, -1).split('|').map((cell) => cell.trim());
|
||||
}
|
||||
|
||||
function isSeparator(cells) {
|
||||
return cells.every((cell) => /^:?-{3,}:?$/.test(cell));
|
||||
}
|
||||
|
||||
function extractDocumentContent(markdown) {
|
||||
const titleMatch = markdown.match(/^#\s+.+\n+/m);
|
||||
const body = markdown
|
||||
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
|
||||
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
|
||||
.replace(/<br\s*\/?>/gi, ' '); // markup, not document text
|
||||
.replace(/<br\s*\/?>/gi, ' ') // markup, not document text
|
||||
// List markers are drawn by the renderer, not stored as text: the browser
|
||||
// paints them from <ol>/<ul> and Word from numbering.xml, so they are
|
||||
// markup here too and must not be expected in the extracted text.
|
||||
.replace(/^[-*]\s+/gm, '')
|
||||
.replace(/^\d+\.\s+/gm, '');
|
||||
return { body, fields: {} };
|
||||
}
|
||||
|
||||
function words(text) {
|
||||
return new Set((text.normalize('NFC').toLocaleLowerCase().match(/[\p{L}\p{N}]+/gu) ?? []));
|
||||
}
|
||||
|
||||
function compact(text) {
|
||||
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
|
||||
}
|
||||
@@ -49,42 +39,98 @@ async function pdfText(pdf) {
|
||||
return stdout;
|
||||
}
|
||||
|
||||
// A .docx is a zip; only one member is needed, so it is read here rather than
|
||||
// taking a dependency for it. Entries are located through the central
|
||||
// directory, then inflated from their own local header.
|
||||
function readZipEntry(buffer, name) {
|
||||
const endOfDirectory = buffer.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
|
||||
if (endOfDirectory < 0) throw new Error('not a zip archive');
|
||||
const entries = buffer.readUInt16LE(endOfDirectory + 10);
|
||||
let cursor = buffer.readUInt32LE(endOfDirectory + 16);
|
||||
|
||||
for (let index = 0; index < entries; index += 1) {
|
||||
if (buffer.readUInt32LE(cursor) !== 0x02014b50) throw new Error('damaged central directory');
|
||||
const method = buffer.readUInt16LE(cursor + 10);
|
||||
const compressedSize = buffer.readUInt32LE(cursor + 20);
|
||||
const nameLength = buffer.readUInt16LE(cursor + 28);
|
||||
const extraLength = buffer.readUInt16LE(cursor + 30);
|
||||
const commentLength = buffer.readUInt16LE(cursor + 32);
|
||||
const localOffset = buffer.readUInt32LE(cursor + 42);
|
||||
const entryName = buffer.toString('utf8', cursor + 46, cursor + 46 + nameLength);
|
||||
|
||||
if (entryName === name) {
|
||||
const localName = buffer.readUInt16LE(localOffset + 26);
|
||||
const localExtra = buffer.readUInt16LE(localOffset + 28);
|
||||
const start = localOffset + 30 + localName + localExtra;
|
||||
const data = buffer.subarray(start, start + compressedSize);
|
||||
return method === 0 ? Buffer.from(data) : inflateRawSync(data);
|
||||
}
|
||||
cursor += 46 + nameLength + extraLength + commentLength;
|
||||
}
|
||||
throw new Error(`${name} not found in archive`);
|
||||
}
|
||||
|
||||
async function docxText(docx) {
|
||||
const xml = readZipEntry(await fs.readFile(docx), 'word/document.xml').toString('utf8');
|
||||
// Paragraph and row ends are text boundaries; only <w:t> carries characters.
|
||||
return xml
|
||||
.replace(/<\/w:p>|<\/w:tr>|<w:br[^>]*\/>|<w:tab[^>]*\/>/g, '\n')
|
||||
.replace(/<w:t[^>]*>([^<]*)<\/w:t>/g, '$1')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
.replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>')
|
||||
.replaceAll('"', '"').replaceAll(''', "'");
|
||||
}
|
||||
|
||||
const READERS = { pdf: pdfText, docx: docxText };
|
||||
|
||||
function tally(text) {
|
||||
const counts = new Map();
|
||||
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
||||
return counts;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const [sourceRoot, deliveryRoot] = process.argv.slice(2);
|
||||
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY');
|
||||
const [sourceRoot, deliveryRoot, ...requested] = process.argv.slice(2);
|
||||
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY [FORMAT…]');
|
||||
const formats = requested.length ? requested : Object.keys(READERS);
|
||||
const unknown = formats.filter((format) => !READERS[format]);
|
||||
if (unknown.length) throw new Error(`Unknown format(s): ${unknown.join(', ')}`);
|
||||
|
||||
const sources = (await markdownFiles(sourceRoot)).sort();
|
||||
const failures = [];
|
||||
let failed = false;
|
||||
|
||||
for (const source of sources) {
|
||||
const relative = path.relative(sourceRoot, source);
|
||||
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
|
||||
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
|
||||
const { body } = extractDocumentContent(markdown);
|
||||
// Thai has no word spaces, so a run can wrap several times inside a narrow
|
||||
// table cell and reach the extracted text as separate fragments. Coverage is
|
||||
// therefore checked by character count: every character of the source must
|
||||
// appear in the PDF at least as many times. Clipped or dropped text loses
|
||||
// glyphs and still fails; re-wrapping does not.
|
||||
const tally = (text) => {
|
||||
const counts = new Map();
|
||||
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
||||
return counts;
|
||||
};
|
||||
const rendered = tally(extracted);
|
||||
const missing = [...tally(body)]
|
||||
.filter(([character, count]) => (rendered.get(character) ?? 0) < count)
|
||||
.map(([character, count]) => `${character}\u00d7${count - (rendered.get(character) ?? 0)}`);
|
||||
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
|
||||
for (const format of formats) {
|
||||
const failures = [];
|
||||
for (const source of sources) {
|
||||
const relative = path.relative(sourceRoot, source);
|
||||
const rendered = path.join(deliveryRoot, relative.replace(/\.md$/, `.${format}`));
|
||||
const [markdown, extracted] = await Promise.all([
|
||||
fs.readFile(source, 'utf8'),
|
||||
READERS[format](rendered)
|
||||
]);
|
||||
const { body } = extractDocumentContent(markdown);
|
||||
// Thai has no word spaces, so a run can wrap several times inside a narrow
|
||||
// table cell and reach the extracted text as separate fragments. Coverage is
|
||||
// therefore checked by character count: every character of the source must
|
||||
// appear in the rendering at least as many times. Clipped or dropped text
|
||||
// loses glyphs and still fails; re-wrapping does not.
|
||||
const counts = tally(extracted);
|
||||
const missing = [...tally(body)]
|
||||
.filter(([character, count]) => (counts.get(character) ?? 0) < count)
|
||||
.map(([character, count]) => `${character}×${count - (counts.get(character) ?? 0)}`);
|
||||
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
|
||||
}
|
||||
|
||||
if (failures.length) {
|
||||
failed = true;
|
||||
console.error(`Content coverage failed for ${failures.length}/${sources.length} ${format.toUpperCase()}s:`);
|
||||
failures.forEach((failure) => console.error(`- ${failure}`));
|
||||
} else {
|
||||
console.log(`Content coverage passed: ${sources.length}/${sources.length} ${format.toUpperCase()}s.`);
|
||||
}
|
||||
}
|
||||
|
||||
if (failures.length) {
|
||||
console.error(`Content coverage failed for ${failures.length}/${sources.length} PDFs:`);
|
||||
failures.forEach((failure) => console.error(`- ${failure}`));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`Content coverage passed: ${sources.length}/${sources.length} PDFs.`);
|
||||
if (failed) process.exit(1);
|
||||
}
|
||||
|
||||
main().catch((error) => { console.error(error); process.exit(1); });
|
||||
|
||||
Reference in New Issue
Block a user