Add Word rendering of the SDLC delivery package
This commit is contained in:
@@ -0,0 +1,57 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert the rendered delivery package from PDF to Word.
|
||||
|
||||
The work products are authored in Markdown and rendered to PDF by
|
||||
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
|
||||
PDF, so the Word file keeps the page the reviewer signed off rather than being
|
||||
laid out a second time from the source.
|
||||
|
||||
Requires pdf2docx: pip install -r requirements.txt
|
||||
|
||||
Usage:
|
||||
convert-to-docx.py DELIVERY_DIRECTORY [PDF…]
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from pdf2docx import Converter
|
||||
|
||||
|
||||
def convert(pdf: Path) -> Path:
|
||||
docx = pdf.with_suffix('.docx')
|
||||
converter = Converter(str(pdf))
|
||||
try:
|
||||
converter.convert(str(docx))
|
||||
finally:
|
||||
converter.close()
|
||||
return docx
|
||||
|
||||
|
||||
def main(argv):
|
||||
if not argv:
|
||||
raise SystemExit(__doc__)
|
||||
|
||||
root = Path(argv[0])
|
||||
pdfs = [Path(p) for p in argv[1:]] or sorted(root.rglob('*.pdf'))
|
||||
if not pdfs:
|
||||
raise SystemExit(f'No PDFs found under {root}')
|
||||
|
||||
failures = []
|
||||
for index, pdf in enumerate(pdfs, start=1):
|
||||
try:
|
||||
docx = convert(pdf)
|
||||
print(f'[{index}/{len(pdfs)}] {docx.relative_to(root)}')
|
||||
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
||||
failures.append((pdf, error))
|
||||
print(f'[{index}/{len(pdfs)}] FAILED {pdf.relative_to(root)}: {error}', file=sys.stderr)
|
||||
|
||||
print(f'{len(pdfs) - len(failures)}/{len(pdfs)} documents converted')
|
||||
if failures:
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# pdf2docx is chatty; the per-document progress above is enough.
|
||||
os.environ.setdefault('PDF2DOCX_QUIET', '1')
|
||||
main(sys.argv[1:])
|
||||
@@ -0,0 +1 @@
|
||||
pdf2docx>=0.5.8
|
||||
@@ -2,34 +2,24 @@ import { execFile } from 'node:child_process';
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { promisify } from 'node:util';
|
||||
import { inflateRawSync } from 'node:zlib';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function isTableLine(line) {
|
||||
return /^\|.*\|\s*$/.test(line.trim());
|
||||
}
|
||||
|
||||
function tableCells(line) {
|
||||
return line.trim().slice(1, -1).split('|').map((cell) => cell.trim());
|
||||
}
|
||||
|
||||
function isSeparator(cells) {
|
||||
return cells.every((cell) => /^:?-{3,}:?$/.test(cell));
|
||||
}
|
||||
|
||||
function extractDocumentContent(markdown) {
|
||||
const titleMatch = markdown.match(/^#\s+.+\n+/m);
|
||||
const body = markdown
|
||||
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
|
||||
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
|
||||
.replace(/<br\s*\/?>/gi, ' '); // markup, not document text
|
||||
.replace(/<br\s*\/?>/gi, ' ') // markup, not document text
|
||||
// List markers are drawn by the renderer, not stored as text: the browser
|
||||
// paints them from <ol>/<ul> and Word from numbering.xml, so they are
|
||||
// markup here too and must not be expected in the extracted text.
|
||||
.replace(/^[-*]\s+/gm, '')
|
||||
.replace(/^\d+\.\s+/gm, '');
|
||||
return { body, fields: {} };
|
||||
}
|
||||
|
||||
function words(text) {
|
||||
return new Set((text.normalize('NFC').toLocaleLowerCase().match(/[\p{L}\p{N}]+/gu) ?? []));
|
||||
}
|
||||
|
||||
function compact(text) {
|
||||
return text.normalize('NFC').toLocaleLowerCase().replace(/[^\p{L}\p{N}]/gu, '');
|
||||
}
|
||||
@@ -49,42 +39,98 @@ async function pdfText(pdf) {
|
||||
return stdout;
|
||||
}
|
||||
|
||||
// A .docx is a zip; only one member is needed, so it is read here rather than
|
||||
// taking a dependency for it. Entries are located through the central
|
||||
// directory, then inflated from their own local header.
|
||||
function readZipEntry(buffer, name) {
|
||||
const endOfDirectory = buffer.lastIndexOf(Buffer.from([0x50, 0x4b, 0x05, 0x06]));
|
||||
if (endOfDirectory < 0) throw new Error('not a zip archive');
|
||||
const entries = buffer.readUInt16LE(endOfDirectory + 10);
|
||||
let cursor = buffer.readUInt32LE(endOfDirectory + 16);
|
||||
|
||||
for (let index = 0; index < entries; index += 1) {
|
||||
if (buffer.readUInt32LE(cursor) !== 0x02014b50) throw new Error('damaged central directory');
|
||||
const method = buffer.readUInt16LE(cursor + 10);
|
||||
const compressedSize = buffer.readUInt32LE(cursor + 20);
|
||||
const nameLength = buffer.readUInt16LE(cursor + 28);
|
||||
const extraLength = buffer.readUInt16LE(cursor + 30);
|
||||
const commentLength = buffer.readUInt16LE(cursor + 32);
|
||||
const localOffset = buffer.readUInt32LE(cursor + 42);
|
||||
const entryName = buffer.toString('utf8', cursor + 46, cursor + 46 + nameLength);
|
||||
|
||||
if (entryName === name) {
|
||||
const localName = buffer.readUInt16LE(localOffset + 26);
|
||||
const localExtra = buffer.readUInt16LE(localOffset + 28);
|
||||
const start = localOffset + 30 + localName + localExtra;
|
||||
const data = buffer.subarray(start, start + compressedSize);
|
||||
return method === 0 ? Buffer.from(data) : inflateRawSync(data);
|
||||
}
|
||||
cursor += 46 + nameLength + extraLength + commentLength;
|
||||
}
|
||||
throw new Error(`${name} not found in archive`);
|
||||
}
|
||||
|
||||
async function docxText(docx) {
|
||||
const xml = readZipEntry(await fs.readFile(docx), 'word/document.xml').toString('utf8');
|
||||
// Paragraph and row ends are text boundaries; only <w:t> carries characters.
|
||||
return xml
|
||||
.replace(/<\/w:p>|<\/w:tr>|<w:br[^>]*\/>|<w:tab[^>]*\/>/g, '\n')
|
||||
.replace(/<w:t[^>]*>([^<]*)<\/w:t>/g, '$1')
|
||||
.replace(/<[^>]+>/g, '')
|
||||
.replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>')
|
||||
.replaceAll('"', '"').replaceAll(''', "'");
|
||||
}
|
||||
|
||||
const READERS = { pdf: pdfText, docx: docxText };
|
||||
|
||||
function tally(text) {
|
||||
const counts = new Map();
|
||||
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
||||
return counts;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const [sourceRoot, deliveryRoot] = process.argv.slice(2);
|
||||
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY');
|
||||
const [sourceRoot, deliveryRoot, ...requested] = process.argv.slice(2);
|
||||
if (!sourceRoot || !deliveryRoot) throw new Error('Usage: verify-content.mjs SOURCE_DIRECTORY DELIVERY_DIRECTORY [FORMAT…]');
|
||||
const formats = requested.length ? requested : Object.keys(READERS);
|
||||
const unknown = formats.filter((format) => !READERS[format]);
|
||||
if (unknown.length) throw new Error(`Unknown format(s): ${unknown.join(', ')}`);
|
||||
|
||||
const sources = (await markdownFiles(sourceRoot)).sort();
|
||||
const failures = [];
|
||||
let failed = false;
|
||||
|
||||
for (const format of formats) {
|
||||
const failures = [];
|
||||
for (const source of sources) {
|
||||
const relative = path.relative(sourceRoot, source);
|
||||
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
|
||||
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
|
||||
const rendered = path.join(deliveryRoot, relative.replace(/\.md$/, `.${format}`));
|
||||
const [markdown, extracted] = await Promise.all([
|
||||
fs.readFile(source, 'utf8'),
|
||||
READERS[format](rendered)
|
||||
]);
|
||||
const { body } = extractDocumentContent(markdown);
|
||||
// Thai has no word spaces, so a run can wrap several times inside a narrow
|
||||
// table cell and reach the extracted text as separate fragments. Coverage is
|
||||
// therefore checked by character count: every character of the source must
|
||||
// appear in the PDF at least as many times. Clipped or dropped text loses
|
||||
// glyphs and still fails; re-wrapping does not.
|
||||
const tally = (text) => {
|
||||
const counts = new Map();
|
||||
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
|
||||
return counts;
|
||||
};
|
||||
const rendered = tally(extracted);
|
||||
// appear in the rendering at least as many times. Clipped or dropped text
|
||||
// loses glyphs and still fails; re-wrapping does not.
|
||||
const counts = tally(extracted);
|
||||
const missing = [...tally(body)]
|
||||
.filter(([character, count]) => (rendered.get(character) ?? 0) < count)
|
||||
.map(([character, count]) => `${character}\u00d7${count - (rendered.get(character) ?? 0)}`);
|
||||
.filter(([character, count]) => (counts.get(character) ?? 0) < count)
|
||||
.map(([character, count]) => `${character}×${count - (counts.get(character) ?? 0)}`);
|
||||
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
|
||||
}
|
||||
|
||||
if (failures.length) {
|
||||
console.error(`Content coverage failed for ${failures.length}/${sources.length} PDFs:`);
|
||||
failed = true;
|
||||
console.error(`Content coverage failed for ${failures.length}/${sources.length} ${format.toUpperCase()}s:`);
|
||||
failures.forEach((failure) => console.error(`- ${failure}`));
|
||||
process.exit(1);
|
||||
} else {
|
||||
console.log(`Content coverage passed: ${sources.length}/${sources.length} ${format.toUpperCase()}s.`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Content coverage passed: ${sources.length}/${sources.length} PDFs.`);
|
||||
if (failed) process.exit(1);
|
||||
}
|
||||
|
||||
main().catch((error) => { console.error(error); process.exit(1); });
|
||||
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Reference in New Issue
Block a user