Give the Word package a real page header for the CI band

This commit is contained in:
Thanakorn
2026-09-11 17:11:49 +07:00
parent 75ff043036
commit 7473115fe2
61 changed files with 264 additions and 15 deletions
+184 -15
View File
@@ -6,52 +6,221 @@ render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
PDF, so the Word file keeps the page the reviewer signed off rather than being
laid out a second time from the source.
The corporate identity is the one thing not taken from the reconstruction.
A PDF has no notion of a running header, so the converter rebuilds the CI band
and the footer rule as ordinary content at the top and bottom of every page,
losing the ribbon geometry and the alignment with it. Both are therefore
stripped out again here and replaced by a real Word header and footer: the band
is the same assets/header.html the PDF is printed from, captured by
render-ci-band.mjs, so it lands in the same place on every page.
Requires pdf2docx: pip install -r requirements.txt
Usage:
convert-to-docx.py DELIVERY_DIRECTORY [PDF…]
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
"""
import json
import subprocess
import sys
import os
import tempfile
from pathlib import Path
from docx import Document
from docx.enum.text import WD_TAB_ALIGNMENT
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import Mm, Pt, RGBColor
from pdf2docx import Converter
HERE = Path(__file__).resolve().parent
def convert(pdf: Path) -> Path:
# Page box measured from the audited reference package, as in sdlc-delivery.css.
BODY_TOP_MM = 32
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
BAND_WIDTH_MM = 210
STANDARD = 'ISO/IEC 29110-4-1:2018'
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
INK = RGBColor(0x11, 0x11, 0x11)
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
# nowhere in the authored sources.
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
def text_of(element):
return ''.join(element.itertext())
def is_blank(element):
return (
element.tag == qn('w:p')
and not text_of(element).strip()
and element.find('.//' + qn('w:sectPr')) is None
and not element.findall('.//' + qn('w:drawing'))
)
def strip_reconstructed_furniture(document):
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
body = document.element.body
removed = 0
for block in list(body):
if block.getparent() is None:
continue
if block.tag == qn('w:tbl'):
# The converter sometimes lays a page's furniture and its content
# into one table, so furniture is taken out a row at a time.
for row in list(block.findall(qn('w:tr'))):
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
block.remove(row)
removed += 1
if block.find(qn('w:tr')) is None:
previous = block.getprevious()
if previous is not None and is_blank(previous):
body.remove(previous)
body.remove(block)
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
body.remove(block)
removed += 1
return removed
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
LOGO_MIN_PX = 100
def strip_logo_images(document):
"""Drop the letterhead logo left loose in the body.
The work products carry no figures of their own, but the converter draws
each bullet as a small image, so only the logo is removed - anything that
size would be furniture, anything smaller is a list marker.
"""
logos = {
relationship_id
for relationship_id, part in document.part.related_parts.items()
if getattr(part, 'image', None) is not None
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
}
removed = 0
for drawing in list(document.element.body.iter(qn('w:drawing'))):
blip = drawing.find('.//' + qn('a:blip'))
if blip is not None and blip.get(R_EMBED) in logos:
drawing.getparent().remove(drawing)
removed += 1
return removed
def style_run(run, points):
run.font.size = Pt(points)
run.font.color.rgb = INK
fonts = OxmlElement('w:rFonts')
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), THAI_FONT)
run._element.get_or_add_rPr().insert(0, fonts)
def add_top_border(paragraph):
borders = OxmlElement('w:pBdr')
top = OxmlElement('w:top')
top.set(qn('w:val'), 'single')
top.set(qn('w:sz'), '6') # .75pt, as printed
top.set(qn('w:space'), '1')
top.set(qn('w:color'), '111111')
borders.append(top)
paragraph._p.get_or_add_pPr().append(borders)
def install_letterhead(document, band, footer_tag):
"""Give every page the CI band and footer rule as real Word furniture."""
for index, section in enumerate(document.sections):
section.top_margin = Mm(BODY_TOP_MM)
section.header_distance = Mm(0)
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
# Only the first section carries the furniture; the rest inherit it.
section.header.is_linked_to_previous = index > 0
section.footer.is_linked_to_previous = index > 0
section = document.sections[0]
header = section.header.paragraphs[0]
# The band spans the page, so it starts outside the text block.
header.paragraph_format.left_indent = -section.left_margin
header.paragraph_format.space_before = Pt(0)
header.paragraph_format.space_after = Pt(0)
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
footer = section.footer.paragraphs[0]
footer.paragraph_format.space_before = Pt(0)
footer.paragraph_format.space_after = Pt(0)
footer.paragraph_format.tab_stops.add_tab_stop(
section.page_width - section.left_margin - section.right_margin,
WD_TAB_ALIGNMENT.RIGHT
)
add_top_border(footer)
style_run(footer.add_run(STANDARD), 9)
style_run(footer.add_run(f'\t{footer_tag}'), 9)
def convert(pdf, band, footer_tag):
docx = pdf.with_suffix('.docx')
converter = Converter(str(pdf))
try:
converter.convert(str(docx))
finally:
converter.close()
document = Document(str(docx))
strip_reconstructed_furniture(document)
strip_logo_images(document)
install_letterhead(document, band, footer_tag)
document.save(str(docx))
return docx
def render_bands(source_root, band_root):
subprocess.run(
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
check=True
)
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
def main(argv):
if not argv:
if len(argv) < 2:
raise SystemExit(__doc__)
root = Path(argv[0])
pdfs = [Path(p) for p in argv[1:]] or sorted(root.rglob('*.pdf'))
if not pdfs:
raise SystemExit(f'No PDFs found under {root}')
source_root, delivery_root = Path(argv[0]), Path(argv[1])
temporary = None
if len(argv) > 2:
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
else:
temporary = tempfile.TemporaryDirectory()
bands = render_bands(source_root, temporary.name)
failures = []
for index, pdf in enumerate(pdfs, start=1):
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
pdf = delivery_root / Path(relative).with_suffix('.pdf')
try:
docx = convert(pdf)
print(f'[{index}/{len(pdfs)}] {docx.relative_to(root)}')
docx = convert(pdf, Path(band['band']), band['footer'])
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
except Exception as error: # noqa: BLE001 - reported, not swallowed
failures.append((pdf, error))
print(f'[{index}/{len(pdfs)}] FAILED {pdf.relative_to(root)}: {error}', file=sys.stderr)
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
print(f'{len(pdfs) - len(failures)}/{len(pdfs)} documents converted')
if temporary:
temporary.cleanup()
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
if failures:
raise SystemExit(1)
if __name__ == '__main__':
# pdf2docx is chatty; the per-document progress above is enough.
os.environ.setdefault('PDF2DOCX_QUIET', '1')
main(sys.argv[1:])
+80
View File
@@ -0,0 +1,80 @@
// Render the corporate identity band used by the delivery package.
//
// The band is the same assets/header.html the PDF is printed with, captured as
// an image so its ribbon geometry survives into Word, where it is placed in a
// real page header instead of being reconstructed in the body.
import fs from 'node:fs/promises';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { chromium } from 'playwright';
const here = path.dirname(fileURLToPath(import.meta.url));
const root = path.resolve(here, '../..');
const assets = path.join(here, 'assets');
export const BAND_MM = { width: 210, height: 27 };
const SCALE = 3; // the band is drawn at 3x so it stays crisp in print
const applyTemplate = (template, values) => template.replace(/{{(\w+)}}/g, (_, key) => values[key] ?? '');
const slug = (title) => title.replace(/[^\w-]+/g, '-').replace(/^-|-$/g, '').toLowerCase() || 'band';
async function markdownFiles(directory) {
const entries = await fs.readdir(directory, { withFileTypes: true });
const results = await Promise.all(entries.map(async (entry) => {
const target = path.join(directory, entry.name);
if (entry.isDirectory()) return markdownFiles(target);
return entry.isFile() && entry.name.endsWith('.md') ? [target] : [];
}));
return results.flat();
}
async function main() {
const [sourceRoot, outputRoot] = process.argv.slice(2);
if (!sourceRoot || !outputRoot) throw new Error('Usage: render-ci-band.mjs SOURCE_DIRECTORY OUTPUT_DIRECTORY');
const [template, logo] = await Promise.all([
fs.readFile(path.join(assets, 'header.html'), 'utf8'),
fs.readFile(path.join(root, 'app/assets/images/logo.svg'))
]);
const values = {
logoPath: `data:image/svg+xml;base64,${logo.toString('base64')}`,
ciBlue: '#0797d5',
ciGrey: '#8e8e8e' // ribbon grey measured from the audited reference documents
};
await fs.mkdir(outputRoot, { recursive: true });
const sources = (await markdownFiles(sourceRoot)).sort();
const bands = {};
const rendered = new Map();
const browser = await chromium.launch({ headless: true });
try {
const page = await browser.newPage({ deviceScaleFactor: SCALE });
for (const source of sources) {
const markdown = await fs.readFile(source, 'utf8');
const title = markdown.match(/^#\s+(.+)$/m)?.[1]?.trim() ?? 'BRN WMS';
const footer = markdown.match(/^<!--\s*footer:\s*(.+?)\s*-->\s*$/m)?.[1] ?? '';
if (!rendered.has(title)) {
const file = path.join(outputRoot, `${slug(title)}.png`);
await page.setContent(
'<!doctype html><meta charset="utf-8"><body style="margin:0;background:#fff">' +
`<div id="band" style="width:${BAND_MM.width}mm;height:${BAND_MM.height}mm;overflow:hidden;position:relative">` +
`${applyTemplate(template, { ...values, documentTitle: title })}</div></body>`,
{ waitUntil: 'networkidle' }
);
await page.locator('#band').screenshot({ type: 'png', path: file });
rendered.set(title, file);
}
bands[path.relative(sourceRoot, source)] = { title, footer, band: rendered.get(title) };
}
} finally {
await browser.close();
}
await fs.writeFile(path.join(outputRoot, 'bands.json'), `${JSON.stringify(bands, null, 2)}\n`);
console.log(`${rendered.size} band${rendered.size === 1 ? '' : 's'} rendered for ${sources.length} documents`);
}
main().catch((error) => { console.error(error); process.exit(1); });