#!/usr/bin/env python3 """Convert the rendered delivery package from PDF to Word. The work products are authored in Markdown and rendered to PDF by render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that PDF, so the Word file keeps the page the reviewer signed off rather than being laid out a second time from the source. The corporate identity is the one thing not taken from the reconstruction. A PDF has no notion of a running header, so the converter rebuilds the CI band and the footer rule as ordinary content at the top and bottom of every page, losing the ribbon geometry and the alignment with it. Both are therefore stripped out again here and replaced by a real Word header and footer: the band is the same assets/header.html the PDF is printed from, captured by render-ci-band.mjs, so it lands in the same place on every page. Requires pdf2docx: pip install -r requirements.txt Usage: convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY] """ import json import subprocess import sys import tempfile from pathlib import Path from docx import Document from docx.enum.text import WD_TAB_ALIGNMENT from docx.oxml import OxmlElement from docx.oxml.ns import qn from docx.shared import Mm, Pt, RGBColor from pdf2docx import Converter HERE = Path(__file__).resolve().parent # Page box measured from the audited reference package, as in sdlc-delivery.css. BODY_TOP_MM = 32 FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page BAND_WIDTH_MM = 210 STANDARD = 'ISO/IEC 29110-4-1:2018' THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai" INK = RGBColor(0x11, 0x11, 0x11) # The furniture pdf2docx rebuilds in the body, recognised by text that appears # nowhere in the authored sources. BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD) LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower') def text_of(element): return ''.join(element.itertext()) def is_blank(element): return ( element.tag == qn('w:p') and not text_of(element).strip() and element.find('.//' + qn('w:sectPr')) is None and not element.findall('.//' + qn('w:drawing')) ) def strip_reconstructed_furniture(document): """Remove the per-page letterhead and footer rule pdf2docx puts in the body.""" body = document.element.body removed = 0 for block in list(body): if block.getparent() is None: continue if block.tag == qn('w:tbl'): # The converter sometimes lays a page's furniture and its content # into one table, so furniture is taken out a row at a time. for row in list(block.findall(qn('w:tr'))): if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS): block.remove(row) removed += 1 if block.find(qn('w:tr')) is None: previous = block.getprevious() if previous is not None and is_blank(previous): body.remove(previous) body.remove(block) elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS): body.remove(block) removed += 1 return removed R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed' LOGO_MIN_PX = 100 def strip_logo_images(document): """Drop the letterhead logo left loose in the body. The work products carry no figures of their own, but the converter draws each bullet as a small image, so only the logo is removed - anything that size would be furniture, anything smaller is a list marker. """ logos = { relationship_id for relationship_id, part in document.part.related_parts.items() if getattr(part, 'image', None) is not None and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX } removed = 0 for drawing in list(document.element.body.iter(qn('w:drawing'))): blip = drawing.find('.//' + qn('a:blip')) if blip is not None and blip.get(R_EMBED) in logos: drawing.getparent().remove(drawing) removed += 1 return removed def style_run(run, points): run.font.size = Pt(points) run.font.color.rgb = INK fonts = OxmlElement('w:rFonts') for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'): fonts.set(qn(attribute), THAI_FONT) run._element.get_or_add_rPr().insert(0, fonts) def add_top_border(paragraph): borders = OxmlElement('w:pBdr') top = OxmlElement('w:top') top.set(qn('w:val'), 'single') top.set(qn('w:sz'), '6') # .75pt, as printed top.set(qn('w:space'), '1') top.set(qn('w:color'), '111111') borders.append(top) paragraph._p.get_or_add_pPr().append(borders) def install_letterhead(document, band, footer_tag): """Give every page the CI band and footer rule as real Word furniture.""" for index, section in enumerate(document.sections): section.top_margin = Mm(BODY_TOP_MM) section.header_distance = Mm(0) section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM) # Only the first section carries the furniture; the rest inherit it. section.header.is_linked_to_previous = index > 0 section.footer.is_linked_to_previous = index > 0 section = document.sections[0] header = section.header.paragraphs[0] # The band spans the page, so it starts outside the text block. header.paragraph_format.left_indent = -section.left_margin header.paragraph_format.space_before = Pt(0) header.paragraph_format.space_after = Pt(0) header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM)) footer = section.footer.paragraphs[0] footer.paragraph_format.space_before = Pt(0) footer.paragraph_format.space_after = Pt(0) footer.paragraph_format.tab_stops.add_tab_stop( section.page_width - section.left_margin - section.right_margin, WD_TAB_ALIGNMENT.RIGHT ) add_top_border(footer) style_run(footer.add_run(STANDARD), 9) style_run(footer.add_run(f'\t{footer_tag}'), 9) def convert(pdf, band, footer_tag): docx = pdf.with_suffix('.docx') converter = Converter(str(pdf)) try: converter.convert(str(docx)) finally: converter.close() document = Document(str(docx)) strip_reconstructed_furniture(document) strip_logo_images(document) install_letterhead(document, band, footer_tag) document.save(str(docx)) return docx def render_bands(source_root, band_root): subprocess.run( ['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)], check=True ) return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8')) def main(argv): if len(argv) < 2: raise SystemExit(__doc__) source_root, delivery_root = Path(argv[0]), Path(argv[1]) temporary = None if len(argv) > 2: bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8')) else: temporary = tempfile.TemporaryDirectory() bands = render_bands(source_root, temporary.name) failures = [] for index, (relative, band) in enumerate(sorted(bands.items()), start=1): pdf = delivery_root / Path(relative).with_suffix('.pdf') try: docx = convert(pdf, Path(band['band']), band['footer']) print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}') except Exception as error: # noqa: BLE001 - reported, not swallowed failures.append((pdf, error)) print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr) if temporary: temporary.cleanup() print(f'{len(bands) - len(failures)}/{len(bands)} documents converted') if failures: raise SystemExit(1) if __name__ == '__main__': main(sys.argv[1:])