#!/usr/bin/env python3 """Convert the rendered delivery package from PDF to Word. The work products are authored in Markdown and rendered to PDF by render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that PDF, so the Word file keeps the page the reviewer signed off rather than being laid out a second time from the source. The corporate identity is the one thing not taken from the reconstruction. A PDF has no notion of a running header, so the converter rebuilds the CI band and the footer rule as ordinary content at the top and bottom of every page, losing the ribbon geometry and the alignment with it. Both are therefore stripped out again here and replaced by a real Word header and footer: the band is the same assets/header.html the PDF is printed from, captured by render-ci-band.mjs, so it lands in the same place on every page. Requires pdf2docx: pip install -r requirements.txt Usage: convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY] """ import json import re import subprocess import sys import tempfile from pathlib import Path from docx import Document from docx.enum.text import WD_TAB_ALIGNMENT from docx.oxml import OxmlElement from docx.oxml.ns import qn from docx.shared import Mm, Pt, RGBColor from pdf2docx import Converter HERE = Path(__file__).resolve().parent # Page box measured from the audited reference package, as in sdlc-delivery.css. BODY_TOP_MM = 32 FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page BAND_WIDTH_MM = 210 STANDARD = 'ISO/IEC 29110-4-1:2018' RULE_SIZE = '4' # .5pt table rule, as printed RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai" INK = RGBColor(0x11, 0x11, 0x11) # The furniture pdf2docx rebuilds in the body, recognised by text that appears # nowhere in the authored sources. BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD) LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower') def text_of(element): return ''.join(element.itertext()) def is_blank(element): return ( element.tag == qn('w:p') and not text_of(element).strip() and element.find('.//' + qn('w:sectPr')) is None and not element.findall('.//' + qn('w:drawing')) ) def strip_reconstructed_furniture(document): """Remove the per-page letterhead and footer rule pdf2docx puts in the body.""" body = document.element.body removed = 0 for block in list(body): if block.getparent() is None: continue if block.tag == qn('w:tbl'): # The converter sometimes lays a page's furniture and its content # into one table, so furniture is taken out a row at a time. for row in list(block.findall(qn('w:tr'))): if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS): block.remove(row) removed += 1 if block.find(qn('w:tr')) is None: previous = block.getprevious() if previous is not None and is_blank(previous): body.remove(previous) body.remove(block) elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS): body.remove(block) removed += 1 return removed DECIMAL = re.compile(r'^-?\d+\.\d+$') BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'} def normalise_ooxml(document): """Repair the measurements the converter writes outside the schema. Widths, sizes and border weights are whole units in OOXML, and a colour is six hex digits with no leading hash. The converter writes the numbers it measured off the page instead - "5.599999999999909", "#585858" - which Word is free to interpret as it likes, so borders come out at weights nobody asked for. """ fixed = 0 for element in document.element.body.iter(): for name, value in list(element.attrib.items()): local = name.rpartition('}')[2] if DECIMAL.match(value): element.set(name, str(round(float(value)))) fixed += 1 elif local in ('color', 'fill') and value.startswith('#'): element.set(name, value[1:]) fixed += 1 return fixed def match_rule_weight(document): """Draw every table rule at the .5pt the PDF is printed with.""" edges = 0 for borders in document.element.body.iter(): if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')): continue for edge in borders: if edge.tag.rpartition('}')[2] not in BORDER_EDGES: continue if edge.get(qn('w:val')) in (None, 'nil', 'none'): continue edge.set(qn('w:sz'), RULE_SIZE) edge.set(qn('w:color'), RULE_COLOUR) edges += 1 return edges def drop_padding(document): """Remove the blank paragraphs the converter pads each page out with. A run of them at the foot of a page is pure whitespace - the section break already ends the page - and a run between two blocks only ever needs one. """ body = document.element.body children = list(body) removed = 0 index = 0 while index < len(children): if not is_blank(children[index]): index += 1 continue start = index while index < len(children) and is_blank(children[index]): index += 1 following = children[index] if index < len(children) else None ends_page = following is None or ( following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None ) for element in children[start + (0 if ends_page else 1):index]: body.remove(element) removed += 1 return removed R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed' LOGO_MIN_PX = 100 def strip_logo_images(document): """Drop the letterhead logo left loose in the body. The work products carry no figures of their own, but the converter draws each bullet as a small image, so only the logo is removed - anything that size would be furniture, anything smaller is a list marker. """ logos = { relationship_id for relationship_id, part in document.part.related_parts.items() if getattr(part, 'image', None) is not None and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX } removed = 0 for drawing in list(document.element.body.iter(qn('w:drawing'))): blip = drawing.find('.//' + qn('a:blip')) if blip is not None and blip.get(R_EMBED) in logos: drawing.getparent().remove(drawing) removed += 1 return removed def style_run(run, points): run.font.size = Pt(points) run.font.color.rgb = INK fonts = OxmlElement('w:rFonts') for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'): fonts.set(qn(attribute), THAI_FONT) run._element.get_or_add_rPr().insert(0, fonts) def add_top_border(paragraph): borders = OxmlElement('w:pBdr') top = OxmlElement('w:top') top.set(qn('w:val'), 'single') top.set(qn('w:sz'), '6') # .75pt, as printed top.set(qn('w:space'), '1') top.set(qn('w:color'), '111111') borders.append(top) paragraph._p.get_or_add_pPr().append(borders) def install_letterhead(document, band, footer_tag): """Give every page the CI band and footer rule as real Word furniture.""" for index, section in enumerate(document.sections): section.top_margin = Mm(BODY_TOP_MM) section.header_distance = Mm(0) section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM) # Only the first section carries the furniture; the rest inherit it. section.header.is_linked_to_previous = index > 0 section.footer.is_linked_to_previous = index > 0 section = document.sections[0] header = section.header.paragraphs[0] # The band spans the page, so it starts outside the text block. header.paragraph_format.left_indent = -section.left_margin header.paragraph_format.space_before = Pt(0) header.paragraph_format.space_after = Pt(0) header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM)) footer = section.footer.paragraphs[0] footer.paragraph_format.space_before = Pt(0) footer.paragraph_format.space_after = Pt(0) footer.paragraph_format.tab_stops.add_tab_stop( section.page_width - section.left_margin - section.right_margin, WD_TAB_ALIGNMENT.RIGHT ) add_top_border(footer) style_run(footer.add_run(STANDARD), 9) style_run(footer.add_run(f'\t{footer_tag}'), 9) def convert(pdf, band, footer_tag): docx = pdf.with_suffix('.docx') converter = Converter(str(pdf)) try: converter.convert(str(docx)) finally: converter.close() document = Document(str(docx)) strip_reconstructed_furniture(document) strip_logo_images(document) drop_padding(document) normalise_ooxml(document) match_rule_weight(document) install_letterhead(document, band, footer_tag) document.save(str(docx)) return docx def render_bands(source_root, band_root): subprocess.run( ['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)], check=True ) return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8')) def main(argv): if len(argv) < 2: raise SystemExit(__doc__) source_root, delivery_root = Path(argv[0]), Path(argv[1]) temporary = None if len(argv) > 2: bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8')) else: temporary = tempfile.TemporaryDirectory() bands = render_bands(source_root, temporary.name) failures = [] for index, (relative, band) in enumerate(sorted(bands.items()), start=1): pdf = delivery_root / Path(relative).with_suffix('.pdf') try: docx = convert(pdf, Path(band['band']), band['footer']) print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}') except Exception as error: # noqa: BLE001 - reported, not swallowed failures.append((pdf, error)) print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr) if temporary: temporary.cleanup() print(f'{len(bands) - len(failures)}/{len(bands)} documents converted') if failures: raise SystemExit(1) if __name__ == '__main__': main(sys.argv[1:])