Files
wms-app/scripts/sdlc-delivery/convert-to-docx.py
T

305 lines
10 KiB
Python

#!/usr/bin/env python3
"""Convert the rendered delivery package from PDF to Word.
The work products are authored in Markdown and rendered to PDF by
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
PDF, so the Word file keeps the page the reviewer signed off rather than being
laid out a second time from the source.
The corporate identity is the one thing not taken from the reconstruction.
A PDF has no notion of a running header, so the converter rebuilds the CI band
and the footer rule as ordinary content at the top and bottom of every page,
losing the ribbon geometry and the alignment with it. Both are therefore
stripped out again here and replaced by a real Word header and footer: the band
is the same assets/header.html the PDF is printed from, captured by
render-ci-band.mjs, so it lands in the same place on every page.
Requires pdf2docx: pip install -r requirements.txt
Usage:
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
"""
import json
import re
import subprocess
import sys
import tempfile
from pathlib import Path
from docx import Document
from docx.enum.text import WD_TAB_ALIGNMENT
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import Mm, Pt, RGBColor
from pdf2docx import Converter
HERE = Path(__file__).resolve().parent
# Page box measured from the audited reference package, as in sdlc-delivery.css.
BODY_TOP_MM = 32
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
BAND_WIDTH_MM = 210
STANDARD = 'ISO/IEC 29110-4-1:2018'
RULE_SIZE = '4' # .5pt table rule, as printed
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
INK = RGBColor(0x11, 0x11, 0x11)
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
# nowhere in the authored sources.
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
def text_of(element):
return ''.join(element.itertext())
def is_blank(element):
return (
element.tag == qn('w:p')
and not text_of(element).strip()
and element.find('.//' + qn('w:sectPr')) is None
and not element.findall('.//' + qn('w:drawing'))
)
def strip_reconstructed_furniture(document):
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
body = document.element.body
removed = 0
for block in list(body):
if block.getparent() is None:
continue
if block.tag == qn('w:tbl'):
# The converter sometimes lays a page's furniture and its content
# into one table, so furniture is taken out a row at a time.
for row in list(block.findall(qn('w:tr'))):
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
block.remove(row)
removed += 1
if block.find(qn('w:tr')) is None:
previous = block.getprevious()
if previous is not None and is_blank(previous):
body.remove(previous)
body.remove(block)
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
body.remove(block)
removed += 1
return removed
DECIMAL = re.compile(r'^-?\d+\.\d+$')
BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'}
def normalise_ooxml(document):
"""Repair the measurements the converter writes outside the schema.
Widths, sizes and border weights are whole units in OOXML, and a colour is
six hex digits with no leading hash. The converter writes the numbers it
measured off the page instead - "5.599999999999909", "#585858" - which Word
is free to interpret as it likes, so borders come out at weights nobody
asked for.
"""
fixed = 0
for element in document.element.body.iter():
for name, value in list(element.attrib.items()):
local = name.rpartition('}')[2]
if DECIMAL.match(value):
element.set(name, str(round(float(value))))
fixed += 1
elif local in ('color', 'fill') and value.startswith('#'):
element.set(name, value[1:])
fixed += 1
return fixed
def match_rule_weight(document):
"""Draw every table rule at the .5pt the PDF is printed with."""
edges = 0
for borders in document.element.body.iter():
if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')):
continue
for edge in borders:
if edge.tag.rpartition('}')[2] not in BORDER_EDGES:
continue
if edge.get(qn('w:val')) in (None, 'nil', 'none'):
continue
edge.set(qn('w:sz'), RULE_SIZE)
edge.set(qn('w:color'), RULE_COLOUR)
edges += 1
return edges
def drop_padding(document):
"""Remove the blank paragraphs the converter pads each page out with.
A run of them at the foot of a page is pure whitespace - the section break
already ends the page - and a run between two blocks only ever needs one.
"""
body = document.element.body
children = list(body)
removed = 0
index = 0
while index < len(children):
if not is_blank(children[index]):
index += 1
continue
start = index
while index < len(children) and is_blank(children[index]):
index += 1
following = children[index] if index < len(children) else None
ends_page = following is None or (
following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None
)
for element in children[start + (0 if ends_page else 1):index]:
body.remove(element)
removed += 1
return removed
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
LOGO_MIN_PX = 100
def strip_logo_images(document):
"""Drop the letterhead logo left loose in the body.
The work products carry no figures of their own, but the converter draws
each bullet as a small image, so only the logo is removed - anything that
size would be furniture, anything smaller is a list marker.
"""
logos = {
relationship_id
for relationship_id, part in document.part.related_parts.items()
if getattr(part, 'image', None) is not None
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
}
removed = 0
for drawing in list(document.element.body.iter(qn('w:drawing'))):
blip = drawing.find('.//' + qn('a:blip'))
if blip is not None and blip.get(R_EMBED) in logos:
drawing.getparent().remove(drawing)
removed += 1
return removed
def style_run(run, points):
run.font.size = Pt(points)
run.font.color.rgb = INK
fonts = OxmlElement('w:rFonts')
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), THAI_FONT)
run._element.get_or_add_rPr().insert(0, fonts)
def add_top_border(paragraph):
borders = OxmlElement('w:pBdr')
top = OxmlElement('w:top')
top.set(qn('w:val'), 'single')
top.set(qn('w:sz'), '6') # .75pt, as printed
top.set(qn('w:space'), '1')
top.set(qn('w:color'), '111111')
borders.append(top)
paragraph._p.get_or_add_pPr().append(borders)
def install_letterhead(document, band, footer_tag):
"""Give every page the CI band and footer rule as real Word furniture."""
for index, section in enumerate(document.sections):
section.top_margin = Mm(BODY_TOP_MM)
section.header_distance = Mm(0)
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
# Only the first section carries the furniture; the rest inherit it.
section.header.is_linked_to_previous = index > 0
section.footer.is_linked_to_previous = index > 0
section = document.sections[0]
header = section.header.paragraphs[0]
# The band spans the page, so it starts outside the text block.
header.paragraph_format.left_indent = -section.left_margin
header.paragraph_format.space_before = Pt(0)
header.paragraph_format.space_after = Pt(0)
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
footer = section.footer.paragraphs[0]
footer.paragraph_format.space_before = Pt(0)
footer.paragraph_format.space_after = Pt(0)
footer.paragraph_format.tab_stops.add_tab_stop(
section.page_width - section.left_margin - section.right_margin,
WD_TAB_ALIGNMENT.RIGHT
)
add_top_border(footer)
style_run(footer.add_run(STANDARD), 9)
style_run(footer.add_run(f'\t{footer_tag}'), 9)
def convert(pdf, band, footer_tag):
docx = pdf.with_suffix('.docx')
converter = Converter(str(pdf))
try:
converter.convert(str(docx))
finally:
converter.close()
document = Document(str(docx))
strip_reconstructed_furniture(document)
strip_logo_images(document)
drop_padding(document)
normalise_ooxml(document)
match_rule_weight(document)
install_letterhead(document, band, footer_tag)
document.save(str(docx))
return docx
def render_bands(source_root, band_root):
subprocess.run(
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
check=True
)
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
def main(argv):
if len(argv) < 2:
raise SystemExit(__doc__)
source_root, delivery_root = Path(argv[0]), Path(argv[1])
temporary = None
if len(argv) > 2:
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
else:
temporary = tempfile.TemporaryDirectory()
bands = render_bands(source_root, temporary.name)
failures = []
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
pdf = delivery_root / Path(relative).with_suffix('.pdf')
try:
docx = convert(pdf, Path(band['band']), band['footer'])
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
except Exception as error: # noqa: BLE001 - reported, not swallowed
failures.append((pdf, error))
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
if temporary:
temporary.cleanup()
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
if failures:
raise SystemExit(1)
if __name__ == '__main__':
main(sys.argv[1:])