Files
wms-app/scripts/sdlc-delivery/convert-to-docx.py
T

227 lines
7.8 KiB
Python

#!/usr/bin/env python3
"""Convert the rendered delivery package from PDF to Word.
The work products are authored in Markdown and rendered to PDF by
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
PDF, so the Word file keeps the page the reviewer signed off rather than being
laid out a second time from the source.
The corporate identity is the one thing not taken from the reconstruction.
A PDF has no notion of a running header, so the converter rebuilds the CI band
and the footer rule as ordinary content at the top and bottom of every page,
losing the ribbon geometry and the alignment with it. Both are therefore
stripped out again here and replaced by a real Word header and footer: the band
is the same assets/header.html the PDF is printed from, captured by
render-ci-band.mjs, so it lands in the same place on every page.
Requires pdf2docx: pip install -r requirements.txt
Usage:
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
"""
import json
import subprocess
import sys
import tempfile
from pathlib import Path
from docx import Document
from docx.enum.text import WD_TAB_ALIGNMENT
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import Mm, Pt, RGBColor
from pdf2docx import Converter
HERE = Path(__file__).resolve().parent
# Page box measured from the audited reference package, as in sdlc-delivery.css.
BODY_TOP_MM = 32
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
BAND_WIDTH_MM = 210
STANDARD = 'ISO/IEC 29110-4-1:2018'
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
INK = RGBColor(0x11, 0x11, 0x11)
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
# nowhere in the authored sources.
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
def text_of(element):
return ''.join(element.itertext())
def is_blank(element):
return (
element.tag == qn('w:p')
and not text_of(element).strip()
and element.find('.//' + qn('w:sectPr')) is None
and not element.findall('.//' + qn('w:drawing'))
)
def strip_reconstructed_furniture(document):
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
body = document.element.body
removed = 0
for block in list(body):
if block.getparent() is None:
continue
if block.tag == qn('w:tbl'):
# The converter sometimes lays a page's furniture and its content
# into one table, so furniture is taken out a row at a time.
for row in list(block.findall(qn('w:tr'))):
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
block.remove(row)
removed += 1
if block.find(qn('w:tr')) is None:
previous = block.getprevious()
if previous is not None and is_blank(previous):
body.remove(previous)
body.remove(block)
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
body.remove(block)
removed += 1
return removed
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
LOGO_MIN_PX = 100
def strip_logo_images(document):
"""Drop the letterhead logo left loose in the body.
The work products carry no figures of their own, but the converter draws
each bullet as a small image, so only the logo is removed - anything that
size would be furniture, anything smaller is a list marker.
"""
logos = {
relationship_id
for relationship_id, part in document.part.related_parts.items()
if getattr(part, 'image', None) is not None
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
}
removed = 0
for drawing in list(document.element.body.iter(qn('w:drawing'))):
blip = drawing.find('.//' + qn('a:blip'))
if blip is not None and blip.get(R_EMBED) in logos:
drawing.getparent().remove(drawing)
removed += 1
return removed
def style_run(run, points):
run.font.size = Pt(points)
run.font.color.rgb = INK
fonts = OxmlElement('w:rFonts')
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), THAI_FONT)
run._element.get_or_add_rPr().insert(0, fonts)
def add_top_border(paragraph):
borders = OxmlElement('w:pBdr')
top = OxmlElement('w:top')
top.set(qn('w:val'), 'single')
top.set(qn('w:sz'), '6') # .75pt, as printed
top.set(qn('w:space'), '1')
top.set(qn('w:color'), '111111')
borders.append(top)
paragraph._p.get_or_add_pPr().append(borders)
def install_letterhead(document, band, footer_tag):
"""Give every page the CI band and footer rule as real Word furniture."""
for index, section in enumerate(document.sections):
section.top_margin = Mm(BODY_TOP_MM)
section.header_distance = Mm(0)
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
# Only the first section carries the furniture; the rest inherit it.
section.header.is_linked_to_previous = index > 0
section.footer.is_linked_to_previous = index > 0
section = document.sections[0]
header = section.header.paragraphs[0]
# The band spans the page, so it starts outside the text block.
header.paragraph_format.left_indent = -section.left_margin
header.paragraph_format.space_before = Pt(0)
header.paragraph_format.space_after = Pt(0)
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
footer = section.footer.paragraphs[0]
footer.paragraph_format.space_before = Pt(0)
footer.paragraph_format.space_after = Pt(0)
footer.paragraph_format.tab_stops.add_tab_stop(
section.page_width - section.left_margin - section.right_margin,
WD_TAB_ALIGNMENT.RIGHT
)
add_top_border(footer)
style_run(footer.add_run(STANDARD), 9)
style_run(footer.add_run(f'\t{footer_tag}'), 9)
def convert(pdf, band, footer_tag):
docx = pdf.with_suffix('.docx')
converter = Converter(str(pdf))
try:
converter.convert(str(docx))
finally:
converter.close()
document = Document(str(docx))
strip_reconstructed_furniture(document)
strip_logo_images(document)
install_letterhead(document, band, footer_tag)
document.save(str(docx))
return docx
def render_bands(source_root, band_root):
subprocess.run(
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
check=True
)
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
def main(argv):
if len(argv) < 2:
raise SystemExit(__doc__)
source_root, delivery_root = Path(argv[0]), Path(argv[1])
temporary = None
if len(argv) > 2:
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
else:
temporary = tempfile.TemporaryDirectory()
bands = render_bands(source_root, temporary.name)
failures = []
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
pdf = delivery_root / Path(relative).with_suffix('.pdf')
try:
docx = convert(pdf, Path(band['band']), band['footer'])
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
except Exception as error: # noqa: BLE001 - reported, not swallowed
failures.append((pdf, error))
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
if temporary:
temporary.cleanup()
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
if failures:
raise SystemExit(1)
if __name__ == '__main__':
main(sys.argv[1:])