305 lines
10 KiB
Python
305 lines
10 KiB
Python
#!/usr/bin/env python3
|
|
"""Convert the rendered delivery package from PDF to Word.
|
|
|
|
The work products are authored in Markdown and rendered to PDF by
|
|
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
|
|
PDF, so the Word file keeps the page the reviewer signed off rather than being
|
|
laid out a second time from the source.
|
|
|
|
The corporate identity is the one thing not taken from the reconstruction.
|
|
A PDF has no notion of a running header, so the converter rebuilds the CI band
|
|
and the footer rule as ordinary content at the top and bottom of every page,
|
|
losing the ribbon geometry and the alignment with it. Both are therefore
|
|
stripped out again here and replaced by a real Word header and footer: the band
|
|
is the same assets/header.html the PDF is printed from, captured by
|
|
render-ci-band.mjs, so it lands in the same place on every page.
|
|
|
|
Requires pdf2docx: pip install -r requirements.txt
|
|
|
|
Usage:
|
|
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
|
|
"""
|
|
import json
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
from pathlib import Path
|
|
|
|
from docx import Document
|
|
from docx.enum.text import WD_TAB_ALIGNMENT
|
|
from docx.oxml import OxmlElement
|
|
from docx.oxml.ns import qn
|
|
from docx.shared import Mm, Pt, RGBColor
|
|
from pdf2docx import Converter
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
|
|
# Page box measured from the audited reference package, as in sdlc-delivery.css.
|
|
BODY_TOP_MM = 32
|
|
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
|
|
BAND_WIDTH_MM = 210
|
|
STANDARD = 'ISO/IEC 29110-4-1:2018'
|
|
RULE_SIZE = '4' # .5pt table rule, as printed
|
|
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
|
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
|
|
INK = RGBColor(0x11, 0x11, 0x11)
|
|
|
|
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
|
|
# nowhere in the authored sources.
|
|
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
|
|
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
|
|
|
|
|
|
def text_of(element):
|
|
return ''.join(element.itertext())
|
|
|
|
|
|
def is_blank(element):
|
|
return (
|
|
element.tag == qn('w:p')
|
|
and not text_of(element).strip()
|
|
and element.find('.//' + qn('w:sectPr')) is None
|
|
and not element.findall('.//' + qn('w:drawing'))
|
|
)
|
|
|
|
|
|
def strip_reconstructed_furniture(document):
|
|
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
|
|
body = document.element.body
|
|
removed = 0
|
|
|
|
for block in list(body):
|
|
if block.getparent() is None:
|
|
continue
|
|
|
|
if block.tag == qn('w:tbl'):
|
|
# The converter sometimes lays a page's furniture and its content
|
|
# into one table, so furniture is taken out a row at a time.
|
|
for row in list(block.findall(qn('w:tr'))):
|
|
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
|
|
block.remove(row)
|
|
removed += 1
|
|
if block.find(qn('w:tr')) is None:
|
|
previous = block.getprevious()
|
|
if previous is not None and is_blank(previous):
|
|
body.remove(previous)
|
|
body.remove(block)
|
|
|
|
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
|
|
body.remove(block)
|
|
removed += 1
|
|
|
|
return removed
|
|
|
|
|
|
DECIMAL = re.compile(r'^-?\d+\.\d+$')
|
|
BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'}
|
|
|
|
|
|
def normalise_ooxml(document):
|
|
"""Repair the measurements the converter writes outside the schema.
|
|
|
|
Widths, sizes and border weights are whole units in OOXML, and a colour is
|
|
six hex digits with no leading hash. The converter writes the numbers it
|
|
measured off the page instead - "5.599999999999909", "#585858" - which Word
|
|
is free to interpret as it likes, so borders come out at weights nobody
|
|
asked for.
|
|
"""
|
|
fixed = 0
|
|
for element in document.element.body.iter():
|
|
for name, value in list(element.attrib.items()):
|
|
local = name.rpartition('}')[2]
|
|
if DECIMAL.match(value):
|
|
element.set(name, str(round(float(value))))
|
|
fixed += 1
|
|
elif local in ('color', 'fill') and value.startswith('#'):
|
|
element.set(name, value[1:])
|
|
fixed += 1
|
|
return fixed
|
|
|
|
|
|
def match_rule_weight(document):
|
|
"""Draw every table rule at the .5pt the PDF is printed with."""
|
|
edges = 0
|
|
for borders in document.element.body.iter():
|
|
if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')):
|
|
continue
|
|
for edge in borders:
|
|
if edge.tag.rpartition('}')[2] not in BORDER_EDGES:
|
|
continue
|
|
if edge.get(qn('w:val')) in (None, 'nil', 'none'):
|
|
continue
|
|
edge.set(qn('w:sz'), RULE_SIZE)
|
|
edge.set(qn('w:color'), RULE_COLOUR)
|
|
edges += 1
|
|
return edges
|
|
|
|
|
|
def drop_padding(document):
|
|
"""Remove the blank paragraphs the converter pads each page out with.
|
|
|
|
A run of them at the foot of a page is pure whitespace - the section break
|
|
already ends the page - and a run between two blocks only ever needs one.
|
|
"""
|
|
body = document.element.body
|
|
children = list(body)
|
|
removed = 0
|
|
index = 0
|
|
|
|
while index < len(children):
|
|
if not is_blank(children[index]):
|
|
index += 1
|
|
continue
|
|
start = index
|
|
while index < len(children) and is_blank(children[index]):
|
|
index += 1
|
|
following = children[index] if index < len(children) else None
|
|
ends_page = following is None or (
|
|
following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None
|
|
)
|
|
for element in children[start + (0 if ends_page else 1):index]:
|
|
body.remove(element)
|
|
removed += 1
|
|
|
|
return removed
|
|
|
|
|
|
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
|
|
LOGO_MIN_PX = 100
|
|
|
|
|
|
def strip_logo_images(document):
|
|
"""Drop the letterhead logo left loose in the body.
|
|
|
|
The work products carry no figures of their own, but the converter draws
|
|
each bullet as a small image, so only the logo is removed - anything that
|
|
size would be furniture, anything smaller is a list marker.
|
|
"""
|
|
logos = {
|
|
relationship_id
|
|
for relationship_id, part in document.part.related_parts.items()
|
|
if getattr(part, 'image', None) is not None
|
|
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
|
|
}
|
|
|
|
removed = 0
|
|
for drawing in list(document.element.body.iter(qn('w:drawing'))):
|
|
blip = drawing.find('.//' + qn('a:blip'))
|
|
if blip is not None and blip.get(R_EMBED) in logos:
|
|
drawing.getparent().remove(drawing)
|
|
removed += 1
|
|
return removed
|
|
|
|
|
|
def style_run(run, points):
|
|
run.font.size = Pt(points)
|
|
run.font.color.rgb = INK
|
|
fonts = OxmlElement('w:rFonts')
|
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
|
fonts.set(qn(attribute), THAI_FONT)
|
|
run._element.get_or_add_rPr().insert(0, fonts)
|
|
|
|
|
|
def add_top_border(paragraph):
|
|
borders = OxmlElement('w:pBdr')
|
|
top = OxmlElement('w:top')
|
|
top.set(qn('w:val'), 'single')
|
|
top.set(qn('w:sz'), '6') # .75pt, as printed
|
|
top.set(qn('w:space'), '1')
|
|
top.set(qn('w:color'), '111111')
|
|
borders.append(top)
|
|
paragraph._p.get_or_add_pPr().append(borders)
|
|
|
|
|
|
def install_letterhead(document, band, footer_tag):
|
|
"""Give every page the CI band and footer rule as real Word furniture."""
|
|
for index, section in enumerate(document.sections):
|
|
section.top_margin = Mm(BODY_TOP_MM)
|
|
section.header_distance = Mm(0)
|
|
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
|
|
# Only the first section carries the furniture; the rest inherit it.
|
|
section.header.is_linked_to_previous = index > 0
|
|
section.footer.is_linked_to_previous = index > 0
|
|
|
|
section = document.sections[0]
|
|
|
|
header = section.header.paragraphs[0]
|
|
# The band spans the page, so it starts outside the text block.
|
|
header.paragraph_format.left_indent = -section.left_margin
|
|
header.paragraph_format.space_before = Pt(0)
|
|
header.paragraph_format.space_after = Pt(0)
|
|
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
|
|
|
|
footer = section.footer.paragraphs[0]
|
|
footer.paragraph_format.space_before = Pt(0)
|
|
footer.paragraph_format.space_after = Pt(0)
|
|
footer.paragraph_format.tab_stops.add_tab_stop(
|
|
section.page_width - section.left_margin - section.right_margin,
|
|
WD_TAB_ALIGNMENT.RIGHT
|
|
)
|
|
add_top_border(footer)
|
|
style_run(footer.add_run(STANDARD), 9)
|
|
style_run(footer.add_run(f'\t{footer_tag}'), 9)
|
|
|
|
|
|
def convert(pdf, band, footer_tag):
|
|
docx = pdf.with_suffix('.docx')
|
|
converter = Converter(str(pdf))
|
|
try:
|
|
converter.convert(str(docx))
|
|
finally:
|
|
converter.close()
|
|
|
|
document = Document(str(docx))
|
|
strip_reconstructed_furniture(document)
|
|
strip_logo_images(document)
|
|
drop_padding(document)
|
|
normalise_ooxml(document)
|
|
match_rule_weight(document)
|
|
install_letterhead(document, band, footer_tag)
|
|
document.save(str(docx))
|
|
return docx
|
|
|
|
|
|
def render_bands(source_root, band_root):
|
|
subprocess.run(
|
|
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
|
|
check=True
|
|
)
|
|
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
|
|
|
|
|
|
def main(argv):
|
|
if len(argv) < 2:
|
|
raise SystemExit(__doc__)
|
|
|
|
source_root, delivery_root = Path(argv[0]), Path(argv[1])
|
|
temporary = None
|
|
if len(argv) > 2:
|
|
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
|
|
else:
|
|
temporary = tempfile.TemporaryDirectory()
|
|
bands = render_bands(source_root, temporary.name)
|
|
|
|
failures = []
|
|
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
|
|
pdf = delivery_root / Path(relative).with_suffix('.pdf')
|
|
try:
|
|
docx = convert(pdf, Path(band['band']), band['footer'])
|
|
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
|
|
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
|
failures.append((pdf, error))
|
|
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
|
|
|
|
if temporary:
|
|
temporary.cleanup()
|
|
|
|
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
|
|
if failures:
|
|
raise SystemExit(1)
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main(sys.argv[1:])
|