Files
wms-app/scripts/sdlc-delivery/convert-to-docx.py
T

855 lines
33 KiB
Python

#!/usr/bin/env python3
"""Convert the rendered delivery package from PDF to Word.
The work products are authored in Markdown and rendered to PDF by
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
PDF, so the Word file keeps the page the reviewer signed off rather than being
laid out a second time from the source.
The corporate identity is the one thing not taken from the reconstruction.
A PDF has no notion of a running header, so the converter rebuilds the CI band
and the footer rule as ordinary content at the top and bottom of every page,
losing the ribbon geometry and the alignment with it. Both are therefore
stripped out again here and replaced by a real Word header and footer: the band
is the same assets/header.html the PDF is printed from, captured by
render-ci-band.mjs, so it lands in the same place on every page.
Requires pdf2docx: pip install -r requirements.txt
Usage:
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
"""
import copy
import json
import re
import subprocess
import sys
import tempfile
import unicodedata
from pathlib import Path
from docx import Document
from docx.enum.text import WD_TAB_ALIGNMENT
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import Mm, Pt, RGBColor, Twips
from docx.text.paragraph import Paragraph
from pdf2docx import Converter
HERE = Path(__file__).resolve().parent
# Page box measured from the audited reference package, as in sdlc-delivery.css.
BODY_TOP_MM = 32
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
BAND_WIDTH_MM = 210
STANDARD = 'ISO/IEC 29110-4-1:2018'
RULE_SIZE = '4' # .5pt table rule, as printed
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
MONO_FONT = 'Courier New' # code spans; DejaVu Sans Mono in the PDF
INK = RGBColor(0x11, 0x11, 0x11)
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
# nowhere in the authored sources.
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
def text_of(element):
"""The text Word shows for an element: its w:t runs, tabs and breaks.
Not itertext(): the converter leaves other text-bearing nodes in its runs,
which would read every paragraph twice over.
"""
parts = []
for node in element.iter(qn('w:t'), qn('w:tab'), qn('w:br')):
if node.tag == qn('w:t'):
parts.append(node.text or '')
elif node.getparent().tag == qn('w:r'):
# A w:tab under w:tabs is a tab-stop definition, not a character.
parts.append('\t' if node.tag == qn('w:tab') else '\n')
return ''.join(parts)
def is_blank(element):
return (
element.tag == qn('w:p')
and not text_of(element).strip()
and element.find('.//' + qn('w:sectPr')) is None
and not element.findall('.//' + qn('w:drawing'))
)
def strip_reconstructed_furniture(document):
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
body = document.element.body
removed = 0
for block in list(body):
if block.getparent() is None:
continue
if block.tag == qn('w:tbl'):
# The converter sometimes lays a page's furniture and its content
# into one table, so furniture is taken out a row at a time.
for row in list(block.findall(qn('w:tr'))):
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
block.remove(row)
removed += 1
if block.find(qn('w:tr')) is None:
previous = block.getprevious()
if previous is not None and is_blank(previous):
body.remove(previous)
body.remove(block)
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
body.remove(block)
removed += 1
return removed
DECIMAL = re.compile(r'^-?\d+\.\d+$')
BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'}
def normalise_ooxml(document):
"""Repair the measurements the converter writes outside the schema.
Widths, sizes and border weights are whole units in OOXML, and a colour is
six hex digits with no leading hash. The converter writes the numbers it
measured off the page instead - "5.599999999999909", "#585858" - which Word
is free to interpret as it likes, so borders come out at weights nobody
asked for.
"""
fixed = 0
for element in document.element.body.iter():
for name, value in list(element.attrib.items()):
local = name.rpartition('}')[2]
if DECIMAL.match(value):
element.set(name, str(round(float(value))))
fixed += 1
elif local in ('color', 'fill') and value.startswith('#'):
element.set(name, value[1:])
fixed += 1
return fixed
def match_rule_weight(document):
"""Draw every table rule at the .5pt the PDF is printed with."""
edges = 0
for borders in document.element.body.iter():
if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')):
continue
for edge in borders:
if edge.tag.rpartition('}')[2] not in BORDER_EDGES:
continue
if edge.get(qn('w:val')) in (None, 'nil', 'none'):
continue
edge.set(qn('w:sz'), RULE_SIZE)
edge.set(qn('w:color'), RULE_COLOUR)
edges += 1
return edges
def drop_padding(document):
"""Remove the blank paragraphs the converter pads each page out with.
A run of them at the foot of a page is pure whitespace - the section break
already ends the page - and a run between two blocks only ever needs one.
"""
body = document.element.body
children = list(body)
removed = 0
index = 0
while index < len(children):
if not is_blank(children[index]):
index += 1
continue
start = index
while index < len(children) and is_blank(children[index]):
index += 1
following = children[index] if index < len(children) else None
ends_page = following is None or (
following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None
)
for element in children[start + (0 if ends_page else 1):index]:
body.remove(element)
removed += 1
return removed
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
LOGO_MIN_PX = 100
def strip_logo_images(document):
"""Drop the letterhead logo left loose in the body.
The work products carry no figures of their own, but the converter draws
each bullet as a small image, so only the logo is removed - anything that
size would be furniture, anything smaller is a list marker.
"""
logos = {
relationship_id
for relationship_id, part in document.part.related_parts.items()
if getattr(part, 'image', None) is not None
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
}
removed = 0
for drawing in list(document.element.body.iter(qn('w:drawing'))):
blip = drawing.find('.//' + qn('a:blip'))
if blip is not None and blip.get(R_EMBED) in logos:
drawing.getparent().remove(drawing)
removed += 1
return removed
# ---------------------------------------------------------------------------
# Fonts
# ---------------------------------------------------------------------------
def normalise_fonts(document):
"""Point every run at a face Word actually has.
The PDF embeds its faces as Type3 fonts, so the converter names runs after
the PDF objects - "Type3 (4 0 R)" - and gives them no complex-script face
or size. Word then sets Latin text in a fallback serif and Thai text at the
default size, which is the mixed, jumping type the conversion showed.
"""
for run in document.element.body.iter(qn('w:r')):
properties = run.find(qn('w:rPr'))
if properties is None:
properties = OxmlElement('w:rPr')
run.insert(0, properties)
fonts = properties.find(qn('w:rFonts'))
face = MONO_FONT if fonts is not None and 'Mono' in (fonts.get(qn('w:ascii')) or '') else THAI_FONT
if fonts is None:
fonts = OxmlElement('w:rFonts')
style = properties.find(qn('w:rStyle'))
properties.insert(1 if style is not None else 0, fonts)
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), face)
size = properties.find(qn('w:sz'))
if size is not None and properties.find(qn('w:szCs')) is None:
complex_size = OxmlElement('w:szCs')
complex_size.set(qn('w:val'), size.get(qn('w:val')))
size.addnext(complex_size)
# ---------------------------------------------------------------------------
# Bullets
# ---------------------------------------------------------------------------
BULLET = '•'
BULLET_TEXT_MM = 7 # ul padding-left in sdlc-delivery.css
BULLET_HANG_MM = 4
def text_block_left_mm(section):
# The PDF text block starts 30mm in, or 15mm on the landscape work products.
return 15 if section.page_width > section.page_height else 30
def unwrap_list_tables(document):
"""Turn the tables the converter builds out of bullet lists back into text.
A short list is sometimes laid out as a grid - bullet image, item, empty
cell - with stray bottom borders that print as loose rules. Each row goes
back into the body as a paragraph of its own.
"""
body = document.element.body
items = []
for table in [child for child in body if child.tag == qn('w:tbl')]:
rows = table.findall(qn('w:tr'))
entries = []
for row in rows:
cells = row.findall(qn('w:tc'))
worded = [cell for cell in cells if text_of(cell).strip()]
marked = [cell for cell in cells if cell.findall('.//' + qn('w:drawing')) and not text_of(cell).strip()]
if len(worded) != 1 or not marked:
entries = None
break
entries.append(worded[0])
if not rows or entries is None:
continue
for cell in entries:
for paragraph in cell.findall(qn('w:p')):
if text_of(paragraph).strip():
table.addprevious(paragraph)
items.append(paragraph)
body.remove(table)
return items
def take_bullet_images(document):
"""Collect the list items drawn with an image marker, dropping the image."""
items = []
for paragraph in [child for child in document.element.body if child.tag == qn('w:p')]:
drawings = paragraph.findall('.//' + qn('w:drawing'))
if not drawings or not text_of(paragraph).strip():
continue
for drawing in drawings:
run = drawing.getparent()
run.remove(drawing)
if run.tag == qn('w:r') and not text_of(run).strip():
run.getparent().remove(run)
items.append(paragraph)
return items
def make_bullet(document, paragraph):
section = document.sections[0]
item = Paragraph(paragraph, document._body)
text_mm = text_block_left_mm(section) + BULLET_TEXT_MM
item.paragraph_format.left_indent = Twips(round(Mm(text_mm).twips - section.left_margin.twips))
item.paragraph_format.first_line_indent = -Mm(BULLET_HANG_MM)
first = paragraph.find(qn('w:r'))
marker = OxmlElement('w:r')
if first is not None and first.find(qn('w:rPr')) is not None:
marker.append(copy.deepcopy(first.find(qn('w:rPr'))))
text = OxmlElement('w:t')
text.text = BULLET
marker.append(text)
marker.append(OxmlElement('w:tab'))
properties = paragraph.find(qn('w:pPr'))
if properties is not None:
properties.addnext(marker)
else:
paragraph.insert(0, marker)
# ---------------------------------------------------------------------------
# Text from the source
# ---------------------------------------------------------------------------
APPROVAL_FIELD = re.compile(r'^(Name|Role|Signature|Date|Position|Company|Project roles|Decision):')
SEPARATOR_CELL = re.compile(r'^:?-{3,}:?$')
INLINE = re.compile(r'\*\*(.+?)\*\*|`([^`]+)`|\\([\\*_`~])')
def letters(text):
"""The characters two renderings of the same text must share."""
return ''.join(
character for character in unicodedata.normalize('NFC', text)
if unicodedata.category(character)[0] in 'LMN'
).lower()
def read_source(markdown):
"""Blocks and table cells of a work product, as the reader sees them."""
body = re.sub(r'^#\s+.+\n+', '', markdown, count=1, flags=re.M)
body = re.sub(r'^<!--.*?-->\s*$', '', body, flags=re.M)
blocks, cells, tables, paragraph = [], [], [], []
bullets = set()
table = None
def flush():
if paragraph:
joined = ''
for index, line in enumerate(paragraph):
hard = re.search(r'\s{2,}$', line) or APPROVAL_FIELD.match(line.strip())
joined += line.strip() + ('\n' if hard else (' ' if index < len(paragraph) - 1 else ''))
blocks.append(joined.rstrip('\n'))
paragraph.clear()
for line in body.splitlines():
stripped = line.strip()
if stripped.startswith('|') and stripped.endswith('|'):
flush()
parts = [part.strip() for part in stripped[1:-1].split('|')]
if table is None:
table = {'after': blocks[-1] if blocks else None, 'rows': []}
tables.append(table)
if not all(SEPARATOR_CELL.match(part) for part in parts):
row = [re.sub(r'<br\s*/?>', '\n', part) for part in parts]
cells.extend(row)
table['rows'].append(row)
continue
table = None
if not stripped:
flush()
continue
heading = re.match(r'^#{1,3}\s+(.+)$', stripped)
bullet = re.match(r'^[-*]\s+(.+)$', stripped)
ordered = re.match(r'^\d+\.\s+.+$', stripped)
if heading or bullet or ordered:
flush()
# A numbered item keeps its number: the PDF prints it as text.
blocks.append(heading.group(1) if heading else bullet.group(1) if bullet else stripped)
if bullet:
bullets.add(len(blocks) - 1)
continue
paragraph.append(line)
flush()
return blocks, cells, tables, bullets
def index_texts(texts):
exact, by_length = {}, {}
for text in texts:
key = letters(text)
if key and key not in exact:
exact[key] = text
by_length.setdefault(len(key), []).append(key)
return exact, by_length
def is_subsequence(short, long):
remaining = iter(long)
return all(character in remaining for character in short)
def source_for(text, exact, by_length):
"""The authored text a converted fragment came from, if it is unambiguous.
Same letters means only spacing and punctuation differ. Failing that, a
fragment that is the source with up to three characters missing - the
converter drops characters, it never invents them - is accepted when just
one source text fits.
"""
key = letters(text)
if not key:
return None
if key in exact:
return exact[key]
if len(key) < 8:
return None
fits = {
exact[candidate]
for length in range(len(key) + 1, len(key) + 4)
for candidate in by_length.get(length, ())
if is_subsequence(key, candidate)
}
return fits.pop() if len(fits) == 1 else None
def write_text(paragraph, text):
"""Replace a paragraph's runs with the authored text, keeping its look."""
template = paragraph.find('.//' + qn('w:rPr'))
template = copy.deepcopy(template) if template is not None else OxmlElement('w:rPr')
for child in list(paragraph):
if child.tag != qn('w:pPr'):
paragraph.remove(child)
def run(content=None, bold=False, code=False, line_break=False):
element = OxmlElement('w:r')
properties = copy.deepcopy(template)
fonts = properties.find(qn('w:rFonts'))
if fonts is not None:
face = MONO_FONT if code else THAI_FONT
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), face)
weight = properties.find(qn('w:b'))
for complex_weight in properties.findall(qn('w:bCs')):
properties.remove(complex_weight)
if bold:
if weight is None:
weight = OxmlElement('w:b')
(fonts.addnext(weight) if fonts is not None else properties.insert(0, weight))
weight.attrib.pop(qn('w:val'), None)
weight.addnext(OxmlElement('w:bCs'))
elif weight is not None:
weight.set(qn('w:val'), '0')
element.append(properties)
if line_break:
element.append(OxmlElement('w:br'))
else:
node = OxmlElement('w:t')
node.text = content
node.set('{http://www.w3.org/XML/1998/namespace}space', 'preserve')
element.append(node)
paragraph.append(element)
for number, line in enumerate(text.split('\n')):
if number:
run(line_break=True)
cursor = 0
for match in INLINE.finditer(line):
if match.start() > cursor:
run(line[cursor:match.start()])
if match.group(1) is not None:
run(match.group(1), bold=True)
elif match.group(2) is not None:
run(match.group(2), code=True)
else:
run(match.group(3))
cursor = match.end()
if cursor < len(line):
run(line[cursor:])
def split_blocks(key, block_keys):
"""The run of consecutive source blocks a converted paragraph is made of."""
for start, first in enumerate(block_keys):
if not first or not key.startswith(first):
continue
combined, end = first, start
while len(combined) < len(key) and end + 1 < len(block_keys) and end - start < 7:
end += 1
combined += block_keys[end]
if combined == key:
return list(range(start, end + 1))
if not key.startswith(combined):
break
return None
def restore_text(document, markdown, bullets):
"""Put the authored wording back where the converter mangled it.
PDF text has no word spaces for Thai, so the conversion runs Thai words
together, breaks a paragraph into one paragraph per printed line, and now
and then drops a character. Wherever a converted paragraph or table cell
can be traced to exactly one piece of the source, its text is rewritten
from the source; anything that cannot be traced is left as converted.
"""
blocks, cells, _, bullet_blocks = read_source(markdown)
block_keys = [letters(block) for block in blocks]
known = {id(paragraph) for paragraph in bullets}
block_exact, block_lengths = index_texts(blocks)
cell_exact, cell_lengths = index_texts(cells)
body = document.element.body
rewritten = merged = 0
for cell in body.iter(qn('w:tc')):
if cell.find('.//' + qn('w:tbl')) is not None:
continue
paragraphs = cell.findall(qn('w:p'))
text = '\n'.join(text_of(paragraph) for paragraph in paragraphs)
source = source_for(text, cell_exact, cell_lengths)
if source is None or '___' in source or source == text:
continue
write_text(paragraphs[0], source)
for extra in paragraphs[1:]:
cell.remove(extra)
rewritten += 1
children = [child for child in body if child.tag == qn('w:p') and text_of(child).strip()]
index = 0
while index < len(children):
paragraph = children[index]
source = source_for(text_of(paragraph), block_exact, block_lengths)
span = 1
if source is None:
# A paragraph the converter split into one paragraph per line.
combined = text_of(paragraph)
follower = paragraph
for count in range(2, 9):
follower = follower.getnext()
if follower is None or follower.tag != qn('w:p') or not text_of(follower).strip():
break
combined += text_of(follower)
if letters(combined) in block_exact:
source, span = block_exact[letters(combined)], count
break
if source is None:
# Several list items or paragraphs the converter ran into one.
parts = split_blocks(letters(text_of(paragraph)), block_keys)
if parts and not any('___' in blocks[part] for part in parts):
previous = paragraph
for number, part in enumerate(parts):
target = paragraph
if number:
target = OxmlElement('w:p')
if paragraph.find(qn('w:pPr')) is not None:
target.append(copy.deepcopy(paragraph.find(qn('w:pPr'))))
template = paragraph.find('.//' + qn('w:rPr'))
run = OxmlElement('w:r')
run.append(copy.deepcopy(template) if template is not None else OxmlElement('w:rPr'))
target.append(run)
previous.addnext(target)
previous = target
write_text(target, blocks[part])
if part in bullet_blocks and id(target) not in known:
bullets.append(target)
known.add(id(target))
rewritten += len(parts)
index += 1
continue
if source is not None and '___' not in source:
for extra in children[index + 1:index + span]:
body.remove(extra)
merged += 1
write_text(paragraph, source)
rewritten += 1
index += span
return rewritten, merged
TINT = 'FFF3F6' # header-row tint used across the reference tables
TABLE_PT = 9
CELL_MARGIN_MM = (1.6, 2) # vertical, horizontal padding of a table cell
PRINTED_TEXT_WIDTH_MM = {False: 160, True: 267} # portrait, landscape
DISTINCTIVE_LETTERS = 15 # shorter cells - labels, IDs, dates - recur elsewhere
DROPPED_CHARACTER_SLACK = 3
def present(key, document_key):
"""Whether a cell's whole text reached the document.
The whole cell, not a fragment of it: short Thai phrases recur across a
work product, so a partial match proves nothing. Its two halves in order,
a few characters apart at most, still count, as the converter sometimes
drops a character mid-cell.
"""
if key in document_key:
return True
half = len(key) // 2
start = document_key.find(key[:half])
while start >= 0:
second = document_key.find(key[half:], start + half)
if 0 <= second - (start + half) <= DROPPED_CHARACTER_SLACK:
return True
start = document_key.find(key[:half], start + 1)
return False
def dropped_tables(document, markdown):
"""Source tables of which not one distinctive cell reached the document."""
_, _, tables, _ = read_source(markdown)
document_key = letters(text_of(document.element.body))
missing = []
for table in tables:
keys = [letters(cell) for row in table['rows'] for cell in row]
keys = [key for key in keys if len(key) >= DISTINCTIVE_LETTERS]
if keys and not any(present(key, document_key) for key in keys):
missing.append(table)
return missing
def build_table(document, rows):
section = document.sections[0]
landscape = section.page_width > section.page_height
width = Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips
columns = max(len(row) for row in rows)
longest = [max(len(letters(row[i])) if i < len(row) else 0 for row in rows) for i in range(columns)]
weights = [min(max(value, 8), 80) for value in longest]
widths = [round(width * weight / sum(weights)) for weight in weights]
def element(tag, **attributes):
node = OxmlElement(tag)
for name, value in attributes.items():
node.set(qn(name), str(value))
return node
table = element('w:tbl')
properties = element('w:tblPr')
properties.append(element('w:tblW', **{'w:w': width, 'w:type': 'dxa'}))
indent = round(Mm(text_block_left_mm(section)).twips - section.left_margin.twips)
properties.append(element('w:tblInd', **{'w:w': indent, 'w:type': 'dxa'}))
borders = element('w:tblBorders')
for edge in ('top', 'left', 'bottom', 'right', 'insideH', 'insideV'):
borders.append(element('w:' + edge, **{'w:val': 'single', 'w:sz': RULE_SIZE, 'w:space': 0, 'w:color': RULE_COLOUR}))
properties.append(borders)
properties.append(element('w:tblLayout', **{'w:type': 'fixed'}))
margins = element('w:tblCellMar')
vertical, horizontal = (round(Mm(value).twips) for value in CELL_MARGIN_MM)
for edge, value in (('top', vertical), ('left', horizontal), ('bottom', vertical), ('right', horizontal)):
margins.append(element('w:' + edge, **{'w:w': value, 'w:type': 'dxa'}))
properties.append(margins)
table.append(properties)
grid = element('w:tblGrid')
for column_width in widths:
grid.append(element('w:gridCol', **{'w:w': column_width}))
table.append(grid)
for number, cells in enumerate(rows):
row = element('w:tr')
row_properties = element('w:trPr')
row_properties.append(element('w:cantSplit'))
if number == 0:
row_properties.append(element('w:tblHeader'))
row.append(row_properties)
for column in range(columns):
cell = element('w:tc')
cell_properties = element('w:tcPr')
cell_properties.append(element('w:tcW', **{'w:w': widths[column], 'w:type': 'dxa'}))
if number == 0:
cell_properties.append(element('w:shd', **{'w:val': 'clear', 'w:color': 'auto', 'w:fill': TINT}))
cell.append(cell_properties)
paragraph = element('w:p')
paragraph_properties = element('w:pPr')
paragraph_properties.append(element('w:spacing', **{'w:before': 0, 'w:after': 0}))
paragraph.append(paragraph_properties)
template = element('w:r')
run_properties = element('w:rPr')
fonts = element('w:rFonts', **{'w:ascii': THAI_FONT, 'w:hAnsi': THAI_FONT, 'w:cs': THAI_FONT, 'w:eastAsia': THAI_FONT})
run_properties.append(fonts)
run_properties.append(element('w:color', **{'w:val': '111111'}))
run_properties.append(element('w:sz', **{'w:val': TABLE_PT * 2}))
run_properties.append(element('w:szCs', **{'w:val': TABLE_PT * 2}))
template.append(run_properties)
paragraph.append(template)
write_text(paragraph, cells[column] if column < len(cells) else '')
cell.append(paragraph)
row.append(cell)
table.append(row)
return table
def restore_dropped_tables(document, markdown):
"""Put back tables the converter left out, where the source places them.
The converter occasionally drops a whole table that is plainly printed in
the PDF. It is rebuilt from the source in the style of the converted
tables, after the paragraph it follows in the source.
"""
restored = 0
body = document.element.body
for table in dropped_tables(document, markdown):
if not table['after'] or not table['rows']:
continue
anchor_key = letters(table['after'])
paragraphs = [child for child in body if child.tag == qn('w:p')]
anchor = next((child for child in paragraphs if letters(text_of(child)) == anchor_key), None)
if anchor is None:
# The block may share a paragraph with what precedes it.
endings = [child for child in paragraphs if anchor_key and letters(text_of(child)).endswith(anchor_key)]
anchor = endings[0] if len(endings) == 1 else None
if anchor is None:
continue
anchor.addnext(build_table(document, table['rows']))
restored += 1
return restored
def style_run(run, points):
run.font.size = Pt(points)
run.font.color.rgb = INK
fonts = OxmlElement('w:rFonts')
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
fonts.set(qn(attribute), THAI_FONT)
run._element.get_or_add_rPr().insert(0, fonts)
def drop_style(paragraph):
# The template's Header and Footer styles carry centre and right tab stops
# of their own, which catch the tab before the footer tag and centre it.
properties = paragraph._p.pPr
if properties is not None and properties.find(qn('w:pStyle')) is not None:
properties.remove(properties.find(qn('w:pStyle')))
def add_top_border(paragraph):
borders = OxmlElement('w:pBdr')
top = OxmlElement('w:top')
top.set(qn('w:val'), 'single')
top.set(qn('w:sz'), '6') # .75pt, as printed
top.set(qn('w:space'), '1')
top.set(qn('w:color'), '111111')
borders.append(top)
paragraph._p.get_or_add_pPr().append(borders)
def install_letterhead(document, band, footer_tag):
"""Give every page the CI band and footer rule as real Word furniture."""
for index, section in enumerate(document.sections):
section.top_margin = Mm(BODY_TOP_MM)
section.header_distance = Mm(0)
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
# Only the first section carries the furniture; the rest inherit it.
section.header.is_linked_to_previous = index > 0
section.footer.is_linked_to_previous = index > 0
section = document.sections[0]
header = section.header.paragraphs[0]
drop_style(header)
# The band spans the page, so it starts outside the text block.
header.paragraph_format.left_indent = -section.left_margin
header.paragraph_format.space_before = Pt(0)
header.paragraph_format.space_after = Pt(0)
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
footer = section.footer.paragraphs[0]
drop_style(footer)
footer.paragraph_format.space_before = Pt(0)
footer.paragraph_format.space_after = Pt(0)
footer.paragraph_format.tab_stops.add_tab_stop(
section.page_width - section.left_margin - section.right_margin,
WD_TAB_ALIGNMENT.RIGHT
)
add_top_border(footer)
style_run(footer.add_run(STANDARD), 9)
style_run(footer.add_run(f'\t{footer_tag}'), 9)
def convert(pdf, source, band, footer_tag):
docx = pdf.with_suffix('.docx')
converter = Converter(str(pdf))
try:
converter.convert(str(docx))
finally:
converter.close()
document = Document(str(docx))
strip_reconstructed_furniture(document)
strip_logo_images(document)
bullets = unwrap_list_tables(document) + take_bullet_images(document)
normalise_ooxml(document)
normalise_fonts(document)
markdown = source.read_text(encoding='utf8')
restore_text(document, markdown, bullets)
restore_dropped_tables(document, markdown)
for paragraph in bullets:
if paragraph.getparent() is not None:
make_bullet(document, paragraph)
drop_padding(document)
match_rule_weight(document)
install_letterhead(document, band, footer_tag)
document.save(str(docx))
return docx
def render_bands(source_root, band_root):
subprocess.run(
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
check=True
)
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
def main(argv):
if len(argv) < 2:
raise SystemExit(__doc__)
source_root, delivery_root = Path(argv[0]), Path(argv[1])
temporary = None
if len(argv) > 2:
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
else:
temporary = tempfile.TemporaryDirectory()
bands = render_bands(source_root, temporary.name)
failures = []
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
pdf = delivery_root / Path(relative).with_suffix('.pdf')
try:
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'])
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
except Exception as error: # noqa: BLE001 - reported, not swallowed
failures.append((pdf, error))
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
if temporary:
temporary.cleanup()
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
if failures:
raise SystemExit(1)
if __name__ == '__main__':
main(sys.argv[1:])