855 lines
33 KiB
Python
855 lines
33 KiB
Python
#!/usr/bin/env python3
|
|
"""Convert the rendered delivery package from PDF to Word.
|
|
|
|
The work products are authored in Markdown and rendered to PDF by
|
|
render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that
|
|
PDF, so the Word file keeps the page the reviewer signed off rather than being
|
|
laid out a second time from the source.
|
|
|
|
The corporate identity is the one thing not taken from the reconstruction.
|
|
A PDF has no notion of a running header, so the converter rebuilds the CI band
|
|
and the footer rule as ordinary content at the top and bottom of every page,
|
|
losing the ribbon geometry and the alignment with it. Both are therefore
|
|
stripped out again here and replaced by a real Word header and footer: the band
|
|
is the same assets/header.html the PDF is printed from, captured by
|
|
render-ci-band.mjs, so it lands in the same place on every page.
|
|
|
|
Requires pdf2docx: pip install -r requirements.txt
|
|
|
|
Usage:
|
|
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
|
|
"""
|
|
import copy
|
|
import json
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
import unicodedata
|
|
from pathlib import Path
|
|
|
|
from docx import Document
|
|
from docx.enum.text import WD_TAB_ALIGNMENT
|
|
from docx.oxml import OxmlElement
|
|
from docx.oxml.ns import qn
|
|
from docx.shared import Mm, Pt, RGBColor, Twips
|
|
from docx.text.paragraph import Paragraph
|
|
from pdf2docx import Converter
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
|
|
# Page box measured from the audited reference package, as in sdlc-delivery.css.
|
|
BODY_TOP_MM = 32
|
|
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
|
|
BAND_WIDTH_MM = 210
|
|
STANDARD = 'ISO/IEC 29110-4-1:2018'
|
|
RULE_SIZE = '4' # .5pt table rule, as printed
|
|
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
|
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
|
|
MONO_FONT = 'Courier New' # code spans; DejaVu Sans Mono in the PDF
|
|
INK = RGBColor(0x11, 0x11, 0x11)
|
|
|
|
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
|
|
# nowhere in the authored sources.
|
|
BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD)
|
|
LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
|
|
|
|
|
|
def text_of(element):
|
|
"""The text Word shows for an element: its w:t runs, tabs and breaks.
|
|
|
|
Not itertext(): the converter leaves other text-bearing nodes in its runs,
|
|
which would read every paragraph twice over.
|
|
"""
|
|
parts = []
|
|
for node in element.iter(qn('w:t'), qn('w:tab'), qn('w:br')):
|
|
if node.tag == qn('w:t'):
|
|
parts.append(node.text or '')
|
|
elif node.getparent().tag == qn('w:r'):
|
|
# A w:tab under w:tabs is a tab-stop definition, not a character.
|
|
parts.append('\t' if node.tag == qn('w:tab') else '\n')
|
|
return ''.join(parts)
|
|
|
|
|
|
def is_blank(element):
|
|
return (
|
|
element.tag == qn('w:p')
|
|
and not text_of(element).strip()
|
|
and element.find('.//' + qn('w:sectPr')) is None
|
|
and not element.findall('.//' + qn('w:drawing'))
|
|
)
|
|
|
|
|
|
def strip_reconstructed_furniture(document):
|
|
"""Remove the per-page letterhead and footer rule pdf2docx puts in the body."""
|
|
body = document.element.body
|
|
removed = 0
|
|
|
|
for block in list(body):
|
|
if block.getparent() is None:
|
|
continue
|
|
|
|
if block.tag == qn('w:tbl'):
|
|
# The converter sometimes lays a page's furniture and its content
|
|
# into one table, so furniture is taken out a row at a time.
|
|
for row in list(block.findall(qn('w:tr'))):
|
|
if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS):
|
|
block.remove(row)
|
|
removed += 1
|
|
if block.find(qn('w:tr')) is None:
|
|
previous = block.getprevious()
|
|
if previous is not None and is_blank(previous):
|
|
body.remove(previous)
|
|
body.remove(block)
|
|
|
|
elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS):
|
|
body.remove(block)
|
|
removed += 1
|
|
|
|
return removed
|
|
|
|
|
|
DECIMAL = re.compile(r'^-?\d+\.\d+$')
|
|
BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'}
|
|
|
|
|
|
def normalise_ooxml(document):
|
|
"""Repair the measurements the converter writes outside the schema.
|
|
|
|
Widths, sizes and border weights are whole units in OOXML, and a colour is
|
|
six hex digits with no leading hash. The converter writes the numbers it
|
|
measured off the page instead - "5.599999999999909", "#585858" - which Word
|
|
is free to interpret as it likes, so borders come out at weights nobody
|
|
asked for.
|
|
"""
|
|
fixed = 0
|
|
for element in document.element.body.iter():
|
|
for name, value in list(element.attrib.items()):
|
|
local = name.rpartition('}')[2]
|
|
if DECIMAL.match(value):
|
|
element.set(name, str(round(float(value))))
|
|
fixed += 1
|
|
elif local in ('color', 'fill') and value.startswith('#'):
|
|
element.set(name, value[1:])
|
|
fixed += 1
|
|
return fixed
|
|
|
|
|
|
def match_rule_weight(document):
|
|
"""Draw every table rule at the .5pt the PDF is printed with."""
|
|
edges = 0
|
|
for borders in document.element.body.iter():
|
|
if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')):
|
|
continue
|
|
for edge in borders:
|
|
if edge.tag.rpartition('}')[2] not in BORDER_EDGES:
|
|
continue
|
|
if edge.get(qn('w:val')) in (None, 'nil', 'none'):
|
|
continue
|
|
edge.set(qn('w:sz'), RULE_SIZE)
|
|
edge.set(qn('w:color'), RULE_COLOUR)
|
|
edges += 1
|
|
return edges
|
|
|
|
|
|
def drop_padding(document):
|
|
"""Remove the blank paragraphs the converter pads each page out with.
|
|
|
|
A run of them at the foot of a page is pure whitespace - the section break
|
|
already ends the page - and a run between two blocks only ever needs one.
|
|
"""
|
|
body = document.element.body
|
|
children = list(body)
|
|
removed = 0
|
|
index = 0
|
|
|
|
while index < len(children):
|
|
if not is_blank(children[index]):
|
|
index += 1
|
|
continue
|
|
start = index
|
|
while index < len(children) and is_blank(children[index]):
|
|
index += 1
|
|
following = children[index] if index < len(children) else None
|
|
ends_page = following is None or (
|
|
following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None
|
|
)
|
|
for element in children[start + (0 if ends_page else 1):index]:
|
|
body.remove(element)
|
|
removed += 1
|
|
|
|
return removed
|
|
|
|
|
|
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
|
|
LOGO_MIN_PX = 100
|
|
|
|
|
|
def strip_logo_images(document):
|
|
"""Drop the letterhead logo left loose in the body.
|
|
|
|
The work products carry no figures of their own, but the converter draws
|
|
each bullet as a small image, so only the logo is removed - anything that
|
|
size would be furniture, anything smaller is a list marker.
|
|
"""
|
|
logos = {
|
|
relationship_id
|
|
for relationship_id, part in document.part.related_parts.items()
|
|
if getattr(part, 'image', None) is not None
|
|
and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX
|
|
}
|
|
|
|
removed = 0
|
|
for drawing in list(document.element.body.iter(qn('w:drawing'))):
|
|
blip = drawing.find('.//' + qn('a:blip'))
|
|
if blip is not None and blip.get(R_EMBED) in logos:
|
|
drawing.getparent().remove(drawing)
|
|
removed += 1
|
|
return removed
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Fonts
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def normalise_fonts(document):
|
|
"""Point every run at a face Word actually has.
|
|
|
|
The PDF embeds its faces as Type3 fonts, so the converter names runs after
|
|
the PDF objects - "Type3 (4 0 R)" - and gives them no complex-script face
|
|
or size. Word then sets Latin text in a fallback serif and Thai text at the
|
|
default size, which is the mixed, jumping type the conversion showed.
|
|
"""
|
|
for run in document.element.body.iter(qn('w:r')):
|
|
properties = run.find(qn('w:rPr'))
|
|
if properties is None:
|
|
properties = OxmlElement('w:rPr')
|
|
run.insert(0, properties)
|
|
|
|
fonts = properties.find(qn('w:rFonts'))
|
|
face = MONO_FONT if fonts is not None and 'Mono' in (fonts.get(qn('w:ascii')) or '') else THAI_FONT
|
|
if fonts is None:
|
|
fonts = OxmlElement('w:rFonts')
|
|
style = properties.find(qn('w:rStyle'))
|
|
properties.insert(1 if style is not None else 0, fonts)
|
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
|
fonts.set(qn(attribute), face)
|
|
|
|
size = properties.find(qn('w:sz'))
|
|
if size is not None and properties.find(qn('w:szCs')) is None:
|
|
complex_size = OxmlElement('w:szCs')
|
|
complex_size.set(qn('w:val'), size.get(qn('w:val')))
|
|
size.addnext(complex_size)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Bullets
|
|
# ---------------------------------------------------------------------------
|
|
|
|
BULLET = '•'
|
|
BULLET_TEXT_MM = 7 # ul padding-left in sdlc-delivery.css
|
|
BULLET_HANG_MM = 4
|
|
|
|
|
|
def text_block_left_mm(section):
|
|
# The PDF text block starts 30mm in, or 15mm on the landscape work products.
|
|
return 15 if section.page_width > section.page_height else 30
|
|
|
|
|
|
def unwrap_list_tables(document):
|
|
"""Turn the tables the converter builds out of bullet lists back into text.
|
|
|
|
A short list is sometimes laid out as a grid - bullet image, item, empty
|
|
cell - with stray bottom borders that print as loose rules. Each row goes
|
|
back into the body as a paragraph of its own.
|
|
"""
|
|
body = document.element.body
|
|
items = []
|
|
for table in [child for child in body if child.tag == qn('w:tbl')]:
|
|
rows = table.findall(qn('w:tr'))
|
|
entries = []
|
|
for row in rows:
|
|
cells = row.findall(qn('w:tc'))
|
|
worded = [cell for cell in cells if text_of(cell).strip()]
|
|
marked = [cell for cell in cells if cell.findall('.//' + qn('w:drawing')) and not text_of(cell).strip()]
|
|
if len(worded) != 1 or not marked:
|
|
entries = None
|
|
break
|
|
entries.append(worded[0])
|
|
if not rows or entries is None:
|
|
continue
|
|
for cell in entries:
|
|
for paragraph in cell.findall(qn('w:p')):
|
|
if text_of(paragraph).strip():
|
|
table.addprevious(paragraph)
|
|
items.append(paragraph)
|
|
body.remove(table)
|
|
return items
|
|
|
|
|
|
def take_bullet_images(document):
|
|
"""Collect the list items drawn with an image marker, dropping the image."""
|
|
items = []
|
|
for paragraph in [child for child in document.element.body if child.tag == qn('w:p')]:
|
|
drawings = paragraph.findall('.//' + qn('w:drawing'))
|
|
if not drawings or not text_of(paragraph).strip():
|
|
continue
|
|
for drawing in drawings:
|
|
run = drawing.getparent()
|
|
run.remove(drawing)
|
|
if run.tag == qn('w:r') and not text_of(run).strip():
|
|
run.getparent().remove(run)
|
|
items.append(paragraph)
|
|
return items
|
|
|
|
|
|
def make_bullet(document, paragraph):
|
|
section = document.sections[0]
|
|
item = Paragraph(paragraph, document._body)
|
|
text_mm = text_block_left_mm(section) + BULLET_TEXT_MM
|
|
item.paragraph_format.left_indent = Twips(round(Mm(text_mm).twips - section.left_margin.twips))
|
|
item.paragraph_format.first_line_indent = -Mm(BULLET_HANG_MM)
|
|
|
|
first = paragraph.find(qn('w:r'))
|
|
marker = OxmlElement('w:r')
|
|
if first is not None and first.find(qn('w:rPr')) is not None:
|
|
marker.append(copy.deepcopy(first.find(qn('w:rPr'))))
|
|
text = OxmlElement('w:t')
|
|
text.text = BULLET
|
|
marker.append(text)
|
|
marker.append(OxmlElement('w:tab'))
|
|
properties = paragraph.find(qn('w:pPr'))
|
|
if properties is not None:
|
|
properties.addnext(marker)
|
|
else:
|
|
paragraph.insert(0, marker)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Text from the source
|
|
# ---------------------------------------------------------------------------
|
|
|
|
APPROVAL_FIELD = re.compile(r'^(Name|Role|Signature|Date|Position|Company|Project roles|Decision):')
|
|
SEPARATOR_CELL = re.compile(r'^:?-{3,}:?$')
|
|
INLINE = re.compile(r'\*\*(.+?)\*\*|`([^`]+)`|\\([\\*_`~])')
|
|
|
|
|
|
def letters(text):
|
|
"""The characters two renderings of the same text must share."""
|
|
return ''.join(
|
|
character for character in unicodedata.normalize('NFC', text)
|
|
if unicodedata.category(character)[0] in 'LMN'
|
|
).lower()
|
|
|
|
|
|
def read_source(markdown):
|
|
"""Blocks and table cells of a work product, as the reader sees them."""
|
|
body = re.sub(r'^#\s+.+\n+', '', markdown, count=1, flags=re.M)
|
|
body = re.sub(r'^<!--.*?-->\s*$', '', body, flags=re.M)
|
|
blocks, cells, tables, paragraph = [], [], [], []
|
|
bullets = set()
|
|
table = None
|
|
|
|
def flush():
|
|
if paragraph:
|
|
joined = ''
|
|
for index, line in enumerate(paragraph):
|
|
hard = re.search(r'\s{2,}$', line) or APPROVAL_FIELD.match(line.strip())
|
|
joined += line.strip() + ('\n' if hard else (' ' if index < len(paragraph) - 1 else ''))
|
|
blocks.append(joined.rstrip('\n'))
|
|
paragraph.clear()
|
|
|
|
for line in body.splitlines():
|
|
stripped = line.strip()
|
|
if stripped.startswith('|') and stripped.endswith('|'):
|
|
flush()
|
|
parts = [part.strip() for part in stripped[1:-1].split('|')]
|
|
if table is None:
|
|
table = {'after': blocks[-1] if blocks else None, 'rows': []}
|
|
tables.append(table)
|
|
if not all(SEPARATOR_CELL.match(part) for part in parts):
|
|
row = [re.sub(r'<br\s*/?>', '\n', part) for part in parts]
|
|
cells.extend(row)
|
|
table['rows'].append(row)
|
|
continue
|
|
table = None
|
|
if not stripped:
|
|
flush()
|
|
continue
|
|
heading = re.match(r'^#{1,3}\s+(.+)$', stripped)
|
|
bullet = re.match(r'^[-*]\s+(.+)$', stripped)
|
|
ordered = re.match(r'^\d+\.\s+.+$', stripped)
|
|
if heading or bullet or ordered:
|
|
flush()
|
|
# A numbered item keeps its number: the PDF prints it as text.
|
|
blocks.append(heading.group(1) if heading else bullet.group(1) if bullet else stripped)
|
|
if bullet:
|
|
bullets.add(len(blocks) - 1)
|
|
continue
|
|
paragraph.append(line)
|
|
flush()
|
|
return blocks, cells, tables, bullets
|
|
|
|
|
|
def index_texts(texts):
|
|
exact, by_length = {}, {}
|
|
for text in texts:
|
|
key = letters(text)
|
|
if key and key not in exact:
|
|
exact[key] = text
|
|
by_length.setdefault(len(key), []).append(key)
|
|
return exact, by_length
|
|
|
|
|
|
def is_subsequence(short, long):
|
|
remaining = iter(long)
|
|
return all(character in remaining for character in short)
|
|
|
|
|
|
def source_for(text, exact, by_length):
|
|
"""The authored text a converted fragment came from, if it is unambiguous.
|
|
|
|
Same letters means only spacing and punctuation differ. Failing that, a
|
|
fragment that is the source with up to three characters missing - the
|
|
converter drops characters, it never invents them - is accepted when just
|
|
one source text fits.
|
|
"""
|
|
key = letters(text)
|
|
if not key:
|
|
return None
|
|
if key in exact:
|
|
return exact[key]
|
|
if len(key) < 8:
|
|
return None
|
|
fits = {
|
|
exact[candidate]
|
|
for length in range(len(key) + 1, len(key) + 4)
|
|
for candidate in by_length.get(length, ())
|
|
if is_subsequence(key, candidate)
|
|
}
|
|
return fits.pop() if len(fits) == 1 else None
|
|
|
|
|
|
def write_text(paragraph, text):
|
|
"""Replace a paragraph's runs with the authored text, keeping its look."""
|
|
template = paragraph.find('.//' + qn('w:rPr'))
|
|
template = copy.deepcopy(template) if template is not None else OxmlElement('w:rPr')
|
|
for child in list(paragraph):
|
|
if child.tag != qn('w:pPr'):
|
|
paragraph.remove(child)
|
|
|
|
def run(content=None, bold=False, code=False, line_break=False):
|
|
element = OxmlElement('w:r')
|
|
properties = copy.deepcopy(template)
|
|
fonts = properties.find(qn('w:rFonts'))
|
|
if fonts is not None:
|
|
face = MONO_FONT if code else THAI_FONT
|
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
|
fonts.set(qn(attribute), face)
|
|
weight = properties.find(qn('w:b'))
|
|
for complex_weight in properties.findall(qn('w:bCs')):
|
|
properties.remove(complex_weight)
|
|
if bold:
|
|
if weight is None:
|
|
weight = OxmlElement('w:b')
|
|
(fonts.addnext(weight) if fonts is not None else properties.insert(0, weight))
|
|
weight.attrib.pop(qn('w:val'), None)
|
|
weight.addnext(OxmlElement('w:bCs'))
|
|
elif weight is not None:
|
|
weight.set(qn('w:val'), '0')
|
|
element.append(properties)
|
|
if line_break:
|
|
element.append(OxmlElement('w:br'))
|
|
else:
|
|
node = OxmlElement('w:t')
|
|
node.text = content
|
|
node.set('{http://www.w3.org/XML/1998/namespace}space', 'preserve')
|
|
element.append(node)
|
|
paragraph.append(element)
|
|
|
|
for number, line in enumerate(text.split('\n')):
|
|
if number:
|
|
run(line_break=True)
|
|
cursor = 0
|
|
for match in INLINE.finditer(line):
|
|
if match.start() > cursor:
|
|
run(line[cursor:match.start()])
|
|
if match.group(1) is not None:
|
|
run(match.group(1), bold=True)
|
|
elif match.group(2) is not None:
|
|
run(match.group(2), code=True)
|
|
else:
|
|
run(match.group(3))
|
|
cursor = match.end()
|
|
if cursor < len(line):
|
|
run(line[cursor:])
|
|
|
|
|
|
def split_blocks(key, block_keys):
|
|
"""The run of consecutive source blocks a converted paragraph is made of."""
|
|
for start, first in enumerate(block_keys):
|
|
if not first or not key.startswith(first):
|
|
continue
|
|
combined, end = first, start
|
|
while len(combined) < len(key) and end + 1 < len(block_keys) and end - start < 7:
|
|
end += 1
|
|
combined += block_keys[end]
|
|
if combined == key:
|
|
return list(range(start, end + 1))
|
|
if not key.startswith(combined):
|
|
break
|
|
return None
|
|
|
|
|
|
def restore_text(document, markdown, bullets):
|
|
"""Put the authored wording back where the converter mangled it.
|
|
|
|
PDF text has no word spaces for Thai, so the conversion runs Thai words
|
|
together, breaks a paragraph into one paragraph per printed line, and now
|
|
and then drops a character. Wherever a converted paragraph or table cell
|
|
can be traced to exactly one piece of the source, its text is rewritten
|
|
from the source; anything that cannot be traced is left as converted.
|
|
"""
|
|
blocks, cells, _, bullet_blocks = read_source(markdown)
|
|
block_keys = [letters(block) for block in blocks]
|
|
known = {id(paragraph) for paragraph in bullets}
|
|
block_exact, block_lengths = index_texts(blocks)
|
|
cell_exact, cell_lengths = index_texts(cells)
|
|
body = document.element.body
|
|
rewritten = merged = 0
|
|
|
|
for cell in body.iter(qn('w:tc')):
|
|
if cell.find('.//' + qn('w:tbl')) is not None:
|
|
continue
|
|
paragraphs = cell.findall(qn('w:p'))
|
|
text = '\n'.join(text_of(paragraph) for paragraph in paragraphs)
|
|
source = source_for(text, cell_exact, cell_lengths)
|
|
if source is None or '___' in source or source == text:
|
|
continue
|
|
write_text(paragraphs[0], source)
|
|
for extra in paragraphs[1:]:
|
|
cell.remove(extra)
|
|
rewritten += 1
|
|
|
|
children = [child for child in body if child.tag == qn('w:p') and text_of(child).strip()]
|
|
index = 0
|
|
while index < len(children):
|
|
paragraph = children[index]
|
|
source = source_for(text_of(paragraph), block_exact, block_lengths)
|
|
span = 1
|
|
if source is None:
|
|
# A paragraph the converter split into one paragraph per line.
|
|
combined = text_of(paragraph)
|
|
follower = paragraph
|
|
for count in range(2, 9):
|
|
follower = follower.getnext()
|
|
if follower is None or follower.tag != qn('w:p') or not text_of(follower).strip():
|
|
break
|
|
combined += text_of(follower)
|
|
if letters(combined) in block_exact:
|
|
source, span = block_exact[letters(combined)], count
|
|
break
|
|
if source is None:
|
|
# Several list items or paragraphs the converter ran into one.
|
|
parts = split_blocks(letters(text_of(paragraph)), block_keys)
|
|
if parts and not any('___' in blocks[part] for part in parts):
|
|
previous = paragraph
|
|
for number, part in enumerate(parts):
|
|
target = paragraph
|
|
if number:
|
|
target = OxmlElement('w:p')
|
|
if paragraph.find(qn('w:pPr')) is not None:
|
|
target.append(copy.deepcopy(paragraph.find(qn('w:pPr'))))
|
|
template = paragraph.find('.//' + qn('w:rPr'))
|
|
run = OxmlElement('w:r')
|
|
run.append(copy.deepcopy(template) if template is not None else OxmlElement('w:rPr'))
|
|
target.append(run)
|
|
previous.addnext(target)
|
|
previous = target
|
|
write_text(target, blocks[part])
|
|
if part in bullet_blocks and id(target) not in known:
|
|
bullets.append(target)
|
|
known.add(id(target))
|
|
rewritten += len(parts)
|
|
index += 1
|
|
continue
|
|
if source is not None and '___' not in source:
|
|
for extra in children[index + 1:index + span]:
|
|
body.remove(extra)
|
|
merged += 1
|
|
write_text(paragraph, source)
|
|
rewritten += 1
|
|
index += span
|
|
|
|
return rewritten, merged
|
|
|
|
|
|
TINT = 'FFF3F6' # header-row tint used across the reference tables
|
|
TABLE_PT = 9
|
|
CELL_MARGIN_MM = (1.6, 2) # vertical, horizontal padding of a table cell
|
|
PRINTED_TEXT_WIDTH_MM = {False: 160, True: 267} # portrait, landscape
|
|
|
|
|
|
DISTINCTIVE_LETTERS = 15 # shorter cells - labels, IDs, dates - recur elsewhere
|
|
DROPPED_CHARACTER_SLACK = 3
|
|
|
|
|
|
def present(key, document_key):
|
|
"""Whether a cell's whole text reached the document.
|
|
|
|
The whole cell, not a fragment of it: short Thai phrases recur across a
|
|
work product, so a partial match proves nothing. Its two halves in order,
|
|
a few characters apart at most, still count, as the converter sometimes
|
|
drops a character mid-cell.
|
|
"""
|
|
if key in document_key:
|
|
return True
|
|
half = len(key) // 2
|
|
start = document_key.find(key[:half])
|
|
while start >= 0:
|
|
second = document_key.find(key[half:], start + half)
|
|
if 0 <= second - (start + half) <= DROPPED_CHARACTER_SLACK:
|
|
return True
|
|
start = document_key.find(key[:half], start + 1)
|
|
return False
|
|
|
|
|
|
def dropped_tables(document, markdown):
|
|
"""Source tables of which not one distinctive cell reached the document."""
|
|
_, _, tables, _ = read_source(markdown)
|
|
document_key = letters(text_of(document.element.body))
|
|
missing = []
|
|
for table in tables:
|
|
keys = [letters(cell) for row in table['rows'] for cell in row]
|
|
keys = [key for key in keys if len(key) >= DISTINCTIVE_LETTERS]
|
|
if keys and not any(present(key, document_key) for key in keys):
|
|
missing.append(table)
|
|
return missing
|
|
|
|
|
|
def build_table(document, rows):
|
|
section = document.sections[0]
|
|
landscape = section.page_width > section.page_height
|
|
width = Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips
|
|
columns = max(len(row) for row in rows)
|
|
longest = [max(len(letters(row[i])) if i < len(row) else 0 for row in rows) for i in range(columns)]
|
|
weights = [min(max(value, 8), 80) for value in longest]
|
|
widths = [round(width * weight / sum(weights)) for weight in weights]
|
|
|
|
def element(tag, **attributes):
|
|
node = OxmlElement(tag)
|
|
for name, value in attributes.items():
|
|
node.set(qn(name), str(value))
|
|
return node
|
|
|
|
table = element('w:tbl')
|
|
properties = element('w:tblPr')
|
|
properties.append(element('w:tblW', **{'w:w': width, 'w:type': 'dxa'}))
|
|
indent = round(Mm(text_block_left_mm(section)).twips - section.left_margin.twips)
|
|
properties.append(element('w:tblInd', **{'w:w': indent, 'w:type': 'dxa'}))
|
|
borders = element('w:tblBorders')
|
|
for edge in ('top', 'left', 'bottom', 'right', 'insideH', 'insideV'):
|
|
borders.append(element('w:' + edge, **{'w:val': 'single', 'w:sz': RULE_SIZE, 'w:space': 0, 'w:color': RULE_COLOUR}))
|
|
properties.append(borders)
|
|
properties.append(element('w:tblLayout', **{'w:type': 'fixed'}))
|
|
margins = element('w:tblCellMar')
|
|
vertical, horizontal = (round(Mm(value).twips) for value in CELL_MARGIN_MM)
|
|
for edge, value in (('top', vertical), ('left', horizontal), ('bottom', vertical), ('right', horizontal)):
|
|
margins.append(element('w:' + edge, **{'w:w': value, 'w:type': 'dxa'}))
|
|
properties.append(margins)
|
|
table.append(properties)
|
|
|
|
grid = element('w:tblGrid')
|
|
for column_width in widths:
|
|
grid.append(element('w:gridCol', **{'w:w': column_width}))
|
|
table.append(grid)
|
|
|
|
for number, cells in enumerate(rows):
|
|
row = element('w:tr')
|
|
row_properties = element('w:trPr')
|
|
row_properties.append(element('w:cantSplit'))
|
|
if number == 0:
|
|
row_properties.append(element('w:tblHeader'))
|
|
row.append(row_properties)
|
|
for column in range(columns):
|
|
cell = element('w:tc')
|
|
cell_properties = element('w:tcPr')
|
|
cell_properties.append(element('w:tcW', **{'w:w': widths[column], 'w:type': 'dxa'}))
|
|
if number == 0:
|
|
cell_properties.append(element('w:shd', **{'w:val': 'clear', 'w:color': 'auto', 'w:fill': TINT}))
|
|
cell.append(cell_properties)
|
|
paragraph = element('w:p')
|
|
paragraph_properties = element('w:pPr')
|
|
paragraph_properties.append(element('w:spacing', **{'w:before': 0, 'w:after': 0}))
|
|
paragraph.append(paragraph_properties)
|
|
template = element('w:r')
|
|
run_properties = element('w:rPr')
|
|
fonts = element('w:rFonts', **{'w:ascii': THAI_FONT, 'w:hAnsi': THAI_FONT, 'w:cs': THAI_FONT, 'w:eastAsia': THAI_FONT})
|
|
run_properties.append(fonts)
|
|
run_properties.append(element('w:color', **{'w:val': '111111'}))
|
|
run_properties.append(element('w:sz', **{'w:val': TABLE_PT * 2}))
|
|
run_properties.append(element('w:szCs', **{'w:val': TABLE_PT * 2}))
|
|
template.append(run_properties)
|
|
paragraph.append(template)
|
|
write_text(paragraph, cells[column] if column < len(cells) else '')
|
|
cell.append(paragraph)
|
|
row.append(cell)
|
|
table.append(row)
|
|
return table
|
|
|
|
|
|
def restore_dropped_tables(document, markdown):
|
|
"""Put back tables the converter left out, where the source places them.
|
|
|
|
The converter occasionally drops a whole table that is plainly printed in
|
|
the PDF. It is rebuilt from the source in the style of the converted
|
|
tables, after the paragraph it follows in the source.
|
|
"""
|
|
restored = 0
|
|
body = document.element.body
|
|
for table in dropped_tables(document, markdown):
|
|
if not table['after'] or not table['rows']:
|
|
continue
|
|
anchor_key = letters(table['after'])
|
|
paragraphs = [child for child in body if child.tag == qn('w:p')]
|
|
anchor = next((child for child in paragraphs if letters(text_of(child)) == anchor_key), None)
|
|
if anchor is None:
|
|
# The block may share a paragraph with what precedes it.
|
|
endings = [child for child in paragraphs if anchor_key and letters(text_of(child)).endswith(anchor_key)]
|
|
anchor = endings[0] if len(endings) == 1 else None
|
|
if anchor is None:
|
|
continue
|
|
anchor.addnext(build_table(document, table['rows']))
|
|
restored += 1
|
|
return restored
|
|
|
|
|
|
def style_run(run, points):
|
|
run.font.size = Pt(points)
|
|
run.font.color.rgb = INK
|
|
fonts = OxmlElement('w:rFonts')
|
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
|
fonts.set(qn(attribute), THAI_FONT)
|
|
run._element.get_or_add_rPr().insert(0, fonts)
|
|
|
|
|
|
def drop_style(paragraph):
|
|
# The template's Header and Footer styles carry centre and right tab stops
|
|
# of their own, which catch the tab before the footer tag and centre it.
|
|
properties = paragraph._p.pPr
|
|
if properties is not None and properties.find(qn('w:pStyle')) is not None:
|
|
properties.remove(properties.find(qn('w:pStyle')))
|
|
|
|
|
|
def add_top_border(paragraph):
|
|
borders = OxmlElement('w:pBdr')
|
|
top = OxmlElement('w:top')
|
|
top.set(qn('w:val'), 'single')
|
|
top.set(qn('w:sz'), '6') # .75pt, as printed
|
|
top.set(qn('w:space'), '1')
|
|
top.set(qn('w:color'), '111111')
|
|
borders.append(top)
|
|
paragraph._p.get_or_add_pPr().append(borders)
|
|
|
|
|
|
def install_letterhead(document, band, footer_tag):
|
|
"""Give every page the CI band and footer rule as real Word furniture."""
|
|
for index, section in enumerate(document.sections):
|
|
section.top_margin = Mm(BODY_TOP_MM)
|
|
section.header_distance = Mm(0)
|
|
section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM)
|
|
# Only the first section carries the furniture; the rest inherit it.
|
|
section.header.is_linked_to_previous = index > 0
|
|
section.footer.is_linked_to_previous = index > 0
|
|
|
|
section = document.sections[0]
|
|
|
|
header = section.header.paragraphs[0]
|
|
drop_style(header)
|
|
# The band spans the page, so it starts outside the text block.
|
|
header.paragraph_format.left_indent = -section.left_margin
|
|
header.paragraph_format.space_before = Pt(0)
|
|
header.paragraph_format.space_after = Pt(0)
|
|
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
|
|
|
|
footer = section.footer.paragraphs[0]
|
|
drop_style(footer)
|
|
footer.paragraph_format.space_before = Pt(0)
|
|
footer.paragraph_format.space_after = Pt(0)
|
|
footer.paragraph_format.tab_stops.add_tab_stop(
|
|
section.page_width - section.left_margin - section.right_margin,
|
|
WD_TAB_ALIGNMENT.RIGHT
|
|
)
|
|
add_top_border(footer)
|
|
style_run(footer.add_run(STANDARD), 9)
|
|
style_run(footer.add_run(f'\t{footer_tag}'), 9)
|
|
|
|
|
|
def convert(pdf, source, band, footer_tag):
|
|
docx = pdf.with_suffix('.docx')
|
|
converter = Converter(str(pdf))
|
|
try:
|
|
converter.convert(str(docx))
|
|
finally:
|
|
converter.close()
|
|
|
|
document = Document(str(docx))
|
|
strip_reconstructed_furniture(document)
|
|
strip_logo_images(document)
|
|
bullets = unwrap_list_tables(document) + take_bullet_images(document)
|
|
normalise_ooxml(document)
|
|
normalise_fonts(document)
|
|
markdown = source.read_text(encoding='utf8')
|
|
restore_text(document, markdown, bullets)
|
|
restore_dropped_tables(document, markdown)
|
|
for paragraph in bullets:
|
|
if paragraph.getparent() is not None:
|
|
make_bullet(document, paragraph)
|
|
drop_padding(document)
|
|
match_rule_weight(document)
|
|
install_letterhead(document, band, footer_tag)
|
|
document.save(str(docx))
|
|
return docx
|
|
|
|
|
|
def render_bands(source_root, band_root):
|
|
subprocess.run(
|
|
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
|
|
check=True
|
|
)
|
|
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
|
|
|
|
|
|
def main(argv):
|
|
if len(argv) < 2:
|
|
raise SystemExit(__doc__)
|
|
|
|
source_root, delivery_root = Path(argv[0]), Path(argv[1])
|
|
temporary = None
|
|
if len(argv) > 2:
|
|
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
|
|
else:
|
|
temporary = tempfile.TemporaryDirectory()
|
|
bands = render_bands(source_root, temporary.name)
|
|
|
|
failures = []
|
|
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
|
|
pdf = delivery_root / Path(relative).with_suffix('.pdf')
|
|
try:
|
|
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'])
|
|
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
|
|
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
|
failures.append((pdf, error))
|
|
print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr)
|
|
|
|
if temporary:
|
|
temporary.cleanup()
|
|
|
|
print(f'{len(bands) - len(failures)}/{len(bands)} documents converted')
|
|
if failures:
|
|
raise SystemExit(1)
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main(sys.argv[1:])
|