Fix Word fonts, bullets and footer tab; restore dropped text from source
This commit is contained in:
@@ -19,18 +19,21 @@ Requires pdf2docx: pip install -r requirements.txt
|
|||||||
Usage:
|
Usage:
|
||||||
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
|
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
|
||||||
"""
|
"""
|
||||||
|
import copy
|
||||||
import json
|
import json
|
||||||
import re
|
import re
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
|
import unicodedata
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from docx import Document
|
from docx import Document
|
||||||
from docx.enum.text import WD_TAB_ALIGNMENT
|
from docx.enum.text import WD_TAB_ALIGNMENT
|
||||||
from docx.oxml import OxmlElement
|
from docx.oxml import OxmlElement
|
||||||
from docx.oxml.ns import qn
|
from docx.oxml.ns import qn
|
||||||
from docx.shared import Mm, Pt, RGBColor
|
from docx.shared import Mm, Pt, RGBColor, Twips
|
||||||
|
from docx.text.paragraph import Paragraph
|
||||||
from pdf2docx import Converter
|
from pdf2docx import Converter
|
||||||
|
|
||||||
HERE = Path(__file__).resolve().parent
|
HERE = Path(__file__).resolve().parent
|
||||||
@@ -43,6 +46,7 @@ STANDARD = 'ISO/IEC 29110-4-1:2018'
|
|||||||
RULE_SIZE = '4' # .5pt table rule, as printed
|
RULE_SIZE = '4' # .5pt table rule, as printed
|
||||||
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
||||||
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
|
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
|
||||||
|
MONO_FONT = 'Courier New' # code spans; DejaVu Sans Mono in the PDF
|
||||||
INK = RGBColor(0x11, 0x11, 0x11)
|
INK = RGBColor(0x11, 0x11, 0x11)
|
||||||
|
|
||||||
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
|
# The furniture pdf2docx rebuilds in the body, recognised by text that appears
|
||||||
@@ -52,7 +56,19 @@ LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower')
|
|||||||
|
|
||||||
|
|
||||||
def text_of(element):
|
def text_of(element):
|
||||||
return ''.join(element.itertext())
|
"""The text Word shows for an element: its w:t runs, tabs and breaks.
|
||||||
|
|
||||||
|
Not itertext(): the converter leaves other text-bearing nodes in its runs,
|
||||||
|
which would read every paragraph twice over.
|
||||||
|
"""
|
||||||
|
parts = []
|
||||||
|
for node in element.iter(qn('w:t'), qn('w:tab'), qn('w:br')):
|
||||||
|
if node.tag == qn('w:t'):
|
||||||
|
parts.append(node.text or '')
|
||||||
|
elif node.getparent().tag == qn('w:r'):
|
||||||
|
# A w:tab under w:tabs is a tab-stop definition, not a character.
|
||||||
|
parts.append('\t' if node.tag == qn('w:tab') else '\n')
|
||||||
|
return ''.join(parts)
|
||||||
|
|
||||||
|
|
||||||
def is_blank(element):
|
def is_blank(element):
|
||||||
@@ -192,6 +208,522 @@ def strip_logo_images(document):
|
|||||||
return removed
|
return removed
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Fonts
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def normalise_fonts(document):
|
||||||
|
"""Point every run at a face Word actually has.
|
||||||
|
|
||||||
|
The PDF embeds its faces as Type3 fonts, so the converter names runs after
|
||||||
|
the PDF objects - "Type3 (4 0 R)" - and gives them no complex-script face
|
||||||
|
or size. Word then sets Latin text in a fallback serif and Thai text at the
|
||||||
|
default size, which is the mixed, jumping type the conversion showed.
|
||||||
|
"""
|
||||||
|
for run in document.element.body.iter(qn('w:r')):
|
||||||
|
properties = run.find(qn('w:rPr'))
|
||||||
|
if properties is None:
|
||||||
|
properties = OxmlElement('w:rPr')
|
||||||
|
run.insert(0, properties)
|
||||||
|
|
||||||
|
fonts = properties.find(qn('w:rFonts'))
|
||||||
|
face = MONO_FONT if fonts is not None and 'Mono' in (fonts.get(qn('w:ascii')) or '') else THAI_FONT
|
||||||
|
if fonts is None:
|
||||||
|
fonts = OxmlElement('w:rFonts')
|
||||||
|
style = properties.find(qn('w:rStyle'))
|
||||||
|
properties.insert(1 if style is not None else 0, fonts)
|
||||||
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
||||||
|
fonts.set(qn(attribute), face)
|
||||||
|
|
||||||
|
size = properties.find(qn('w:sz'))
|
||||||
|
if size is not None and properties.find(qn('w:szCs')) is None:
|
||||||
|
complex_size = OxmlElement('w:szCs')
|
||||||
|
complex_size.set(qn('w:val'), size.get(qn('w:val')))
|
||||||
|
size.addnext(complex_size)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Bullets
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
BULLET = '•'
|
||||||
|
BULLET_TEXT_MM = 7 # ul padding-left in sdlc-delivery.css
|
||||||
|
BULLET_HANG_MM = 4
|
||||||
|
|
||||||
|
|
||||||
|
def text_block_left_mm(section):
|
||||||
|
# The PDF text block starts 30mm in, or 15mm on the landscape work products.
|
||||||
|
return 15 if section.page_width > section.page_height else 30
|
||||||
|
|
||||||
|
|
||||||
|
def unwrap_list_tables(document):
|
||||||
|
"""Turn the tables the converter builds out of bullet lists back into text.
|
||||||
|
|
||||||
|
A short list is sometimes laid out as a grid - bullet image, item, empty
|
||||||
|
cell - with stray bottom borders that print as loose rules. Each row goes
|
||||||
|
back into the body as a paragraph of its own.
|
||||||
|
"""
|
||||||
|
body = document.element.body
|
||||||
|
items = []
|
||||||
|
for table in [child for child in body if child.tag == qn('w:tbl')]:
|
||||||
|
rows = table.findall(qn('w:tr'))
|
||||||
|
entries = []
|
||||||
|
for row in rows:
|
||||||
|
cells = row.findall(qn('w:tc'))
|
||||||
|
worded = [cell for cell in cells if text_of(cell).strip()]
|
||||||
|
marked = [cell for cell in cells if cell.findall('.//' + qn('w:drawing')) and not text_of(cell).strip()]
|
||||||
|
if len(worded) != 1 or not marked:
|
||||||
|
entries = None
|
||||||
|
break
|
||||||
|
entries.append(worded[0])
|
||||||
|
if not rows or entries is None:
|
||||||
|
continue
|
||||||
|
for cell in entries:
|
||||||
|
for paragraph in cell.findall(qn('w:p')):
|
||||||
|
if text_of(paragraph).strip():
|
||||||
|
table.addprevious(paragraph)
|
||||||
|
items.append(paragraph)
|
||||||
|
body.remove(table)
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def take_bullet_images(document):
|
||||||
|
"""Collect the list items drawn with an image marker, dropping the image."""
|
||||||
|
items = []
|
||||||
|
for paragraph in [child for child in document.element.body if child.tag == qn('w:p')]:
|
||||||
|
drawings = paragraph.findall('.//' + qn('w:drawing'))
|
||||||
|
if not drawings or not text_of(paragraph).strip():
|
||||||
|
continue
|
||||||
|
for drawing in drawings:
|
||||||
|
run = drawing.getparent()
|
||||||
|
run.remove(drawing)
|
||||||
|
if run.tag == qn('w:r') and not text_of(run).strip():
|
||||||
|
run.getparent().remove(run)
|
||||||
|
items.append(paragraph)
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def make_bullet(document, paragraph):
|
||||||
|
section = document.sections[0]
|
||||||
|
item = Paragraph(paragraph, document._body)
|
||||||
|
text_mm = text_block_left_mm(section) + BULLET_TEXT_MM
|
||||||
|
item.paragraph_format.left_indent = Twips(round(Mm(text_mm).twips - section.left_margin.twips))
|
||||||
|
item.paragraph_format.first_line_indent = -Mm(BULLET_HANG_MM)
|
||||||
|
|
||||||
|
first = paragraph.find(qn('w:r'))
|
||||||
|
marker = OxmlElement('w:r')
|
||||||
|
if first is not None and first.find(qn('w:rPr')) is not None:
|
||||||
|
marker.append(copy.deepcopy(first.find(qn('w:rPr'))))
|
||||||
|
text = OxmlElement('w:t')
|
||||||
|
text.text = BULLET
|
||||||
|
marker.append(text)
|
||||||
|
marker.append(OxmlElement('w:tab'))
|
||||||
|
properties = paragraph.find(qn('w:pPr'))
|
||||||
|
if properties is not None:
|
||||||
|
properties.addnext(marker)
|
||||||
|
else:
|
||||||
|
paragraph.insert(0, marker)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Text from the source
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
APPROVAL_FIELD = re.compile(r'^(Name|Role|Signature|Date|Position|Company|Project roles|Decision):')
|
||||||
|
SEPARATOR_CELL = re.compile(r'^:?-{3,}:?$')
|
||||||
|
INLINE = re.compile(r'\*\*(.+?)\*\*|`([^`]+)`|\\([\\*_`~])')
|
||||||
|
|
||||||
|
|
||||||
|
def letters(text):
|
||||||
|
"""The characters two renderings of the same text must share."""
|
||||||
|
return ''.join(
|
||||||
|
character for character in unicodedata.normalize('NFC', text)
|
||||||
|
if unicodedata.category(character)[0] in 'LMN'
|
||||||
|
).lower()
|
||||||
|
|
||||||
|
|
||||||
|
def read_source(markdown):
|
||||||
|
"""Blocks and table cells of a work product, as the reader sees them."""
|
||||||
|
body = re.sub(r'^#\s+.+\n+', '', markdown, count=1, flags=re.M)
|
||||||
|
body = re.sub(r'^<!--.*?-->\s*$', '', body, flags=re.M)
|
||||||
|
blocks, cells, tables, paragraph = [], [], [], []
|
||||||
|
bullets = set()
|
||||||
|
table = None
|
||||||
|
|
||||||
|
def flush():
|
||||||
|
if paragraph:
|
||||||
|
joined = ''
|
||||||
|
for index, line in enumerate(paragraph):
|
||||||
|
hard = re.search(r'\s{2,}$', line) or APPROVAL_FIELD.match(line.strip())
|
||||||
|
joined += line.strip() + ('\n' if hard else (' ' if index < len(paragraph) - 1 else ''))
|
||||||
|
blocks.append(joined.rstrip('\n'))
|
||||||
|
paragraph.clear()
|
||||||
|
|
||||||
|
for line in body.splitlines():
|
||||||
|
stripped = line.strip()
|
||||||
|
if stripped.startswith('|') and stripped.endswith('|'):
|
||||||
|
flush()
|
||||||
|
parts = [part.strip() for part in stripped[1:-1].split('|')]
|
||||||
|
if table is None:
|
||||||
|
table = {'after': blocks[-1] if blocks else None, 'rows': []}
|
||||||
|
tables.append(table)
|
||||||
|
if not all(SEPARATOR_CELL.match(part) for part in parts):
|
||||||
|
row = [re.sub(r'<br\s*/?>', '\n', part) for part in parts]
|
||||||
|
cells.extend(row)
|
||||||
|
table['rows'].append(row)
|
||||||
|
continue
|
||||||
|
table = None
|
||||||
|
if not stripped:
|
||||||
|
flush()
|
||||||
|
continue
|
||||||
|
heading = re.match(r'^#{1,3}\s+(.+)$', stripped)
|
||||||
|
bullet = re.match(r'^[-*]\s+(.+)$', stripped)
|
||||||
|
ordered = re.match(r'^\d+\.\s+.+$', stripped)
|
||||||
|
if heading or bullet or ordered:
|
||||||
|
flush()
|
||||||
|
# A numbered item keeps its number: the PDF prints it as text.
|
||||||
|
blocks.append(heading.group(1) if heading else bullet.group(1) if bullet else stripped)
|
||||||
|
if bullet:
|
||||||
|
bullets.add(len(blocks) - 1)
|
||||||
|
continue
|
||||||
|
paragraph.append(line)
|
||||||
|
flush()
|
||||||
|
return blocks, cells, tables, bullets
|
||||||
|
|
||||||
|
|
||||||
|
def index_texts(texts):
|
||||||
|
exact, by_length = {}, {}
|
||||||
|
for text in texts:
|
||||||
|
key = letters(text)
|
||||||
|
if key and key not in exact:
|
||||||
|
exact[key] = text
|
||||||
|
by_length.setdefault(len(key), []).append(key)
|
||||||
|
return exact, by_length
|
||||||
|
|
||||||
|
|
||||||
|
def is_subsequence(short, long):
|
||||||
|
remaining = iter(long)
|
||||||
|
return all(character in remaining for character in short)
|
||||||
|
|
||||||
|
|
||||||
|
def source_for(text, exact, by_length):
|
||||||
|
"""The authored text a converted fragment came from, if it is unambiguous.
|
||||||
|
|
||||||
|
Same letters means only spacing and punctuation differ. Failing that, a
|
||||||
|
fragment that is the source with up to three characters missing - the
|
||||||
|
converter drops characters, it never invents them - is accepted when just
|
||||||
|
one source text fits.
|
||||||
|
"""
|
||||||
|
key = letters(text)
|
||||||
|
if not key:
|
||||||
|
return None
|
||||||
|
if key in exact:
|
||||||
|
return exact[key]
|
||||||
|
if len(key) < 8:
|
||||||
|
return None
|
||||||
|
fits = {
|
||||||
|
exact[candidate]
|
||||||
|
for length in range(len(key) + 1, len(key) + 4)
|
||||||
|
for candidate in by_length.get(length, ())
|
||||||
|
if is_subsequence(key, candidate)
|
||||||
|
}
|
||||||
|
return fits.pop() if len(fits) == 1 else None
|
||||||
|
|
||||||
|
|
||||||
|
def write_text(paragraph, text):
|
||||||
|
"""Replace a paragraph's runs with the authored text, keeping its look."""
|
||||||
|
template = paragraph.find('.//' + qn('w:rPr'))
|
||||||
|
template = copy.deepcopy(template) if template is not None else OxmlElement('w:rPr')
|
||||||
|
for child in list(paragraph):
|
||||||
|
if child.tag != qn('w:pPr'):
|
||||||
|
paragraph.remove(child)
|
||||||
|
|
||||||
|
def run(content=None, bold=False, code=False, line_break=False):
|
||||||
|
element = OxmlElement('w:r')
|
||||||
|
properties = copy.deepcopy(template)
|
||||||
|
fonts = properties.find(qn('w:rFonts'))
|
||||||
|
if fonts is not None:
|
||||||
|
face = MONO_FONT if code else THAI_FONT
|
||||||
|
for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'):
|
||||||
|
fonts.set(qn(attribute), face)
|
||||||
|
weight = properties.find(qn('w:b'))
|
||||||
|
for complex_weight in properties.findall(qn('w:bCs')):
|
||||||
|
properties.remove(complex_weight)
|
||||||
|
if bold:
|
||||||
|
if weight is None:
|
||||||
|
weight = OxmlElement('w:b')
|
||||||
|
(fonts.addnext(weight) if fonts is not None else properties.insert(0, weight))
|
||||||
|
weight.attrib.pop(qn('w:val'), None)
|
||||||
|
weight.addnext(OxmlElement('w:bCs'))
|
||||||
|
elif weight is not None:
|
||||||
|
weight.set(qn('w:val'), '0')
|
||||||
|
element.append(properties)
|
||||||
|
if line_break:
|
||||||
|
element.append(OxmlElement('w:br'))
|
||||||
|
else:
|
||||||
|
node = OxmlElement('w:t')
|
||||||
|
node.text = content
|
||||||
|
node.set('{http://www.w3.org/XML/1998/namespace}space', 'preserve')
|
||||||
|
element.append(node)
|
||||||
|
paragraph.append(element)
|
||||||
|
|
||||||
|
for number, line in enumerate(text.split('\n')):
|
||||||
|
if number:
|
||||||
|
run(line_break=True)
|
||||||
|
cursor = 0
|
||||||
|
for match in INLINE.finditer(line):
|
||||||
|
if match.start() > cursor:
|
||||||
|
run(line[cursor:match.start()])
|
||||||
|
if match.group(1) is not None:
|
||||||
|
run(match.group(1), bold=True)
|
||||||
|
elif match.group(2) is not None:
|
||||||
|
run(match.group(2), code=True)
|
||||||
|
else:
|
||||||
|
run(match.group(3))
|
||||||
|
cursor = match.end()
|
||||||
|
if cursor < len(line):
|
||||||
|
run(line[cursor:])
|
||||||
|
|
||||||
|
|
||||||
|
def split_blocks(key, block_keys):
|
||||||
|
"""The run of consecutive source blocks a converted paragraph is made of."""
|
||||||
|
for start, first in enumerate(block_keys):
|
||||||
|
if not first or not key.startswith(first):
|
||||||
|
continue
|
||||||
|
combined, end = first, start
|
||||||
|
while len(combined) < len(key) and end + 1 < len(block_keys) and end - start < 7:
|
||||||
|
end += 1
|
||||||
|
combined += block_keys[end]
|
||||||
|
if combined == key:
|
||||||
|
return list(range(start, end + 1))
|
||||||
|
if not key.startswith(combined):
|
||||||
|
break
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def restore_text(document, markdown, bullets):
|
||||||
|
"""Put the authored wording back where the converter mangled it.
|
||||||
|
|
||||||
|
PDF text has no word spaces for Thai, so the conversion runs Thai words
|
||||||
|
together, breaks a paragraph into one paragraph per printed line, and now
|
||||||
|
and then drops a character. Wherever a converted paragraph or table cell
|
||||||
|
can be traced to exactly one piece of the source, its text is rewritten
|
||||||
|
from the source; anything that cannot be traced is left as converted.
|
||||||
|
"""
|
||||||
|
blocks, cells, _, bullet_blocks = read_source(markdown)
|
||||||
|
block_keys = [letters(block) for block in blocks]
|
||||||
|
known = {id(paragraph) for paragraph in bullets}
|
||||||
|
block_exact, block_lengths = index_texts(blocks)
|
||||||
|
cell_exact, cell_lengths = index_texts(cells)
|
||||||
|
body = document.element.body
|
||||||
|
rewritten = merged = 0
|
||||||
|
|
||||||
|
for cell in body.iter(qn('w:tc')):
|
||||||
|
if cell.find('.//' + qn('w:tbl')) is not None:
|
||||||
|
continue
|
||||||
|
paragraphs = cell.findall(qn('w:p'))
|
||||||
|
text = '\n'.join(text_of(paragraph) for paragraph in paragraphs)
|
||||||
|
source = source_for(text, cell_exact, cell_lengths)
|
||||||
|
if source is None or '___' in source or source == text:
|
||||||
|
continue
|
||||||
|
write_text(paragraphs[0], source)
|
||||||
|
for extra in paragraphs[1:]:
|
||||||
|
cell.remove(extra)
|
||||||
|
rewritten += 1
|
||||||
|
|
||||||
|
children = [child for child in body if child.tag == qn('w:p') and text_of(child).strip()]
|
||||||
|
index = 0
|
||||||
|
while index < len(children):
|
||||||
|
paragraph = children[index]
|
||||||
|
source = source_for(text_of(paragraph), block_exact, block_lengths)
|
||||||
|
span = 1
|
||||||
|
if source is None:
|
||||||
|
# A paragraph the converter split into one paragraph per line.
|
||||||
|
combined = text_of(paragraph)
|
||||||
|
follower = paragraph
|
||||||
|
for count in range(2, 9):
|
||||||
|
follower = follower.getnext()
|
||||||
|
if follower is None or follower.tag != qn('w:p') or not text_of(follower).strip():
|
||||||
|
break
|
||||||
|
combined += text_of(follower)
|
||||||
|
if letters(combined) in block_exact:
|
||||||
|
source, span = block_exact[letters(combined)], count
|
||||||
|
break
|
||||||
|
if source is None:
|
||||||
|
# Several list items or paragraphs the converter ran into one.
|
||||||
|
parts = split_blocks(letters(text_of(paragraph)), block_keys)
|
||||||
|
if parts and not any('___' in blocks[part] for part in parts):
|
||||||
|
previous = paragraph
|
||||||
|
for number, part in enumerate(parts):
|
||||||
|
target = paragraph
|
||||||
|
if number:
|
||||||
|
target = OxmlElement('w:p')
|
||||||
|
if paragraph.find(qn('w:pPr')) is not None:
|
||||||
|
target.append(copy.deepcopy(paragraph.find(qn('w:pPr'))))
|
||||||
|
template = paragraph.find('.//' + qn('w:rPr'))
|
||||||
|
run = OxmlElement('w:r')
|
||||||
|
run.append(copy.deepcopy(template) if template is not None else OxmlElement('w:rPr'))
|
||||||
|
target.append(run)
|
||||||
|
previous.addnext(target)
|
||||||
|
previous = target
|
||||||
|
write_text(target, blocks[part])
|
||||||
|
if part in bullet_blocks and id(target) not in known:
|
||||||
|
bullets.append(target)
|
||||||
|
known.add(id(target))
|
||||||
|
rewritten += len(parts)
|
||||||
|
index += 1
|
||||||
|
continue
|
||||||
|
if source is not None and '___' not in source:
|
||||||
|
for extra in children[index + 1:index + span]:
|
||||||
|
body.remove(extra)
|
||||||
|
merged += 1
|
||||||
|
write_text(paragraph, source)
|
||||||
|
rewritten += 1
|
||||||
|
index += span
|
||||||
|
|
||||||
|
return rewritten, merged
|
||||||
|
|
||||||
|
|
||||||
|
TINT = 'FFF3F6' # header-row tint used across the reference tables
|
||||||
|
TABLE_PT = 9
|
||||||
|
CELL_MARGIN_MM = (1.6, 2) # vertical, horizontal padding of a table cell
|
||||||
|
PRINTED_TEXT_WIDTH_MM = {False: 160, True: 267} # portrait, landscape
|
||||||
|
|
||||||
|
|
||||||
|
DISTINCTIVE_LETTERS = 15 # shorter cells - labels, IDs, dates - recur elsewhere
|
||||||
|
DROPPED_CHARACTER_SLACK = 3
|
||||||
|
|
||||||
|
|
||||||
|
def present(key, document_key):
|
||||||
|
"""Whether a cell's whole text reached the document.
|
||||||
|
|
||||||
|
The whole cell, not a fragment of it: short Thai phrases recur across a
|
||||||
|
work product, so a partial match proves nothing. Its two halves in order,
|
||||||
|
a few characters apart at most, still count, as the converter sometimes
|
||||||
|
drops a character mid-cell.
|
||||||
|
"""
|
||||||
|
if key in document_key:
|
||||||
|
return True
|
||||||
|
half = len(key) // 2
|
||||||
|
start = document_key.find(key[:half])
|
||||||
|
while start >= 0:
|
||||||
|
second = document_key.find(key[half:], start + half)
|
||||||
|
if 0 <= second - (start + half) <= DROPPED_CHARACTER_SLACK:
|
||||||
|
return True
|
||||||
|
start = document_key.find(key[:half], start + 1)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def dropped_tables(document, markdown):
|
||||||
|
"""Source tables of which not one distinctive cell reached the document."""
|
||||||
|
_, _, tables, _ = read_source(markdown)
|
||||||
|
document_key = letters(text_of(document.element.body))
|
||||||
|
missing = []
|
||||||
|
for table in tables:
|
||||||
|
keys = [letters(cell) for row in table['rows'] for cell in row]
|
||||||
|
keys = [key for key in keys if len(key) >= DISTINCTIVE_LETTERS]
|
||||||
|
if keys and not any(present(key, document_key) for key in keys):
|
||||||
|
missing.append(table)
|
||||||
|
return missing
|
||||||
|
|
||||||
|
|
||||||
|
def build_table(document, rows):
|
||||||
|
section = document.sections[0]
|
||||||
|
landscape = section.page_width > section.page_height
|
||||||
|
width = Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips
|
||||||
|
columns = max(len(row) for row in rows)
|
||||||
|
longest = [max(len(letters(row[i])) if i < len(row) else 0 for row in rows) for i in range(columns)]
|
||||||
|
weights = [min(max(value, 8), 80) for value in longest]
|
||||||
|
widths = [round(width * weight / sum(weights)) for weight in weights]
|
||||||
|
|
||||||
|
def element(tag, **attributes):
|
||||||
|
node = OxmlElement(tag)
|
||||||
|
for name, value in attributes.items():
|
||||||
|
node.set(qn(name), str(value))
|
||||||
|
return node
|
||||||
|
|
||||||
|
table = element('w:tbl')
|
||||||
|
properties = element('w:tblPr')
|
||||||
|
properties.append(element('w:tblW', **{'w:w': width, 'w:type': 'dxa'}))
|
||||||
|
indent = round(Mm(text_block_left_mm(section)).twips - section.left_margin.twips)
|
||||||
|
properties.append(element('w:tblInd', **{'w:w': indent, 'w:type': 'dxa'}))
|
||||||
|
borders = element('w:tblBorders')
|
||||||
|
for edge in ('top', 'left', 'bottom', 'right', 'insideH', 'insideV'):
|
||||||
|
borders.append(element('w:' + edge, **{'w:val': 'single', 'w:sz': RULE_SIZE, 'w:space': 0, 'w:color': RULE_COLOUR}))
|
||||||
|
properties.append(borders)
|
||||||
|
properties.append(element('w:tblLayout', **{'w:type': 'fixed'}))
|
||||||
|
margins = element('w:tblCellMar')
|
||||||
|
vertical, horizontal = (round(Mm(value).twips) for value in CELL_MARGIN_MM)
|
||||||
|
for edge, value in (('top', vertical), ('left', horizontal), ('bottom', vertical), ('right', horizontal)):
|
||||||
|
margins.append(element('w:' + edge, **{'w:w': value, 'w:type': 'dxa'}))
|
||||||
|
properties.append(margins)
|
||||||
|
table.append(properties)
|
||||||
|
|
||||||
|
grid = element('w:tblGrid')
|
||||||
|
for column_width in widths:
|
||||||
|
grid.append(element('w:gridCol', **{'w:w': column_width}))
|
||||||
|
table.append(grid)
|
||||||
|
|
||||||
|
for number, cells in enumerate(rows):
|
||||||
|
row = element('w:tr')
|
||||||
|
row_properties = element('w:trPr')
|
||||||
|
row_properties.append(element('w:cantSplit'))
|
||||||
|
if number == 0:
|
||||||
|
row_properties.append(element('w:tblHeader'))
|
||||||
|
row.append(row_properties)
|
||||||
|
for column in range(columns):
|
||||||
|
cell = element('w:tc')
|
||||||
|
cell_properties = element('w:tcPr')
|
||||||
|
cell_properties.append(element('w:tcW', **{'w:w': widths[column], 'w:type': 'dxa'}))
|
||||||
|
if number == 0:
|
||||||
|
cell_properties.append(element('w:shd', **{'w:val': 'clear', 'w:color': 'auto', 'w:fill': TINT}))
|
||||||
|
cell.append(cell_properties)
|
||||||
|
paragraph = element('w:p')
|
||||||
|
paragraph_properties = element('w:pPr')
|
||||||
|
paragraph_properties.append(element('w:spacing', **{'w:before': 0, 'w:after': 0}))
|
||||||
|
paragraph.append(paragraph_properties)
|
||||||
|
template = element('w:r')
|
||||||
|
run_properties = element('w:rPr')
|
||||||
|
fonts = element('w:rFonts', **{'w:ascii': THAI_FONT, 'w:hAnsi': THAI_FONT, 'w:cs': THAI_FONT, 'w:eastAsia': THAI_FONT})
|
||||||
|
run_properties.append(fonts)
|
||||||
|
run_properties.append(element('w:color', **{'w:val': '111111'}))
|
||||||
|
run_properties.append(element('w:sz', **{'w:val': TABLE_PT * 2}))
|
||||||
|
run_properties.append(element('w:szCs', **{'w:val': TABLE_PT * 2}))
|
||||||
|
template.append(run_properties)
|
||||||
|
paragraph.append(template)
|
||||||
|
write_text(paragraph, cells[column] if column < len(cells) else '')
|
||||||
|
cell.append(paragraph)
|
||||||
|
row.append(cell)
|
||||||
|
table.append(row)
|
||||||
|
return table
|
||||||
|
|
||||||
|
|
||||||
|
def restore_dropped_tables(document, markdown):
|
||||||
|
"""Put back tables the converter left out, where the source places them.
|
||||||
|
|
||||||
|
The converter occasionally drops a whole table that is plainly printed in
|
||||||
|
the PDF. It is rebuilt from the source in the style of the converted
|
||||||
|
tables, after the paragraph it follows in the source.
|
||||||
|
"""
|
||||||
|
restored = 0
|
||||||
|
body = document.element.body
|
||||||
|
for table in dropped_tables(document, markdown):
|
||||||
|
if not table['after'] or not table['rows']:
|
||||||
|
continue
|
||||||
|
anchor_key = letters(table['after'])
|
||||||
|
paragraphs = [child for child in body if child.tag == qn('w:p')]
|
||||||
|
anchor = next((child for child in paragraphs if letters(text_of(child)) == anchor_key), None)
|
||||||
|
if anchor is None:
|
||||||
|
# The block may share a paragraph with what precedes it.
|
||||||
|
endings = [child for child in paragraphs if anchor_key and letters(text_of(child)).endswith(anchor_key)]
|
||||||
|
anchor = endings[0] if len(endings) == 1 else None
|
||||||
|
if anchor is None:
|
||||||
|
continue
|
||||||
|
anchor.addnext(build_table(document, table['rows']))
|
||||||
|
restored += 1
|
||||||
|
return restored
|
||||||
|
|
||||||
|
|
||||||
def style_run(run, points):
|
def style_run(run, points):
|
||||||
run.font.size = Pt(points)
|
run.font.size = Pt(points)
|
||||||
run.font.color.rgb = INK
|
run.font.color.rgb = INK
|
||||||
@@ -201,6 +733,14 @@ def style_run(run, points):
|
|||||||
run._element.get_or_add_rPr().insert(0, fonts)
|
run._element.get_or_add_rPr().insert(0, fonts)
|
||||||
|
|
||||||
|
|
||||||
|
def drop_style(paragraph):
|
||||||
|
# The template's Header and Footer styles carry centre and right tab stops
|
||||||
|
# of their own, which catch the tab before the footer tag and centre it.
|
||||||
|
properties = paragraph._p.pPr
|
||||||
|
if properties is not None and properties.find(qn('w:pStyle')) is not None:
|
||||||
|
properties.remove(properties.find(qn('w:pStyle')))
|
||||||
|
|
||||||
|
|
||||||
def add_top_border(paragraph):
|
def add_top_border(paragraph):
|
||||||
borders = OxmlElement('w:pBdr')
|
borders = OxmlElement('w:pBdr')
|
||||||
top = OxmlElement('w:top')
|
top = OxmlElement('w:top')
|
||||||
@@ -225,6 +765,7 @@ def install_letterhead(document, band, footer_tag):
|
|||||||
section = document.sections[0]
|
section = document.sections[0]
|
||||||
|
|
||||||
header = section.header.paragraphs[0]
|
header = section.header.paragraphs[0]
|
||||||
|
drop_style(header)
|
||||||
# The band spans the page, so it starts outside the text block.
|
# The band spans the page, so it starts outside the text block.
|
||||||
header.paragraph_format.left_indent = -section.left_margin
|
header.paragraph_format.left_indent = -section.left_margin
|
||||||
header.paragraph_format.space_before = Pt(0)
|
header.paragraph_format.space_before = Pt(0)
|
||||||
@@ -232,6 +773,7 @@ def install_letterhead(document, band, footer_tag):
|
|||||||
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
|
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
|
||||||
|
|
||||||
footer = section.footer.paragraphs[0]
|
footer = section.footer.paragraphs[0]
|
||||||
|
drop_style(footer)
|
||||||
footer.paragraph_format.space_before = Pt(0)
|
footer.paragraph_format.space_before = Pt(0)
|
||||||
footer.paragraph_format.space_after = Pt(0)
|
footer.paragraph_format.space_after = Pt(0)
|
||||||
footer.paragraph_format.tab_stops.add_tab_stop(
|
footer.paragraph_format.tab_stops.add_tab_stop(
|
||||||
@@ -243,7 +785,7 @@ def install_letterhead(document, band, footer_tag):
|
|||||||
style_run(footer.add_run(f'\t{footer_tag}'), 9)
|
style_run(footer.add_run(f'\t{footer_tag}'), 9)
|
||||||
|
|
||||||
|
|
||||||
def convert(pdf, band, footer_tag):
|
def convert(pdf, source, band, footer_tag):
|
||||||
docx = pdf.with_suffix('.docx')
|
docx = pdf.with_suffix('.docx')
|
||||||
converter = Converter(str(pdf))
|
converter = Converter(str(pdf))
|
||||||
try:
|
try:
|
||||||
@@ -254,8 +796,16 @@ def convert(pdf, band, footer_tag):
|
|||||||
document = Document(str(docx))
|
document = Document(str(docx))
|
||||||
strip_reconstructed_furniture(document)
|
strip_reconstructed_furniture(document)
|
||||||
strip_logo_images(document)
|
strip_logo_images(document)
|
||||||
drop_padding(document)
|
bullets = unwrap_list_tables(document) + take_bullet_images(document)
|
||||||
normalise_ooxml(document)
|
normalise_ooxml(document)
|
||||||
|
normalise_fonts(document)
|
||||||
|
markdown = source.read_text(encoding='utf8')
|
||||||
|
restore_text(document, markdown, bullets)
|
||||||
|
restore_dropped_tables(document, markdown)
|
||||||
|
for paragraph in bullets:
|
||||||
|
if paragraph.getparent() is not None:
|
||||||
|
make_bullet(document, paragraph)
|
||||||
|
drop_padding(document)
|
||||||
match_rule_weight(document)
|
match_rule_weight(document)
|
||||||
install_letterhead(document, band, footer_tag)
|
install_letterhead(document, band, footer_tag)
|
||||||
document.save(str(docx))
|
document.save(str(docx))
|
||||||
@@ -286,7 +836,7 @@ def main(argv):
|
|||||||
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
|
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
|
||||||
pdf = delivery_root / Path(relative).with_suffix('.pdf')
|
pdf = delivery_root / Path(relative).with_suffix('.pdf')
|
||||||
try:
|
try:
|
||||||
docx = convert(pdf, Path(band['band']), band['footer'])
|
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'])
|
||||||
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
|
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
|
||||||
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
||||||
failures.append((pdf, error))
|
failures.append((pdf, error))
|
||||||
|
|||||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Reference in New Issue
Block a user