Draw the Word tables the way the PDF prints them
This commit is contained in:
@@ -20,6 +20,7 @@ Usage:
|
||||
convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY]
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
@@ -39,6 +40,8 @@ BODY_TOP_MM = 32
|
||||
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
|
||||
BAND_WIDTH_MM = 210
|
||||
STANDARD = 'ISO/IEC 29110-4-1:2018'
|
||||
RULE_SIZE = '4' # .5pt table rule, as printed
|
||||
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
||||
THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai"
|
||||
INK = RGBColor(0x11, 0x11, 0x11)
|
||||
|
||||
@@ -90,6 +93,78 @@ def strip_reconstructed_furniture(document):
|
||||
return removed
|
||||
|
||||
|
||||
DECIMAL = re.compile(r'^-?\d+\.\d+$')
|
||||
BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'}
|
||||
|
||||
|
||||
def normalise_ooxml(document):
|
||||
"""Repair the measurements the converter writes outside the schema.
|
||||
|
||||
Widths, sizes and border weights are whole units in OOXML, and a colour is
|
||||
six hex digits with no leading hash. The converter writes the numbers it
|
||||
measured off the page instead - "5.599999999999909", "#585858" - which Word
|
||||
is free to interpret as it likes, so borders come out at weights nobody
|
||||
asked for.
|
||||
"""
|
||||
fixed = 0
|
||||
for element in document.element.body.iter():
|
||||
for name, value in list(element.attrib.items()):
|
||||
local = name.rpartition('}')[2]
|
||||
if DECIMAL.match(value):
|
||||
element.set(name, str(round(float(value))))
|
||||
fixed += 1
|
||||
elif local in ('color', 'fill') and value.startswith('#'):
|
||||
element.set(name, value[1:])
|
||||
fixed += 1
|
||||
return fixed
|
||||
|
||||
|
||||
def match_rule_weight(document):
|
||||
"""Draw every table rule at the .5pt the PDF is printed with."""
|
||||
edges = 0
|
||||
for borders in document.element.body.iter():
|
||||
if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')):
|
||||
continue
|
||||
for edge in borders:
|
||||
if edge.tag.rpartition('}')[2] not in BORDER_EDGES:
|
||||
continue
|
||||
if edge.get(qn('w:val')) in (None, 'nil', 'none'):
|
||||
continue
|
||||
edge.set(qn('w:sz'), RULE_SIZE)
|
||||
edge.set(qn('w:color'), RULE_COLOUR)
|
||||
edges += 1
|
||||
return edges
|
||||
|
||||
|
||||
def drop_padding(document):
|
||||
"""Remove the blank paragraphs the converter pads each page out with.
|
||||
|
||||
A run of them at the foot of a page is pure whitespace - the section break
|
||||
already ends the page - and a run between two blocks only ever needs one.
|
||||
"""
|
||||
body = document.element.body
|
||||
children = list(body)
|
||||
removed = 0
|
||||
index = 0
|
||||
|
||||
while index < len(children):
|
||||
if not is_blank(children[index]):
|
||||
index += 1
|
||||
continue
|
||||
start = index
|
||||
while index < len(children) and is_blank(children[index]):
|
||||
index += 1
|
||||
following = children[index] if index < len(children) else None
|
||||
ends_page = following is None or (
|
||||
following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None
|
||||
)
|
||||
for element in children[start + (0 if ends_page else 1):index]:
|
||||
body.remove(element)
|
||||
removed += 1
|
||||
|
||||
return removed
|
||||
|
||||
|
||||
R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed'
|
||||
LOGO_MIN_PX = 100
|
||||
|
||||
@@ -179,6 +254,9 @@ def convert(pdf, band, footer_tag):
|
||||
document = Document(str(docx))
|
||||
strip_reconstructed_furniture(document)
|
||||
strip_logo_images(document)
|
||||
drop_padding(document)
|
||||
normalise_ooxml(document)
|
||||
match_rule_weight(document)
|
||||
install_letterhead(document, band, footer_tag)
|
||||
document.save(str(docx))
|
||||
return docx
|
||||
|
||||
Reference in New Issue
Block a user