#!/usr/bin/env python3 """Convert the rendered delivery package from PDF to Word. The work products are authored in Markdown and rendered to PDF by render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that PDF, so the Word file keeps the page the reviewer signed off rather than being laid out a second time from the source. The corporate identity is the one thing not taken from the reconstruction. A PDF has no notion of a running header, so the converter rebuilds the CI band and the footer rule as ordinary content at the top and bottom of every page, losing the ribbon geometry and the alignment with it. Both are therefore stripped out again here and replaced by a real Word header and footer: the band is the same assets/header.html the PDF is printed from, captured by render-ci-band.mjs, so it lands in the same place on every page. Requires pdf2docx: pip install -r requirements.txt Usage: convert-to-docx.py SOURCE_DIRECTORY DELIVERY_DIRECTORY [BAND_DIRECTORY] """ import copy import json import re import subprocess import sys import tempfile import unicodedata from pathlib import Path from docx import Document from docx.enum.text import WD_TAB_ALIGNMENT from docx.oxml import OxmlElement from docx.oxml.ns import qn from docx.shared import Mm, Pt, RGBColor, Twips from docx.text.paragraph import Paragraph from pdf2docx import Converter HERE = Path(__file__).resolve().parent # Page box measured from the audited reference package, as in sdlc-delivery.css. BODY_TOP_MM = 32 FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page BAND_WIDTH_MM = 210 STANDARD = 'ISO/IEC 29110-4-1:2018' RULE_SIZE = '4' # .5pt table rule, as printed RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css THAI_FONT = 'Leelawadee UI' # the face embedded in the PDF as "BRN Thai" MONO_FONT = 'Courier New' # code spans; DejaVu Sans Mono in the PDF INK = RGBColor(0x11, 0x11, 0x11) # The furniture pdf2docx rebuilds in the body, recognised by text that appears # nowhere in the authored sources. BAND_MARKS = ('B.R.N. ENTERPRISE', STANDARD) LINE_MARKS = ('Tel. (+66)', '1011 Supalai Grand Tower') def text_of(element): """The text Word shows for an element: its w:t runs, tabs and breaks. Not itertext(): the converter leaves other text-bearing nodes in its runs, which would read every paragraph twice over. """ parts = [] for node in element.iter(qn('w:t'), qn('w:tab'), qn('w:br')): if node.tag == qn('w:t'): parts.append(node.text or '') elif node.getparent().tag == qn('w:r'): # A w:tab under w:tabs is a tab-stop definition, not a character. parts.append('\t' if node.tag == qn('w:tab') else '\n') return ''.join(parts) def is_blank(element): return ( element.tag == qn('w:p') and not text_of(element).strip() and element.find('.//' + qn('w:sectPr')) is None and not element.findall('.//' + qn('w:drawing')) ) def strip_reconstructed_furniture(document): """Remove the per-page letterhead and footer rule pdf2docx puts in the body.""" body = document.element.body removed = 0 for block in list(body): if block.getparent() is None: continue if block.tag == qn('w:tbl'): # The converter sometimes lays a page's furniture and its content # into one table, so furniture is taken out a row at a time. for row in list(block.findall(qn('w:tr'))): if any(mark in text_of(row) for mark in BAND_MARKS + LINE_MARKS): block.remove(row) removed += 1 if block.find(qn('w:tr')) is None: previous = block.getprevious() if previous is not None and is_blank(previous): body.remove(previous) body.remove(block) elif block.tag == qn('w:p') and any(mark in text_of(block) for mark in LINE_MARKS): body.remove(block) removed += 1 return removed DECIMAL = re.compile(r'^-?\d+\.\d+$') BORDER_EDGES = {'top', 'bottom', 'left', 'right', 'insideH', 'insideV', 'tl2br', 'tr2bl'} def normalise_ooxml(document): """Repair the measurements the converter writes outside the schema. Widths, sizes and border weights are whole units in OOXML, and a colour is six hex digits with no leading hash. The converter writes the numbers it measured off the page instead - "5.599999999999909", "#585858" - which Word is free to interpret as it likes, so borders come out at weights nobody asked for. """ fixed = 0 for element in document.element.body.iter(): for name, value in list(element.attrib.items()): local = name.rpartition('}')[2] if DECIMAL.match(value): element.set(name, str(round(float(value)))) fixed += 1 elif local in ('color', 'fill') and value.startswith('#'): element.set(name, value[1:]) fixed += 1 return fixed def match_rule_weight(document): """Draw every table rule at the .5pt the PDF is printed with.""" edges = 0 for borders in document.element.body.iter(): if borders.tag not in (qn('w:tblBorders'), qn('w:tcBorders')): continue for edge in borders: if edge.tag.rpartition('}')[2] not in BORDER_EDGES: continue if edge.get(qn('w:val')) in (None, 'nil', 'none'): continue edge.set(qn('w:sz'), RULE_SIZE) edge.set(qn('w:color'), RULE_COLOUR) edges += 1 return edges def drop_padding(document): """Remove the blank paragraphs the converter pads each page out with. A run of them at the foot of a page is pure whitespace - the section break already ends the page - and a run between two blocks only ever needs one. """ body = document.element.body children = list(body) removed = 0 index = 0 while index < len(children): if not is_blank(children[index]): index += 1 continue start = index while index < len(children) and is_blank(children[index]): index += 1 following = children[index] if index < len(children) else None ends_page = following is None or ( following.tag == qn('w:p') and following.find('.//' + qn('w:sectPr')) is not None ) for element in children[start + (0 if ends_page else 1):index]: body.remove(element) removed += 1 return removed R_EMBED = '{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed' LOGO_MIN_PX = 100 def strip_logo_images(document): """Drop the letterhead logo left loose in the body. The work products carry no figures of their own, but the converter draws each bullet as a small image, so only the logo is removed - anything that size would be furniture, anything smaller is a list marker. """ logos = { relationship_id for relationship_id, part in document.part.related_parts.items() if getattr(part, 'image', None) is not None and max(part.image.px_width, part.image.px_height) >= LOGO_MIN_PX } removed = 0 for drawing in list(document.element.body.iter(qn('w:drawing'))): blip = drawing.find('.//' + qn('a:blip')) if blip is not None and blip.get(R_EMBED) in logos: drawing.getparent().remove(drawing) removed += 1 return removed # --------------------------------------------------------------------------- # Fonts # --------------------------------------------------------------------------- def normalise_fonts(document): """Point every run at a face Word actually has. The PDF embeds its faces as Type3 fonts, so the converter names runs after the PDF objects - "Type3 (4 0 R)" - and gives them no complex-script face or size. Word then sets Latin text in a fallback serif and Thai text at the default size, which is the mixed, jumping type the conversion showed. """ for run in document.element.body.iter(qn('w:r')): properties = run.find(qn('w:rPr')) if properties is None: properties = OxmlElement('w:rPr') run.insert(0, properties) fonts = properties.find(qn('w:rFonts')) face = MONO_FONT if fonts is not None and 'Mono' in (fonts.get(qn('w:ascii')) or '') else THAI_FONT if fonts is None: fonts = OxmlElement('w:rFonts') style = properties.find(qn('w:rStyle')) properties.insert(1 if style is not None else 0, fonts) for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'): fonts.set(qn(attribute), face) size = properties.find(qn('w:sz')) if size is not None and properties.find(qn('w:szCs')) is None: complex_size = OxmlElement('w:szCs') complex_size.set(qn('w:val'), size.get(qn('w:val'))) size.addnext(complex_size) # --------------------------------------------------------------------------- # Bullets # --------------------------------------------------------------------------- BULLET = '•' BULLET_TEXT_MM = 7 # ul padding-left in sdlc-delivery.css BULLET_HANG_MM = 4 def text_block_left_mm(section): # The PDF text block starts 30mm in, or 15mm on the landscape work products. return 15 if section.page_width > section.page_height else 30 def unwrap_list_tables(document): """Turn the tables the converter builds out of bullet lists back into text. A short list is sometimes laid out as a grid - bullet image, item, empty cell - with stray bottom borders that print as loose rules. Each row goes back into the body as a paragraph of its own. """ body = document.element.body items = [] for table in [child for child in body if child.tag == qn('w:tbl')]: rows = table.findall(qn('w:tr')) entries = [] for row in rows: cells = row.findall(qn('w:tc')) worded = [cell for cell in cells if text_of(cell).strip()] marked = [cell for cell in cells if cell.findall('.//' + qn('w:drawing')) and not text_of(cell).strip()] if len(worded) != 1 or not marked: entries = None break entries.append(worded[0]) if not rows or entries is None: continue for cell in entries: for paragraph in cell.findall(qn('w:p')): if text_of(paragraph).strip(): table.addprevious(paragraph) items.append(paragraph) body.remove(table) return items def take_bullet_images(document): """Collect the list items drawn with an image marker, dropping the image.""" items = [] for paragraph in [child for child in document.element.body if child.tag == qn('w:p')]: drawings = paragraph.findall('.//' + qn('w:drawing')) if not drawings or not text_of(paragraph).strip(): continue for drawing in drawings: run = drawing.getparent() run.remove(drawing) if run.tag == qn('w:r') and not text_of(run).strip(): run.getparent().remove(run) items.append(paragraph) return items def make_bullet(document, paragraph): section = document.sections[0] item = Paragraph(paragraph, document._body) text_mm = text_block_left_mm(section) + BULLET_TEXT_MM item.paragraph_format.left_indent = Twips(round(Mm(text_mm).twips - section.left_margin.twips)) item.paragraph_format.first_line_indent = -Mm(BULLET_HANG_MM) first = paragraph.find(qn('w:r')) marker = OxmlElement('w:r') if first is not None and first.find(qn('w:rPr')) is not None: marker.append(copy.deepcopy(first.find(qn('w:rPr')))) text = OxmlElement('w:t') text.text = BULLET marker.append(text) marker.append(OxmlElement('w:tab')) properties = paragraph.find(qn('w:pPr')) if properties is not None: properties.addnext(marker) else: paragraph.insert(0, marker) # --------------------------------------------------------------------------- # Text from the source # --------------------------------------------------------------------------- APPROVAL_FIELD = re.compile(r'^(Name|Role|Signature|Date|Position|Company|Project roles|Decision):') SEPARATOR_CELL = re.compile(r'^:?-{3,}:?$') INLINE = re.compile(r'\*\*(.+?)\*\*|`([^`]+)`|\\([\\*_`~])') def letters(text): """The characters two renderings of the same text must share.""" return ''.join( character for character in unicodedata.normalize('NFC', text) if unicodedata.category(character)[0] in 'LMN' ).lower() def read_source(markdown): """Blocks and table cells of a work product, as the reader sees them.""" body = re.sub(r'^#\s+.+\n+', '', markdown, count=1, flags=re.M) body = re.sub(r'^\s*$', '', body, flags=re.M) blocks, cells, tables, paragraph = [], [], [], [] bullets = set() table = None def flush(): if paragraph: joined = '' for index, line in enumerate(paragraph): hard = re.search(r'\s{2,}$', line) or APPROVAL_FIELD.match(line.strip()) joined += line.strip() + ('\n' if hard else (' ' if index < len(paragraph) - 1 else '')) blocks.append(joined.rstrip('\n')) paragraph.clear() for line in body.splitlines(): stripped = line.strip() if stripped.startswith('|') and stripped.endswith('|'): flush() parts = [part.strip() for part in stripped[1:-1].split('|')] if table is None: table = {'after': blocks[-1] if blocks else None, 'rows': []} tables.append(table) if not all(SEPARATOR_CELL.match(part) for part in parts): row = [re.sub(r'', '\n', part) for part in parts] cells.extend(row) table['rows'].append(row) continue table = None if not stripped: flush() continue heading = re.match(r'^#{1,3}\s+(.+)$', stripped) bullet = re.match(r'^[-*]\s+(.+)$', stripped) ordered = re.match(r'^\d+\.\s+.+$', stripped) if heading or bullet or ordered: flush() # A numbered item keeps its number: the PDF prints it as text. blocks.append(heading.group(1) if heading else bullet.group(1) if bullet else stripped) if bullet: bullets.add(len(blocks) - 1) continue paragraph.append(line) flush() return blocks, cells, tables, bullets def index_texts(texts): exact, by_length = {}, {} for text in texts: key = letters(text) if key and key not in exact: exact[key] = text by_length.setdefault(len(key), []).append(key) return exact, by_length def is_subsequence(short, long): remaining = iter(long) return all(character in remaining for character in short) def source_for(text, exact, by_length): """The authored text a converted fragment came from, if it is unambiguous. Same letters means only spacing and punctuation differ. Failing that, a fragment that is the source with up to three characters missing - the converter drops characters, it never invents them - is accepted when just one source text fits. """ key = letters(text) if not key: return None if key in exact: return exact[key] if len(key) < 8: return None fits = { exact[candidate] for length in range(len(key) + 1, len(key) + 4) for candidate in by_length.get(length, ()) if is_subsequence(key, candidate) } return fits.pop() if len(fits) == 1 else None def write_text(paragraph, text): """Replace a paragraph's runs with the authored text, keeping its look.""" template = paragraph.find('.//' + qn('w:rPr')) template = copy.deepcopy(template) if template is not None else OxmlElement('w:rPr') for child in list(paragraph): if child.tag != qn('w:pPr'): paragraph.remove(child) def run(content=None, bold=False, code=False, line_break=False): element = OxmlElement('w:r') properties = copy.deepcopy(template) fonts = properties.find(qn('w:rFonts')) if fonts is not None: face = MONO_FONT if code else THAI_FONT for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'): fonts.set(qn(attribute), face) weight = properties.find(qn('w:b')) for complex_weight in properties.findall(qn('w:bCs')): properties.remove(complex_weight) if bold: if weight is None: weight = OxmlElement('w:b') (fonts.addnext(weight) if fonts is not None else properties.insert(0, weight)) weight.attrib.pop(qn('w:val'), None) weight.addnext(OxmlElement('w:bCs')) elif weight is not None: weight.set(qn('w:val'), '0') element.append(properties) if line_break: element.append(OxmlElement('w:br')) else: node = OxmlElement('w:t') node.text = content node.set('{http://www.w3.org/XML/1998/namespace}space', 'preserve') element.append(node) paragraph.append(element) for number, line in enumerate(text.split('\n')): if number: run(line_break=True) cursor = 0 for match in INLINE.finditer(line): if match.start() > cursor: run(line[cursor:match.start()]) if match.group(1) is not None: run(match.group(1), bold=True) elif match.group(2) is not None: run(match.group(2), code=True) else: run(match.group(3)) cursor = match.end() if cursor < len(line): run(line[cursor:]) def split_blocks(key, block_keys): """The run of consecutive source blocks a converted paragraph is made of.""" for start, first in enumerate(block_keys): if not first or not key.startswith(first): continue combined, end = first, start while len(combined) < len(key) and end + 1 < len(block_keys) and end - start < 7: end += 1 combined += block_keys[end] if combined == key: return list(range(start, end + 1)) if not key.startswith(combined): break return None def restore_text(document, markdown, bullets): """Put the authored wording back where the converter mangled it. PDF text has no word spaces for Thai, so the conversion runs Thai words together, breaks a paragraph into one paragraph per printed line, and now and then drops a character. Wherever a converted paragraph or table cell can be traced to exactly one piece of the source, its text is rewritten from the source; anything that cannot be traced is left as converted. """ blocks, cells, _, bullet_blocks = read_source(markdown) block_keys = [letters(block) for block in blocks] known = {id(paragraph) for paragraph in bullets} block_exact, block_lengths = index_texts(blocks) cell_exact, cell_lengths = index_texts(cells) body = document.element.body rewritten = merged = 0 for cell in body.iter(qn('w:tc')): if cell.find('.//' + qn('w:tbl')) is not None: continue paragraphs = cell.findall(qn('w:p')) text = '\n'.join(text_of(paragraph) for paragraph in paragraphs) source = source_for(text, cell_exact, cell_lengths) if source is None or '___' in source or source == text: continue write_text(paragraphs[0], source) for extra in paragraphs[1:]: cell.remove(extra) rewritten += 1 children = [child for child in body if child.tag == qn('w:p') and text_of(child).strip()] index = 0 while index < len(children): paragraph = children[index] source = source_for(text_of(paragraph), block_exact, block_lengths) span = 1 if source is None: # A paragraph the converter split into one paragraph per line. combined = text_of(paragraph) follower = paragraph for count in range(2, 9): follower = follower.getnext() if follower is None or follower.tag != qn('w:p') or not text_of(follower).strip(): break combined += text_of(follower) if letters(combined) in block_exact: source, span = block_exact[letters(combined)], count break if source is None: # Several list items or paragraphs the converter ran into one. parts = split_blocks(letters(text_of(paragraph)), block_keys) if parts and not any('___' in blocks[part] for part in parts): previous = paragraph for number, part in enumerate(parts): target = paragraph if number: target = OxmlElement('w:p') if paragraph.find(qn('w:pPr')) is not None: target.append(copy.deepcopy(paragraph.find(qn('w:pPr')))) template = paragraph.find('.//' + qn('w:rPr')) run = OxmlElement('w:r') run.append(copy.deepcopy(template) if template is not None else OxmlElement('w:rPr')) target.append(run) previous.addnext(target) previous = target write_text(target, blocks[part]) if part in bullet_blocks and id(target) not in known: bullets.append(target) known.add(id(target)) rewritten += len(parts) index += 1 continue if source is not None and '___' not in source: for extra in children[index + 1:index + span]: body.remove(extra) merged += 1 write_text(paragraph, source) rewritten += 1 index += span return rewritten, merged TINT = 'FFF3F6' # header-row tint used across the reference tables TABLE_PT = 9 CELL_MARGIN_MM = (1.6, 2) # vertical, horizontal padding of a table cell PRINTED_TEXT_WIDTH_MM = {False: 160, True: 267} # portrait, landscape DISTINCTIVE_LETTERS = 15 # shorter cells - labels, IDs, dates - recur elsewhere DROPPED_CHARACTER_SLACK = 3 def present(key, document_key): """Whether a cell's whole text reached the document. The whole cell, not a fragment of it: short Thai phrases recur across a work product, so a partial match proves nothing. Its two halves in order, a few characters apart at most, still count, as the converter sometimes drops a character mid-cell. """ if key in document_key: return True half = len(key) // 2 start = document_key.find(key[:half]) while start >= 0: second = document_key.find(key[half:], start + half) if 0 <= second - (start + half) <= DROPPED_CHARACTER_SLACK: return True start = document_key.find(key[:half], start + 1) return False def dropped_tables(document, markdown): """Source tables of which not one distinctive cell reached the document.""" _, _, tables, _ = read_source(markdown) document_key = letters(text_of(document.element.body)) missing = [] for table in tables: keys = [letters(cell) for row in table['rows'] for cell in row] keys = [key for key in keys if len(key) >= DISTINCTIVE_LETTERS] if keys and not any(present(key, document_key) for key in keys): missing.append(table) return missing def build_table(document, rows): section = document.sections[0] landscape = section.page_width > section.page_height width = Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips columns = max(len(row) for row in rows) longest = [max(len(letters(row[i])) if i < len(row) else 0 for row in rows) for i in range(columns)] weights = [min(max(value, 8), 80) for value in longest] widths = [round(width * weight / sum(weights)) for weight in weights] def element(tag, **attributes): node = OxmlElement(tag) for name, value in attributes.items(): node.set(qn(name), str(value)) return node table = element('w:tbl') properties = element('w:tblPr') properties.append(element('w:tblW', **{'w:w': width, 'w:type': 'dxa'})) indent = round(Mm(text_block_left_mm(section)).twips - section.left_margin.twips) properties.append(element('w:tblInd', **{'w:w': indent, 'w:type': 'dxa'})) borders = element('w:tblBorders') for edge in ('top', 'left', 'bottom', 'right', 'insideH', 'insideV'): borders.append(element('w:' + edge, **{'w:val': 'single', 'w:sz': RULE_SIZE, 'w:space': 0, 'w:color': RULE_COLOUR})) properties.append(borders) properties.append(element('w:tblLayout', **{'w:type': 'fixed'})) margins = element('w:tblCellMar') vertical, horizontal = (round(Mm(value).twips) for value in CELL_MARGIN_MM) for edge, value in (('top', vertical), ('left', horizontal), ('bottom', vertical), ('right', horizontal)): margins.append(element('w:' + edge, **{'w:w': value, 'w:type': 'dxa'})) properties.append(margins) table.append(properties) grid = element('w:tblGrid') for column_width in widths: grid.append(element('w:gridCol', **{'w:w': column_width})) table.append(grid) for number, cells in enumerate(rows): row = element('w:tr') row_properties = element('w:trPr') row_properties.append(element('w:cantSplit')) if number == 0: row_properties.append(element('w:tblHeader')) row.append(row_properties) for column in range(columns): cell = element('w:tc') cell_properties = element('w:tcPr') cell_properties.append(element('w:tcW', **{'w:w': widths[column], 'w:type': 'dxa'})) if number == 0: cell_properties.append(element('w:shd', **{'w:val': 'clear', 'w:color': 'auto', 'w:fill': TINT})) cell.append(cell_properties) paragraph = element('w:p') paragraph_properties = element('w:pPr') paragraph_properties.append(element('w:spacing', **{'w:before': 0, 'w:after': 0})) paragraph.append(paragraph_properties) template = element('w:r') run_properties = element('w:rPr') fonts = element('w:rFonts', **{'w:ascii': THAI_FONT, 'w:hAnsi': THAI_FONT, 'w:cs': THAI_FONT, 'w:eastAsia': THAI_FONT}) run_properties.append(fonts) run_properties.append(element('w:color', **{'w:val': '111111'})) run_properties.append(element('w:sz', **{'w:val': TABLE_PT * 2})) run_properties.append(element('w:szCs', **{'w:val': TABLE_PT * 2})) template.append(run_properties) paragraph.append(template) write_text(paragraph, cells[column] if column < len(cells) else '') cell.append(paragraph) row.append(cell) table.append(row) return table def restore_dropped_tables(document, markdown): """Put back tables the converter left out, where the source places them. The converter occasionally drops a whole table that is plainly printed in the PDF. It is rebuilt from the source in the style of the converted tables, after the paragraph it follows in the source. """ restored = 0 body = document.element.body for table in dropped_tables(document, markdown): if not table['after'] or not table['rows']: continue anchor_key = letters(table['after']) paragraphs = [child for child in body if child.tag == qn('w:p')] anchor = next((child for child in paragraphs if letters(text_of(child)) == anchor_key), None) if anchor is None: # The block may share a paragraph with what precedes it. endings = [child for child in paragraphs if anchor_key and letters(text_of(child)).endswith(anchor_key)] anchor = endings[0] if len(endings) == 1 else None if anchor is None: continue anchor.addnext(build_table(document, table['rows'])) restored += 1 return restored def style_run(run, points): run.font.size = Pt(points) run.font.color.rgb = INK fonts = OxmlElement('w:rFonts') for attribute in ('w:ascii', 'w:hAnsi', 'w:cs', 'w:eastAsia'): fonts.set(qn(attribute), THAI_FONT) run._element.get_or_add_rPr().insert(0, fonts) def drop_style(paragraph): # The template's Header and Footer styles carry centre and right tab stops # of their own, which catch the tab before the footer tag and centre it. properties = paragraph._p.pPr if properties is not None and properties.find(qn('w:pStyle')) is not None: properties.remove(properties.find(qn('w:pStyle'))) def add_top_border(paragraph): borders = OxmlElement('w:pBdr') top = OxmlElement('w:top') top.set(qn('w:val'), 'single') top.set(qn('w:sz'), '6') # .75pt, as printed top.set(qn('w:space'), '1') top.set(qn('w:color'), '111111') borders.append(top) paragraph._p.get_or_add_pPr().append(borders) def install_letterhead(document, band, footer_tag): """Give every page the CI band and footer rule as real Word furniture.""" for index, section in enumerate(document.sections): section.top_margin = Mm(BODY_TOP_MM) section.header_distance = Mm(0) section.footer_distance = Mm(FOOTER_FROM_BOTTOM_MM) # Only the first section carries the furniture; the rest inherit it. section.header.is_linked_to_previous = index > 0 section.footer.is_linked_to_previous = index > 0 section = document.sections[0] header = section.header.paragraphs[0] drop_style(header) # The band spans the page, so it starts outside the text block. header.paragraph_format.left_indent = -section.left_margin header.paragraph_format.space_before = Pt(0) header.paragraph_format.space_after = Pt(0) header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM)) footer = section.footer.paragraphs[0] drop_style(footer) footer.paragraph_format.space_before = Pt(0) footer.paragraph_format.space_after = Pt(0) footer.paragraph_format.tab_stops.add_tab_stop( section.page_width - section.left_margin - section.right_margin, WD_TAB_ALIGNMENT.RIGHT ) add_top_border(footer) style_run(footer.add_run(STANDARD), 9) style_run(footer.add_run(f'\t{footer_tag}'), 9) def convert(pdf, source, band, footer_tag): docx = pdf.with_suffix('.docx') converter = Converter(str(pdf)) try: converter.convert(str(docx)) finally: converter.close() document = Document(str(docx)) strip_reconstructed_furniture(document) strip_logo_images(document) bullets = unwrap_list_tables(document) + take_bullet_images(document) normalise_ooxml(document) normalise_fonts(document) markdown = source.read_text(encoding='utf8') restore_text(document, markdown, bullets) restore_dropped_tables(document, markdown) for paragraph in bullets: if paragraph.getparent() is not None: make_bullet(document, paragraph) drop_padding(document) match_rule_weight(document) install_letterhead(document, band, footer_tag) document.save(str(docx)) return docx def render_bands(source_root, band_root): subprocess.run( ['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)], check=True ) return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8')) def main(argv): if len(argv) < 2: raise SystemExit(__doc__) source_root, delivery_root = Path(argv[0]), Path(argv[1]) temporary = None if len(argv) > 2: bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8')) else: temporary = tempfile.TemporaryDirectory() bands = render_bands(source_root, temporary.name) failures = [] for index, (relative, band) in enumerate(sorted(bands.items()), start=1): pdf = delivery_root / Path(relative).with_suffix('.pdf') try: docx = convert(pdf, source_root / relative, Path(band['band']), band['footer']) print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}') except Exception as error: # noqa: BLE001 - reported, not swallowed failures.append((pdf, error)) print(f'[{index}/{len(bands)}] FAILED {relative}: {error}', file=sys.stderr) if temporary: temporary.cleanup() print(f'{len(bands) - len(failures)}/{len(bands)} documents converted') if failures: raise SystemExit(1) if __name__ == '__main__': main(sys.argv[1:])