docs(sdlc): content-sized columns in Word, trace record with test cases, PSRs presented at minuted meetings
This commit is contained in:
@@ -41,7 +41,6 @@ HERE = Path(__file__).resolve().parent
|
||||
# Page box measured from the audited reference package, as in sdlc-delivery.css.
|
||||
BODY_TOP_MM = 32
|
||||
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
|
||||
BAND_WIDTH_MM = 210
|
||||
STANDARD = 'ISO/IEC 29110-4-1:2018'
|
||||
RULE_SIZE = '4' # .5pt table rule, as printed
|
||||
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
|
||||
@@ -702,6 +701,68 @@ def build_table(document, rows):
|
||||
return table
|
||||
|
||||
|
||||
SIGNATURE_HEADER = tuple(letters(cell) for cell in ('ชื่อ', 'ตำแหน่ง', 'ลายเซ็น', 'วันที่'))
|
||||
|
||||
|
||||
def apply_column_widths(document, sized):
|
||||
"""Give each table the column widths the PDF was printed with.
|
||||
|
||||
pdf2docx writes its own grid, usually equal columns, so a table that reads
|
||||
well in the PDF comes out cramped in Word. Tables are matched to the source
|
||||
by their header row; pdf2docx repeats the header on each page's piece of a
|
||||
long table, so every piece is matched. Returns the headers left unmatched.
|
||||
"""
|
||||
section = document.sections[0]
|
||||
landscape = section.page_width > section.page_height
|
||||
width = round(Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips)
|
||||
by_header = {tuple(letters(cell) for cell in table['header']): table['widths'] for table in sized}
|
||||
unmatched = []
|
||||
for table in document.element.body.iter(qn('w:tbl')):
|
||||
rows = table.findall(qn('w:tr'))
|
||||
if not rows:
|
||||
continue
|
||||
header = tuple(letters(text_of(cell)) for cell in rows[0].findall(qn('w:tc')))
|
||||
if len(header) < 2:
|
||||
continue
|
||||
percents = by_header.get(header)
|
||||
if percents is None:
|
||||
# The document-control block and the signature tables keep their own layout.
|
||||
if header[0] != letters('Document No') and header != SIGNATURE_HEADER:
|
||||
unmatched.append(' | '.join(header))
|
||||
continue
|
||||
total = sum(percents)
|
||||
twips = [round(width * p / total) for p in percents]
|
||||
properties = table.find(qn('w:tblPr'))
|
||||
for tag, attributes in (('w:tblW', {'w:w': width, 'w:type': 'dxa'}), ('w:tblLayout', {'w:type': 'fixed'})):
|
||||
node = properties.find(qn(tag))
|
||||
if node is None:
|
||||
node = OxmlElement(tag)
|
||||
properties.append(node)
|
||||
for name, value in attributes.items():
|
||||
node.set(qn(name), str(value))
|
||||
grid = table.find(qn('w:tblGrid'))
|
||||
for column in list(grid):
|
||||
grid.remove(column)
|
||||
for value in twips:
|
||||
column = OxmlElement('w:gridCol')
|
||||
column.set(qn('w:w'), str(value))
|
||||
grid.append(column)
|
||||
for row in rows:
|
||||
index = 0
|
||||
for cell in row.findall(qn('w:tc')):
|
||||
cell_properties = cell.get_or_add_tcPr()
|
||||
span = cell_properties.find(qn('w:gridSpan'))
|
||||
count = int(span.get(qn('w:val'))) if span is not None else 1
|
||||
cell_width = cell_properties.find(qn('w:tcW'))
|
||||
if cell_width is None:
|
||||
cell_width = OxmlElement('w:tcW')
|
||||
cell_properties.insert(0, cell_width)
|
||||
cell_width.set(qn('w:w'), str(sum(twips[index:index + count])))
|
||||
cell_width.set(qn('w:type'), 'dxa')
|
||||
index += count
|
||||
return unmatched
|
||||
|
||||
|
||||
def restore_dropped_tables(document, markdown):
|
||||
"""Put back tables the converter left out, where the source places them.
|
||||
|
||||
@@ -774,7 +835,7 @@ def install_letterhead(document, band, footer_tag):
|
||||
header.paragraph_format.left_indent = -section.left_margin
|
||||
header.paragraph_format.space_before = Pt(0)
|
||||
header.paragraph_format.space_after = Pt(0)
|
||||
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
|
||||
header.add_run().add_picture(str(band), width=section.page_width)
|
||||
|
||||
footer = section.footer.paragraphs[0]
|
||||
drop_style(footer)
|
||||
@@ -789,7 +850,7 @@ def install_letterhead(document, band, footer_tag):
|
||||
style_run(footer.add_run(f'\t{footer_tag}'), 9)
|
||||
|
||||
|
||||
def convert(pdf, source, band, footer_tag):
|
||||
def convert(pdf, source, band, footer_tag, sized=()):
|
||||
docx = pdf.with_suffix('.docx')
|
||||
converter = Converter(str(pdf))
|
||||
try:
|
||||
@@ -806,6 +867,8 @@ def convert(pdf, source, band, footer_tag):
|
||||
markdown = source.read_text(encoding='utf8')
|
||||
restore_text(document, markdown, bullets)
|
||||
restore_dropped_tables(document, markdown)
|
||||
for header in apply_column_widths(document, sized):
|
||||
print(f' WARNING {pdf.name}: no source widths for table "{header[:80]}"', file=sys.stderr)
|
||||
for paragraph in bullets:
|
||||
if paragraph.getparent() is not None:
|
||||
make_bullet(document, paragraph)
|
||||
@@ -816,9 +879,9 @@ def convert(pdf, source, band, footer_tag):
|
||||
return docx
|
||||
|
||||
|
||||
def render_bands(source_root, band_root):
|
||||
def render_bands(source_root, band_root, pdf_root):
|
||||
subprocess.run(
|
||||
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
|
||||
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root), str(pdf_root)],
|
||||
check=True
|
||||
)
|
||||
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
|
||||
@@ -831,16 +894,20 @@ def main(argv):
|
||||
source_root, delivery_root = Path(argv[0]), Path(argv[1])
|
||||
temporary = None
|
||||
if len(argv) > 2:
|
||||
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
|
||||
band_root = Path(argv[2])
|
||||
bands = json.loads((band_root / 'bands.json').read_text(encoding='utf8'))
|
||||
else:
|
||||
temporary = tempfile.TemporaryDirectory()
|
||||
bands = render_bands(source_root, temporary.name)
|
||||
band_root = Path(temporary.name)
|
||||
bands = render_bands(source_root, temporary.name, delivery_root)
|
||||
|
||||
failures = []
|
||||
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
|
||||
pdf = delivery_root / Path(relative).with_suffix('.pdf')
|
||||
try:
|
||||
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'])
|
||||
widths = band_root / 'widths' / Path(relative).with_suffix('.json')
|
||||
sized = json.loads(widths.read_text(encoding='utf8')) if widths.exists() else []
|
||||
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'], sized)
|
||||
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
|
||||
except Exception as error: # noqa: BLE001 - reported, not swallowed
|
||||
failures.append((pdf, error))
|
||||
|
||||
Reference in New Issue
Block a user