docs(sdlc): content-sized columns in Word, trace record with test cases, PSRs presented at minuted meetings

This commit is contained in:
Thanakorn
2026-10-02 16:35:31 +07:00
parent 64b1be55d8
commit 203b541c59
139 changed files with 322 additions and 170 deletions
+75 -8
View File
@@ -41,7 +41,6 @@ HERE = Path(__file__).resolve().parent
# Page box measured from the audited reference package, as in sdlc-delivery.css.
BODY_TOP_MM = 32
FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page
BAND_WIDTH_MM = 210
STANDARD = 'ISO/IEC 29110-4-1:2018'
RULE_SIZE = '4' # .5pt table rule, as printed
RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css
@@ -702,6 +701,68 @@ def build_table(document, rows):
return table
SIGNATURE_HEADER = tuple(letters(cell) for cell in ('ชื่อ', 'ตำแหน่ง', 'ลายเซ็น', 'วันที่'))
def apply_column_widths(document, sized):
"""Give each table the column widths the PDF was printed with.
pdf2docx writes its own grid, usually equal columns, so a table that reads
well in the PDF comes out cramped in Word. Tables are matched to the source
by their header row; pdf2docx repeats the header on each page's piece of a
long table, so every piece is matched. Returns the headers left unmatched.
"""
section = document.sections[0]
landscape = section.page_width > section.page_height
width = round(Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips)
by_header = {tuple(letters(cell) for cell in table['header']): table['widths'] for table in sized}
unmatched = []
for table in document.element.body.iter(qn('w:tbl')):
rows = table.findall(qn('w:tr'))
if not rows:
continue
header = tuple(letters(text_of(cell)) for cell in rows[0].findall(qn('w:tc')))
if len(header) < 2:
continue
percents = by_header.get(header)
if percents is None:
# The document-control block and the signature tables keep their own layout.
if header[0] != letters('Document No') and header != SIGNATURE_HEADER:
unmatched.append(' | '.join(header))
continue
total = sum(percents)
twips = [round(width * p / total) for p in percents]
properties = table.find(qn('w:tblPr'))
for tag, attributes in (('w:tblW', {'w:w': width, 'w:type': 'dxa'}), ('w:tblLayout', {'w:type': 'fixed'})):
node = properties.find(qn(tag))
if node is None:
node = OxmlElement(tag)
properties.append(node)
for name, value in attributes.items():
node.set(qn(name), str(value))
grid = table.find(qn('w:tblGrid'))
for column in list(grid):
grid.remove(column)
for value in twips:
column = OxmlElement('w:gridCol')
column.set(qn('w:w'), str(value))
grid.append(column)
for row in rows:
index = 0
for cell in row.findall(qn('w:tc')):
cell_properties = cell.get_or_add_tcPr()
span = cell_properties.find(qn('w:gridSpan'))
count = int(span.get(qn('w:val'))) if span is not None else 1
cell_width = cell_properties.find(qn('w:tcW'))
if cell_width is None:
cell_width = OxmlElement('w:tcW')
cell_properties.insert(0, cell_width)
cell_width.set(qn('w:w'), str(sum(twips[index:index + count])))
cell_width.set(qn('w:type'), 'dxa')
index += count
return unmatched
def restore_dropped_tables(document, markdown):
"""Put back tables the converter left out, where the source places them.
@@ -774,7 +835,7 @@ def install_letterhead(document, band, footer_tag):
header.paragraph_format.left_indent = -section.left_margin
header.paragraph_format.space_before = Pt(0)
header.paragraph_format.space_after = Pt(0)
header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM))
header.add_run().add_picture(str(band), width=section.page_width)
footer = section.footer.paragraphs[0]
drop_style(footer)
@@ -789,7 +850,7 @@ def install_letterhead(document, band, footer_tag):
style_run(footer.add_run(f'\t{footer_tag}'), 9)
def convert(pdf, source, band, footer_tag):
def convert(pdf, source, band, footer_tag, sized=()):
docx = pdf.with_suffix('.docx')
converter = Converter(str(pdf))
try:
@@ -806,6 +867,8 @@ def convert(pdf, source, band, footer_tag):
markdown = source.read_text(encoding='utf8')
restore_text(document, markdown, bullets)
restore_dropped_tables(document, markdown)
for header in apply_column_widths(document, sized):
print(f' WARNING {pdf.name}: no source widths for table "{header[:80]}"', file=sys.stderr)
for paragraph in bullets:
if paragraph.getparent() is not None:
make_bullet(document, paragraph)
@@ -816,9 +879,9 @@ def convert(pdf, source, band, footer_tag):
return docx
def render_bands(source_root, band_root):
def render_bands(source_root, band_root, pdf_root):
subprocess.run(
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)],
['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root), str(pdf_root)],
check=True
)
return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8'))
@@ -831,16 +894,20 @@ def main(argv):
source_root, delivery_root = Path(argv[0]), Path(argv[1])
temporary = None
if len(argv) > 2:
bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8'))
band_root = Path(argv[2])
bands = json.loads((band_root / 'bands.json').read_text(encoding='utf8'))
else:
temporary = tempfile.TemporaryDirectory()
bands = render_bands(source_root, temporary.name)
band_root = Path(temporary.name)
bands = render_bands(source_root, temporary.name, delivery_root)
failures = []
for index, (relative, band) in enumerate(sorted(bands.items()), start=1):
pdf = delivery_root / Path(relative).with_suffix('.pdf')
try:
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'])
widths = band_root / 'widths' / Path(relative).with_suffix('.json')
sized = json.loads(widths.read_text(encoding='utf8')) if widths.exists() else []
docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'], sized)
print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}')
except Exception as error: # noqa: BLE001 - reported, not swallowed
failures.append((pdf, error))