diff --git a/scripts/build-sdlc-delivery.sh b/scripts/build-sdlc-delivery.sh index 7907b37..4ecfaf0 100755 --- a/scripts/build-sdlc-delivery.sh +++ b/scripts/build-sdlc-delivery.sh @@ -26,22 +26,23 @@ node "$SCRIPT_DIR/sdlc-seed/build.mjs" rm -rf "$STAGING_DIR" mkdir -p "$STAGING_DIR" +# Scratch space for the CI bands and the table column widths handed to the converter. +BAND_DIR="$(mktemp -d)" +trap 'rm -rf "$BAND_DIR"' EXIT while IFS= read -r -d '' source; do relative="${source#"$ROOT_DIR/sdlc/"}" destination="$STAGING_DIR/${relative%.md}.pdf" echo "Rendering ${relative%.md}.pdf" - node "$TOOL_DIR/render-sdlc.mjs" "$source" "$destination" + SDLC_WIDTHS_FILE="$BAND_DIR/widths/${relative%.md}.json" node "$TOOL_DIR/render-sdlc.mjs" "$source" "$destination" done < <(find "$SOURCE_DIR" -mindepth 2 -type f -name '*.md' -print0 | sort -z) # Every .pdf gets a Word copy beside it, reconstructed from the rendered page. # The CI header bands need Playwright, so they are rendered here with node and # handed to the converter, which then only needs Python with pdf2docx: the local # interpreter ($PYTHON) when it has it, otherwise a throwaway python container. -BAND_DIR="$(mktemp -d)" -trap 'rm -rf "$BAND_DIR"' EXIT echo "Rendering CI header bands..." -node "$TOOL_DIR/render-ci-band.mjs" "$SOURCE_DIR" "$BAND_DIR" +node "$TOOL_DIR/render-ci-band.mjs" "$SOURCE_DIR" "$BAND_DIR" "$STAGING_DIR" echo "Converting PDFs to Word..." PYTHON="${PYTHON:-python3}" diff --git a/scripts/sdlc-delivery/assets/sdlc-delivery.css b/scripts/sdlc-delivery/assets/sdlc-delivery.css index 3ccd35d..4908214 100644 --- a/scripts/sdlc-delivery/assets/sdlc-delivery.css +++ b/scripts/sdlc-delivery/assets/sdlc-delivery.css @@ -74,7 +74,7 @@ table:not(.document-control) { border-collapse: collapse; border: .5pt solid var(--brn-line); break-inside: auto; - font-size: 9pt; + font-size: 8.5pt; margin: 0 0 5mm; table-layout: auto; width: 100%; diff --git a/scripts/sdlc-delivery/convert-to-docx.py b/scripts/sdlc-delivery/convert-to-docx.py index 70abe15..58b385d 100644 --- a/scripts/sdlc-delivery/convert-to-docx.py +++ b/scripts/sdlc-delivery/convert-to-docx.py @@ -41,7 +41,6 @@ HERE = Path(__file__).resolve().parent # Page box measured from the audited reference package, as in sdlc-delivery.css. BODY_TOP_MM = 32 FOOTER_FROM_BOTTOM_MM = 24.2 # footer rule at 272.8mm on a 297mm page -BAND_WIDTH_MM = 210 STANDARD = 'ISO/IEC 29110-4-1:2018' RULE_SIZE = '4' # .5pt table rule, as printed RULE_COLOUR = '595959' # --brn-line from sdlc-delivery.css @@ -702,6 +701,68 @@ def build_table(document, rows): return table +SIGNATURE_HEADER = tuple(letters(cell) for cell in ('ชื่อ', 'ตำแหน่ง', 'ลายเซ็น', 'วันที่')) + + +def apply_column_widths(document, sized): + """Give each table the column widths the PDF was printed with. + + pdf2docx writes its own grid, usually equal columns, so a table that reads + well in the PDF comes out cramped in Word. Tables are matched to the source + by their header row; pdf2docx repeats the header on each page's piece of a + long table, so every piece is matched. Returns the headers left unmatched. + """ + section = document.sections[0] + landscape = section.page_width > section.page_height + width = round(Mm(PRINTED_TEXT_WIDTH_MM[landscape]).twips) + by_header = {tuple(letters(cell) for cell in table['header']): table['widths'] for table in sized} + unmatched = [] + for table in document.element.body.iter(qn('w:tbl')): + rows = table.findall(qn('w:tr')) + if not rows: + continue + header = tuple(letters(text_of(cell)) for cell in rows[0].findall(qn('w:tc'))) + if len(header) < 2: + continue + percents = by_header.get(header) + if percents is None: + # The document-control block and the signature tables keep their own layout. + if header[0] != letters('Document No') and header != SIGNATURE_HEADER: + unmatched.append(' | '.join(header)) + continue + total = sum(percents) + twips = [round(width * p / total) for p in percents] + properties = table.find(qn('w:tblPr')) + for tag, attributes in (('w:tblW', {'w:w': width, 'w:type': 'dxa'}), ('w:tblLayout', {'w:type': 'fixed'})): + node = properties.find(qn(tag)) + if node is None: + node = OxmlElement(tag) + properties.append(node) + for name, value in attributes.items(): + node.set(qn(name), str(value)) + grid = table.find(qn('w:tblGrid')) + for column in list(grid): + grid.remove(column) + for value in twips: + column = OxmlElement('w:gridCol') + column.set(qn('w:w'), str(value)) + grid.append(column) + for row in rows: + index = 0 + for cell in row.findall(qn('w:tc')): + cell_properties = cell.get_or_add_tcPr() + span = cell_properties.find(qn('w:gridSpan')) + count = int(span.get(qn('w:val'))) if span is not None else 1 + cell_width = cell_properties.find(qn('w:tcW')) + if cell_width is None: + cell_width = OxmlElement('w:tcW') + cell_properties.insert(0, cell_width) + cell_width.set(qn('w:w'), str(sum(twips[index:index + count]))) + cell_width.set(qn('w:type'), 'dxa') + index += count + return unmatched + + def restore_dropped_tables(document, markdown): """Put back tables the converter left out, where the source places them. @@ -774,7 +835,7 @@ def install_letterhead(document, band, footer_tag): header.paragraph_format.left_indent = -section.left_margin header.paragraph_format.space_before = Pt(0) header.paragraph_format.space_after = Pt(0) - header.add_run().add_picture(str(band), width=Mm(BAND_WIDTH_MM)) + header.add_run().add_picture(str(band), width=section.page_width) footer = section.footer.paragraphs[0] drop_style(footer) @@ -789,7 +850,7 @@ def install_letterhead(document, band, footer_tag): style_run(footer.add_run(f'\t{footer_tag}'), 9) -def convert(pdf, source, band, footer_tag): +def convert(pdf, source, band, footer_tag, sized=()): docx = pdf.with_suffix('.docx') converter = Converter(str(pdf)) try: @@ -806,6 +867,8 @@ def convert(pdf, source, band, footer_tag): markdown = source.read_text(encoding='utf8') restore_text(document, markdown, bullets) restore_dropped_tables(document, markdown) + for header in apply_column_widths(document, sized): + print(f' WARNING {pdf.name}: no source widths for table "{header[:80]}"', file=sys.stderr) for paragraph in bullets: if paragraph.getparent() is not None: make_bullet(document, paragraph) @@ -816,9 +879,9 @@ def convert(pdf, source, band, footer_tag): return docx -def render_bands(source_root, band_root): +def render_bands(source_root, band_root, pdf_root): subprocess.run( - ['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root)], + ['node', str(HERE / 'render-ci-band.mjs'), str(source_root), str(band_root), str(pdf_root)], check=True ) return json.loads((Path(band_root) / 'bands.json').read_text(encoding='utf8')) @@ -831,16 +894,20 @@ def main(argv): source_root, delivery_root = Path(argv[0]), Path(argv[1]) temporary = None if len(argv) > 2: - bands = json.loads((Path(argv[2]) / 'bands.json').read_text(encoding='utf8')) + band_root = Path(argv[2]) + bands = json.loads((band_root / 'bands.json').read_text(encoding='utf8')) else: temporary = tempfile.TemporaryDirectory() - bands = render_bands(source_root, temporary.name) + band_root = Path(temporary.name) + bands = render_bands(source_root, temporary.name, delivery_root) failures = [] for index, (relative, band) in enumerate(sorted(bands.items()), start=1): pdf = delivery_root / Path(relative).with_suffix('.pdf') try: - docx = convert(pdf, source_root / relative, Path(band['band']), band['footer']) + widths = band_root / 'widths' / Path(relative).with_suffix('.json') + sized = json.loads(widths.read_text(encoding='utf8')) if widths.exists() else [] + docx = convert(pdf, source_root / relative, Path(band['band']), band['footer'], sized) print(f'[{index}/{len(bands)}] {docx.relative_to(delivery_root)}') except Exception as error: # noqa: BLE001 - reported, not swallowed failures.append((pdf, error)) diff --git a/scripts/sdlc-delivery/letterhead.mjs b/scripts/sdlc-delivery/letterhead.mjs new file mode 100644 index 0000000..db271ca --- /dev/null +++ b/scripts/sdlc-delivery/letterhead.mjs @@ -0,0 +1,28 @@ +// Page geometry shared by the PDF header (render-sdlc.mjs) and the Word header +// band (render-ci-band.mjs), so both draw the same letterhead for a page. +// +// Portrait is the audited reference layout. Landscape pages are 87mm wider: the +// grey ribbon stretches and the blue banner keeps its distance from the right +// edge, while the logo and company block stay where they are. +export const PAGE_MM = { portrait: 210, landscape: 297 }; +const EXTRA_MM = PAGE_MM.landscape - PAGE_MM.portrait; + +export function headerValues(title, landscape) { + const extra = landscape ? EXTRA_MM : 0; + return { + documentTitle: title, + // The banner holds about 22 characters at 13.5pt; longer titles shrink to fit. + titleSize: `${Math.min(13.5, (13.5 * 22) / title.length).toFixed(1)}pt`, + pageWidth: `${landscape ? PAGE_MM.landscape : PAGE_MM.portrait}mm`, + ribbonWidth: `${64.7 + extra}mm`, + bannerLeft: `${127.2 + extra}mm` + }; +} + +/** Orientation of a rendered PDF, read from its first /MediaBox. */ +export function isLandscapePdf(bytes) { + const box = bytes.toString('latin1').match(/\/MediaBox\s*\[\s*([\d.]+)\s+([\d.]+)\s+([\d.]+)\s+([\d.]+)\s*\]/); + if (!box) throw new Error('PDF has no /MediaBox'); + const [x0, y0, x1, y1] = box.slice(1).map(Number); + return x1 - x0 > y1 - y0; +} diff --git a/scripts/sdlc-delivery/render-ci-band.mjs b/scripts/sdlc-delivery/render-ci-band.mjs index 2d29047..5cc6b3b 100644 --- a/scripts/sdlc-delivery/render-ci-band.mjs +++ b/scripts/sdlc-delivery/render-ci-band.mjs @@ -7,12 +7,13 @@ import fs from 'node:fs/promises'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { chromium } from 'playwright'; +import { PAGE_MM, headerValues, isLandscapePdf } from './letterhead.mjs'; const here = path.dirname(fileURLToPath(import.meta.url)); const root = path.resolve(here, '../..'); const assets = path.join(here, 'assets'); -export const BAND_MM = { width: 210, height: 27 }; +export const BAND_MM = { height: 27 }; const SCALE = 3; // the band is drawn at 3x so it stays crisp in print const applyTemplate = (template, values) => template.replace(/{{(\w+)}}/g, (_, key) => values[key] ?? ''); @@ -30,8 +31,9 @@ async function markdownFiles(directory, depth = 0) { } async function main() { - const [sourceRoot, outputRoot] = process.argv.slice(2); - if (!sourceRoot || !outputRoot) throw new Error('Usage: render-ci-band.mjs SOURCE_DIRECTORY OUTPUT_DIRECTORY'); + // PDF_DIRECTORY holds the rendered PDFs: each band matches its PDF's orientation. + const [sourceRoot, outputRoot, pdfRoot] = process.argv.slice(2); + if (!sourceRoot || !outputRoot || !pdfRoot) throw new Error('Usage: render-ci-band.mjs SOURCE_DIRECTORY OUTPUT_DIRECTORY PDF_DIRECTORY'); const [template, logo] = await Promise.all([ fs.readFile(path.join(assets, 'header.html'), 'utf8'), @@ -56,19 +58,25 @@ async function main() { const title = markdown.match(/^#\s+(.+)$/m)?.[1]?.trim() ?? 'BRN WMS'; const footer = markdown.match(/^\s*$/m)?.[1] ?? ''; - if (!rendered.has(title)) { - const file = path.join(outputRoot, `${slug(title)}.png`); + const relative = path.relative(sourceRoot, source); + const pdf = path.join(pdfRoot, relative.replace(/\.md$/, '.pdf')); + const landscape = isLandscapePdf(await fs.readFile(pdf)); + const key = `${title}|${landscape ? 'landscape' : 'portrait'}`; + const widthMm = landscape ? PAGE_MM.landscape : PAGE_MM.portrait; + + if (!rendered.has(key)) { + const file = path.join(outputRoot, `${slug(title)}-${landscape ? 'landscape' : 'portrait'}.png`); await page.setContent( '
' + - `