#!/usr/bin/env python3 """Convert the rendered delivery package from PDF to Word. The work products are authored in Markdown and rendered to PDF by render-sdlc.mjs; the .docx beside each .pdf is a layout reconstruction of that PDF, so the Word file keeps the page the reviewer signed off rather than being laid out a second time from the source. Requires pdf2docx: pip install -r requirements.txt Usage: convert-to-docx.py DELIVERY_DIRECTORY [PDF…] """ import sys import os from pathlib import Path from pdf2docx import Converter def convert(pdf: Path) -> Path: docx = pdf.with_suffix('.docx') converter = Converter(str(pdf)) try: converter.convert(str(docx)) finally: converter.close() return docx def main(argv): if not argv: raise SystemExit(__doc__) root = Path(argv[0]) pdfs = [Path(p) for p in argv[1:]] or sorted(root.rglob('*.pdf')) if not pdfs: raise SystemExit(f'No PDFs found under {root}') failures = [] for index, pdf in enumerate(pdfs, start=1): try: docx = convert(pdf) print(f'[{index}/{len(pdfs)}] {docx.relative_to(root)}') except Exception as error: # noqa: BLE001 - reported, not swallowed failures.append((pdf, error)) print(f'[{index}/{len(pdfs)}] FAILED {pdf.relative_to(root)}: {error}', file=sys.stderr) print(f'{len(pdfs) - len(failures)}/{len(pdfs)} documents converted') if failures: raise SystemExit(1) if __name__ == '__main__': # pdf2docx is chatty; the per-document progress above is enough. os.environ.setdefault('PDF2DOCX_QUIET', '1') main(sys.argv[1:])