docs(sdlc): match reference PDF layout and company CI

This commit is contained in:
Thanakorn
2026-09-09 13:54:45 +07:00
parent 9553d94ec6
commit 97dc634c19
81 changed files with 149 additions and 156 deletions
+16 -4
View File
@@ -21,7 +21,8 @@ function extractDocumentContent(markdown) {
const titleMatch = markdown.match(/^#\s+.+\n+/m);
const body = markdown
.slice((titleMatch?.index ?? 0) + (titleMatch?.[0].length ?? 0))
.replace(/^<!--\s*footer:.*?-->\s*$/m, '');
.replace(/^<!--\s*footer:.*?-->\s*$/m, '')
.replace(/<br\s*\/?>/gi, ' '); // markup, not document text
return { body, fields: {} };
}
@@ -60,9 +61,20 @@ async function main() {
const pdf = path.join(deliveryRoot, relative.replace(/\.md$/, '.pdf'));
const [markdown, extracted] = await Promise.all([fs.readFile(source, 'utf8'), pdfText(pdf)]);
const { body } = extractDocumentContent(markdown);
const expected = words(body);
const actual = compact(extracted);
const missing = [...expected].filter((word) => !actual.includes(word));
// Thai has no word spaces, so a run can wrap several times inside a narrow
// table cell and reach the extracted text as separate fragments. Coverage is
// therefore checked by character count: every character of the source must
// appear in the PDF at least as many times. Clipped or dropped text loses
// glyphs and still fails; re-wrapping does not.
const tally = (text) => {
const counts = new Map();
for (const character of compact(text)) counts.set(character, (counts.get(character) ?? 0) + 1);
return counts;
};
const rendered = tally(extracted);
const missing = [...tally(body)]
.filter(([character, count]) => (rendered.get(character) ?? 0) < count)
.map(([character, count]) => `${character}\u00d7${count - (rendered.get(character) ?? 0)}`);
if (missing.length) failures.push(`${relative}: missing ${missing.slice(0, 12).join(', ')}${missing.length > 12 ? ', …' : ''}`);
}