import assert from "node:assert/strict"; import { mkdtemp, rm, writeFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import test from "node:test"; import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js"; import { DETECTION_POLICY_VERSION, classifyOcrPage, computeNativeMetrics, isNativeTextSufficient, selectPdfPagesForOcr } from "../../src/modules/ocr/detection.js"; import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js"; import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js"; function buildPdf(pageTexts: string[]): Buffer { const fontId = 3 + pageTexts.length * 2; const objects = [ "<< /Type /Catalog /Pages 2 0 R >>", `<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>` ]; for (const [index, text] of pageTexts.entries()) { const pageId = 3 + index * 2; const contentId = pageId + 1; const escaped = text.replace(/([\\()])/g, "\\$1"); const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : ""; objects.push( `<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`, `<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream` ); } objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"); let pdf = "%PDF-1.4\n"; const offsets = [0]; objects.forEach((object, index) => { offsets.push(Buffer.byteLength(pdf)); pdf += `${index + 1} 0 obj\n${object}\nendobj\n`; }); const xref = Buffer.byteLength(pdf); pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`; pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join(""); pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`; return Buffer.from(pdf); } const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" "); test("native detection applies every pdf-detection-v2 boundary per page", () => { const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 }; assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v2"); assert.deepEqual( [boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }] .map(isNativeTextSufficient), [true, false, false, false, false] ); assert.deepEqual(computeNativeMetrics("Árbol 12\nword�\u0001"), { nonWhitespaceCharacters: 12, alphanumericCharacters: 10, wordCount: 3, replacementControlRatio: 2 / 12 }); }); test("text-rich pages route at the raster threshold while small logos stay native", () => { const page = { page: 1, text: sufficientText }; assert.deepEqual(selectPdfPagesForOcr("visual.pdf", [{ ...page, rasterCoverage: 0.05 }]), [1]); assert.deepEqual(selectPdfPagesForOcr("logo.pdf", [{ ...page, rasterCoverage: 0.0499 }]), []); }); test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => { const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-")); const filePath = path.join(directory, "mixed.PDF"); try { await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"])); const pages = await parsePdfPages(filePath); assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]); for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) { assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []); } assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]); } finally { await rm(directory, { recursive: true, force: true }); } }); test("OCR blank and quality gates fail closed at exact thresholds", () => { const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 }; assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" }); assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" }); assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" }); for (const metrics of [ { ...passing, nonWhitespaceCharacters: 39 }, { ...passing, medianConfidence: 0.799 }, { ...passing, p10Confidence: 0.499 }, { ...passing, lowConfidenceLineRatio: 0.201 } ]) { assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked"); } }); test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => { const input = [ { page: 3, method: "ocr" as const, nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06", rawOcrText: "raw service text", lines: [ { lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] }, { lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] }, { lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] } ] }, { page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] }, { page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] } ]; const first = composeCandidate(input); const second = composeCandidate(structuredClone(input)); assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7"); assert.equal(first.textSha256, sha256Hex(first.text)); assert.deepEqual(first, second); assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [ { page: 1, candidateText: "Native first page.", risks: [] }, { page: 2, candidateText: "", risks: [] }, { page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] } ]); assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages))); assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a")); assert.equal(first.pages[0]?.rawOcrText, "ignored OCR"); assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06"); }); test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => { assert.deepEqual( prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"), ["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"] ); });