rag-service/tests/ocr/detection.test.ts

144 lines
7 KiB
TypeScript
Raw Blame History

import assert from "node:assert/strict";
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import test from "node:test";
import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js";
import {
DETECTION_POLICY_VERSION,
classifyOcrPage,
computeNativeMetrics,
isNativeTextSufficient,
selectPdfPagesForOcr
} from "../../src/modules/ocr/detection.js";
import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js";
import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js";
function buildPdf(pageTexts: string[]): Buffer {
const fontId = 3 + pageTexts.length * 2;
const objects = [
"<< /Type /Catalog /Pages 2 0 R >>",
`<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>`
];
for (const [index, text] of pageTexts.entries()) {
const pageId = 3 + index * 2;
const contentId = pageId + 1;
const escaped = text.replace(/([\\()])/g, "\\$1");
const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : "";
objects.push(
`<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`,
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`
);
}
objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>");
let pdf = "%PDF-1.4\n";
const offsets = [0];
objects.forEach((object, index) => {
offsets.push(Buffer.byteLength(pdf));
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
});
const xref = Buffer.byteLength(pdf);
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
return Buffer.from(pdf);
}
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
test("native detection applies every pdf-detection-v2 boundary per page", () => {
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v2");
assert.deepEqual(
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
.map(isNativeTextSufficient),
[true, false, false, false, false]
);
assert.deepEqual(computeNativeMetrics("Árbol 12\nword<72>\u0001"), {
nonWhitespaceCharacters: 12,
alphanumericCharacters: 10,
wordCount: 3,
replacementControlRatio: 2 / 12
});
});
test("text-rich pages route at the raster threshold while small logos stay native", () => {
const page = { page: 1, text: sufficientText };
assert.deepEqual(selectPdfPagesForOcr("visual.pdf", [{ ...page, rasterCoverage: 0.05 }]), [1]);
assert.deepEqual(selectPdfPagesForOcr("logo.pdf", [{ ...page, rasterCoverage: 0.0499 }]), []);
});
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
const filePath = path.join(directory, "mixed.PDF");
try {
await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"]));
const pages = await parsePdfPages(filePath);
assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]);
for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) {
assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []);
}
assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]);
} finally {
await rm(directory, { recursive: true, force: true });
}
});
test("OCR blank and quality gates fail closed at exact thresholds", () => {
const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 };
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" });
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" });
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" });
for (const metrics of [
{ ...passing, nonWhitespaceCharacters: 39 },
{ ...passing, medianConfidence: 0.799 },
{ ...passing, p10Confidence: 0.499 },
{ ...passing, lowConfidenceLineRatio: 0.201 }
]) {
assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked");
}
});
test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => {
const input = [
{
page: 3,
method: "ocr" as const,
nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06",
rawOcrText: "raw service text",
lines: [
{ lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] },
{ lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] },
{ lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] }
]
},
{ page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] },
{ page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] }
];
const first = composeCandidate(input);
const second = composeCandidate(structuredClone(input));
assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7");
assert.equal(first.textSha256, sha256Hex(first.text));
assert.deepEqual(first, second);
assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [
{ page: 1, candidateText: "Native first page.", risks: [] },
{ page: 2, candidateText: "", risks: [] },
{ page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] }
]);
assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages)));
assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a"));
assert.equal(first.pages[0]?.rawOcrText, "ignored OCR");
assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06");
});
test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => {
assert.deepEqual(
prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"),
["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"]
);
});