From badd9829048ce5a8e5fbab68fdcafff2f49a0342 Mon Sep 17 00:00:00 2001 From: Paco POR-CORREO Date: Mon, 14 Sep 2026 14:20:52 +0200 Subject: [PATCH] feat(ocr): detect and compose OCR pages --- src/modules/ocr/composition.ts | 114 +++++++++++++++++++++++++++ src/modules/ocr/detection.ts | 70 +++++++++++++++++ tests/ocr/detection.test.ts | 138 +++++++++++++++++++++++++++++++++ 3 files changed, 322 insertions(+) create mode 100644 src/modules/ocr/composition.ts create mode 100644 src/modules/ocr/detection.ts create mode 100644 tests/ocr/detection.test.ts diff --git a/src/modules/ocr/composition.ts b/src/modules/ocr/composition.ts new file mode 100644 index 0000000..2470e3a --- /dev/null +++ b/src/modules/ocr/composition.ts @@ -0,0 +1,114 @@ +import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js"; + +export interface CandidateLineInput { + lineId: string; + text: string; + confidence: number; + bbox: [number, number, number, number]; +} + +export interface CandidatePageInput { + page: number; + method: "native" | "ocr" | "blank"; + nativeText: string; + rawOcrText: string; + lines: CandidateLineInput[]; +} + +export interface CandidateLine extends CandidateLineInput { + lineSha256: string; +} + +export interface CandidatePage { + page: number; + method: CandidatePageInput["method"]; + nativeText: string; + rawOcrText: string; + lines: CandidateLine[]; + candidateText: string; + candidateTextSha256: string; + risks: string[]; +} + +const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g; +const CONTEXT_WORD = /(?:codigo|error|regla|sqlstate|estado|identificador)\s*[:#-]?\s*$/iu; + +export function prioritizeRiskTokens(text: string, comparisonText = ""): string[] { + const matches = [...text.matchAll(RISK_TOKEN)]; + const counts = new Map(); + for (const match of matches) counts.set(match[0], (counts.get(match[0]) ?? 0) + 1); + + const unique = new Map(); + for (const match of matches) { + const token = match[0]; + if (unique.has(token)) continue; + const position = match.index ?? 0; + const differs = !comparisonText.includes(token); + const mixedCase = /[A-Z]/u.test(token) && /[a-z]/u.test(token); + const ambiguous = /[Oo0Ii1lSs5]/u.test(token); + const contextAdjacent = CONTEXT_WORD.test(text.slice(Math.max(0, position - 40), position)); + unique.set(token, { + token, + position, + score: Number(differs) * 2 + Number(contextAdjacent) * 2 + Number(mixedCase) + Number(ambiguous) + Number(counts.get(token) === 1) + }); + } + return [...unique.values()] + .sort((left, right) => right.score - left.score || left.position - right.position) + .map(({ token }) => token); +} + +export function composeCandidate(inputPages: CandidatePageInput[]): { + text: string; + textSha256: string; + candidateSha256: string; + pages: CandidatePage[]; +} { + const pageNumbers = new Set(); + const pages = [...inputPages] + .sort((left, right) => left.page - right.page) + .map((input): CandidatePage => { + if (!Number.isInteger(input.page) || input.page < 1 || pageNumbers.has(input.page)) { + throw new Error("Candidate pages must be unique positive one-based integers"); + } + pageNumbers.add(input.page); + const lines = input.method === "ocr" + ? input.lines + .map((line, originalIndex) => ({ line, originalIndex })) + .sort((left, right) => left.line.bbox[1] - right.line.bbox[1] + || left.line.bbox[0] - right.line.bbox[0] + || left.originalIndex - right.originalIndex) + .map(({ line }) => ({ + ...line, + bbox: [line.bbox[0], line.bbox[1], line.bbox[2], line.bbox[3]] as [number, number, number, number], + lineSha256: sha256Hex(line.text) + })) + : []; + const candidateText = input.method === "native" + ? input.nativeText + : input.method === "ocr" + ? lines.map(({ text }) => text).join("\n") + : ""; + return { + page: input.page, + method: input.method, + nativeText: input.nativeText, + rawOcrText: input.rawOcrText, + lines, + candidateText, + candidateTextSha256: sha256Hex(candidateText), + risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText) + }; + }); + + const nonBlank = pages.filter(({ candidateText }) => candidateText.length > 0); + const text = nonBlank.map(({ page, candidateText }, index) => index === 0 + ? candidateText + : `--- Page ${page} ---\n\n${candidateText}`).join("\n\n"); + return { + text, + textSha256: sha256Hex(text), + candidateSha256: sha256Hex(canonicalJson(pages)), + pages + }; +} diff --git a/src/modules/ocr/detection.ts b/src/modules/ocr/detection.ts new file mode 100644 index 0000000..8a77edc --- /dev/null +++ b/src/modules/ocr/detection.ts @@ -0,0 +1,70 @@ +import path from "node:path"; + +export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const; + +export interface NativeTextMetrics { + nonWhitespaceCharacters: number; + alphanumericCharacters: number; + wordCount: number; + replacementControlRatio: number; +} + +export interface OcrQualityMetrics { + nonWhitespaceCharacters: number; + medianConfidence: number; + p10Confidence: number; + lowConfidenceLineRatio: number; +} + +const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u; + +export function computeNativeMetrics(text: string): NativeTextMetrics { + const normalized = text.normalize("NFC"); + const characters = Array.from(normalized); + const nonWhitespaceCharacters = characters.filter((character) => /\S/u.test(character) && !CONTROL_CHARACTER.test(character)).length; + const replacementOrControl = characters.filter((character) => character === "\uFFFD" || CONTROL_CHARACTER.test(character)).length; + return { + nonWhitespaceCharacters, + alphanumericCharacters: characters.filter((character) => /[A-Za-z0-9]/u.test(character)).length, + wordCount: normalized.trim() ? normalized.trim().split(/\s+/u).length : 0, + replacementControlRatio: nonWhitespaceCharacters === 0 ? 0 : replacementOrControl / nonWhitespaceCharacters + }; +} + +export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean { + return metrics.nonWhitespaceCharacters >= 120 + && metrics.alphanumericCharacters >= 80 + && metrics.wordCount >= 20 + && metrics.replacementControlRatio <= 0.01; +} + +export function selectPdfPagesForOcr( + filePath: string, + pages: ReadonlyArray<{ page: number; text: string }> +): number[] { + if (path.extname(filePath).toLowerCase() !== ".pdf") return []; + + const selected = new Set(); + for (const { page, text } of pages) { + if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers"); + if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page); + } + return [...selected].sort((left, right) => left - right); +} + +export function classifyOcrPage(input: { + inkCoverage: number; + metrics: OcrQualityMetrics; +}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } { + const { inkCoverage, metrics } = input; + if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" }; + if ( + metrics.nonWhitespaceCharacters >= 40 + && metrics.medianConfidence >= 0.8 + && metrics.p10Confidence >= 0.5 + && metrics.lowConfidenceLineRatio <= 0.2 + ) { + return { method: "ocr" }; + } + return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" }; +} diff --git a/tests/ocr/detection.test.ts b/tests/ocr/detection.test.ts new file mode 100644 index 0000000..43ff965 --- /dev/null +++ b/tests/ocr/detection.test.ts @@ -0,0 +1,138 @@ +import assert from "node:assert/strict"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js"; +import { + DETECTION_POLICY_VERSION, + classifyOcrPage, + computeNativeMetrics, + isNativeTextSufficient, + selectPdfPagesForOcr +} from "../../src/modules/ocr/detection.js"; +import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js"; +import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js"; + +function buildPdf(pageTexts: string[]): Buffer { + const fontId = 3 + pageTexts.length * 2; + const objects = [ + "<< /Type /Catalog /Pages 2 0 R >>", + `<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>` + ]; + for (const [index, text] of pageTexts.entries()) { + const pageId = 3 + index * 2; + const contentId = pageId + 1; + const escaped = text.replace(/([\\()])/g, "\\$1"); + const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : ""; + objects.push( + `<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`, + `<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream` + ); + } + objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"); + + let pdf = "%PDF-1.4\n"; + const offsets = [0]; + objects.forEach((object, index) => { + offsets.push(Buffer.byteLength(pdf)); + pdf += `${index + 1} 0 obj\n${object}\nendobj\n`; + }); + const xref = Buffer.byteLength(pdf); + pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`; + pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join(""); + pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`; + return Buffer.from(pdf); +} + +const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" "); + +test("native detection applies every pdf-detection-v1 boundary per page", () => { + const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 }; + + assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1"); + assert.deepEqual( + [boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }] + .map(isNativeTextSufficient), + [true, false, false, false, false] + ); + assert.deepEqual(computeNativeMetrics("Árbol 12\nword�\u0001"), { + nonWhitespaceCharacters: 12, + alphanumericCharacters: 10, + wordCount: 3, + replacementControlRatio: 2 / 12 + }); +}); + +test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => { + const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-")); + const filePath = path.join(directory, "mixed.PDF"); + try { + await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"])); + const pages = await parsePdfPages(filePath); + + assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]); + for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) { + assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []); + } + assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]); + } finally { + await rm(directory, { recursive: true, force: true }); + } +}); + +test("OCR blank and quality gates fail closed at exact thresholds", () => { + const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 }; + + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" }); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" }); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" }); + for (const metrics of [ + { ...passing, nonWhitespaceCharacters: 39 }, + { ...passing, medianConfidence: 0.799 }, + { ...passing, p10Confidence: 0.499 }, + { ...passing, lowConfidenceLineRatio: 0.201 } + ]) { + assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked"); + } +}); + +test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => { + const input = [ + { + page: 3, + method: "ocr" as const, + nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06", + rawOcrText: "raw service text", + lines: [ + { lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] }, + { lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] }, + { lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] } + ] + }, + { page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] }, + { page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] } + ]; + const first = composeCandidate(input); + const second = composeCandidate(structuredClone(input)); + + assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7"); + assert.equal(first.textSha256, sha256Hex(first.text)); + assert.deepEqual(first, second); + assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [ + { page: 1, candidateText: "Native first page.", risks: [] }, + { page: 2, candidateText: "", risks: [] }, + { page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] } + ]); + assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages))); + assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a")); + assert.equal(first.pages[0]?.rawOcrText, "ignored OCR"); + assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06"); +}); + +test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => { + assert.deepEqual( + prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"), + ["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"] + ); +});