feat(ocr): detect and compose OCR pages

This commit is contained in:
Paco POR-CORREO 2026-09-14 14:20:52 +02:00
parent 905f8854db
commit badd982904
3 changed files with 322 additions and 0 deletions

View file

@ -0,0 +1,114 @@
import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js";
export interface CandidateLineInput {
lineId: string;
text: string;
confidence: number;
bbox: [number, number, number, number];
}
export interface CandidatePageInput {
page: number;
method: "native" | "ocr" | "blank";
nativeText: string;
rawOcrText: string;
lines: CandidateLineInput[];
}
export interface CandidateLine extends CandidateLineInput {
lineSha256: string;
}
export interface CandidatePage {
page: number;
method: CandidatePageInput["method"];
nativeText: string;
rawOcrText: string;
lines: CandidateLine[];
candidateText: string;
candidateTextSha256: string;
risks: string[];
}
const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g;
const CONTEXT_WORD = /(?:codigo|error|regla|sqlstate|estado|identificador)\s*[:#-]?\s*$/iu;
export function prioritizeRiskTokens(text: string, comparisonText = ""): string[] {
const matches = [...text.matchAll(RISK_TOKEN)];
const counts = new Map<string, number>();
for (const match of matches) counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
const unique = new Map<string, { token: string; position: number; score: number }>();
for (const match of matches) {
const token = match[0];
if (unique.has(token)) continue;
const position = match.index ?? 0;
const differs = !comparisonText.includes(token);
const mixedCase = /[A-Z]/u.test(token) && /[a-z]/u.test(token);
const ambiguous = /[Oo0Ii1lSs5]/u.test(token);
const contextAdjacent = CONTEXT_WORD.test(text.slice(Math.max(0, position - 40), position));
unique.set(token, {
token,
position,
score: Number(differs) * 2 + Number(contextAdjacent) * 2 + Number(mixedCase) + Number(ambiguous) + Number(counts.get(token) === 1)
});
}
return [...unique.values()]
.sort((left, right) => right.score - left.score || left.position - right.position)
.map(({ token }) => token);
}
export function composeCandidate(inputPages: CandidatePageInput[]): {
text: string;
textSha256: string;
candidateSha256: string;
pages: CandidatePage[];
} {
const pageNumbers = new Set<number>();
const pages = [...inputPages]
.sort((left, right) => left.page - right.page)
.map((input): CandidatePage => {
if (!Number.isInteger(input.page) || input.page < 1 || pageNumbers.has(input.page)) {
throw new Error("Candidate pages must be unique positive one-based integers");
}
pageNumbers.add(input.page);
const lines = input.method === "ocr"
? input.lines
.map((line, originalIndex) => ({ line, originalIndex }))
.sort((left, right) => left.line.bbox[1] - right.line.bbox[1]
|| left.line.bbox[0] - right.line.bbox[0]
|| left.originalIndex - right.originalIndex)
.map(({ line }) => ({
...line,
bbox: [line.bbox[0], line.bbox[1], line.bbox[2], line.bbox[3]] as [number, number, number, number],
lineSha256: sha256Hex(line.text)
}))
: [];
const candidateText = input.method === "native"
? input.nativeText
: input.method === "ocr"
? lines.map(({ text }) => text).join("\n")
: "";
return {
page: input.page,
method: input.method,
nativeText: input.nativeText,
rawOcrText: input.rawOcrText,
lines,
candidateText,
candidateTextSha256: sha256Hex(candidateText),
risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText)
};
});
const nonBlank = pages.filter(({ candidateText }) => candidateText.length > 0);
const text = nonBlank.map(({ page, candidateText }, index) => index === 0
? candidateText
: `--- Page ${page} ---\n\n${candidateText}`).join("\n\n");
return {
text,
textSha256: sha256Hex(text),
candidateSha256: sha256Hex(canonicalJson(pages)),
pages
};
}

View file

@ -0,0 +1,70 @@
import path from "node:path";
export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const;
export interface NativeTextMetrics {
nonWhitespaceCharacters: number;
alphanumericCharacters: number;
wordCount: number;
replacementControlRatio: number;
}
export interface OcrQualityMetrics {
nonWhitespaceCharacters: number;
medianConfidence: number;
p10Confidence: number;
lowConfidenceLineRatio: number;
}
const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u;
export function computeNativeMetrics(text: string): NativeTextMetrics {
const normalized = text.normalize("NFC");
const characters = Array.from(normalized);
const nonWhitespaceCharacters = characters.filter((character) => /\S/u.test(character) && !CONTROL_CHARACTER.test(character)).length;
const replacementOrControl = characters.filter((character) => character === "\uFFFD" || CONTROL_CHARACTER.test(character)).length;
return {
nonWhitespaceCharacters,
alphanumericCharacters: characters.filter((character) => /[A-Za-z0-9]/u.test(character)).length,
wordCount: normalized.trim() ? normalized.trim().split(/\s+/u).length : 0,
replacementControlRatio: nonWhitespaceCharacters === 0 ? 0 : replacementOrControl / nonWhitespaceCharacters
};
}
export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean {
return metrics.nonWhitespaceCharacters >= 120
&& metrics.alphanumericCharacters >= 80
&& metrics.wordCount >= 20
&& metrics.replacementControlRatio <= 0.01;
}
export function selectPdfPagesForOcr(
filePath: string,
pages: ReadonlyArray<{ page: number; text: string }>
): number[] {
if (path.extname(filePath).toLowerCase() !== ".pdf") return [];
const selected = new Set<number>();
for (const { page, text } of pages) {
if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers");
if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page);
}
return [...selected].sort((left, right) => left - right);
}
export function classifyOcrPage(input: {
inkCoverage: number;
metrics: OcrQualityMetrics;
}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } {
const { inkCoverage, metrics } = input;
if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" };
if (
metrics.nonWhitespaceCharacters >= 40
&& metrics.medianConfidence >= 0.8
&& metrics.p10Confidence >= 0.5
&& metrics.lowConfidenceLineRatio <= 0.2
) {
return { method: "ocr" };
}
return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" };
}

138
tests/ocr/detection.test.ts Normal file
View file

@ -0,0 +1,138 @@
import assert from "node:assert/strict";
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import test from "node:test";
import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js";
import {
DETECTION_POLICY_VERSION,
classifyOcrPage,
computeNativeMetrics,
isNativeTextSufficient,
selectPdfPagesForOcr
} from "../../src/modules/ocr/detection.js";
import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js";
import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js";
function buildPdf(pageTexts: string[]): Buffer {
const fontId = 3 + pageTexts.length * 2;
const objects = [
"<< /Type /Catalog /Pages 2 0 R >>",
`<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>`
];
for (const [index, text] of pageTexts.entries()) {
const pageId = 3 + index * 2;
const contentId = pageId + 1;
const escaped = text.replace(/([\\()])/g, "\\$1");
const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : "";
objects.push(
`<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`,
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`
);
}
objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>");
let pdf = "%PDF-1.4\n";
const offsets = [0];
objects.forEach((object, index) => {
offsets.push(Buffer.byteLength(pdf));
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
});
const xref = Buffer.byteLength(pdf);
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
return Buffer.from(pdf);
}
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
test("native detection applies every pdf-detection-v1 boundary per page", () => {
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1");
assert.deepEqual(
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
.map(isNativeTextSufficient),
[true, false, false, false, false]
);
assert.deepEqual(computeNativeMetrics("Árbol 12\nword<72>\u0001"), {
nonWhitespaceCharacters: 12,
alphanumericCharacters: 10,
wordCount: 3,
replacementControlRatio: 2 / 12
});
});
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
const filePath = path.join(directory, "mixed.PDF");
try {
await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"]));
const pages = await parsePdfPages(filePath);
assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]);
for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) {
assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []);
}
assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]);
} finally {
await rm(directory, { recursive: true, force: true });
}
});
test("OCR blank and quality gates fail closed at exact thresholds", () => {
const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 };
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" });
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" });
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" });
for (const metrics of [
{ ...passing, nonWhitespaceCharacters: 39 },
{ ...passing, medianConfidence: 0.799 },
{ ...passing, p10Confidence: 0.499 },
{ ...passing, lowConfidenceLineRatio: 0.201 }
]) {
assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked");
}
});
test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => {
const input = [
{
page: 3,
method: "ocr" as const,
nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06",
rawOcrText: "raw service text",
lines: [
{ lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] },
{ lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] },
{ lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] }
]
},
{ page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] },
{ page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] }
];
const first = composeCandidate(input);
const second = composeCandidate(structuredClone(input));
assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7");
assert.equal(first.textSha256, sha256Hex(first.text));
assert.deepEqual(first, second);
assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [
{ page: 1, candidateText: "Native first page.", risks: [] },
{ page: 2, candidateText: "", risks: [] },
{ page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] }
]);
assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages)));
assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a"));
assert.equal(first.pages[0]?.rawOcrText, "ignored OCR");
assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06");
});
test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => {
assert.deepEqual(
prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"),
["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"]
);
});