feat(ocr): detect and compose OCR pages
This commit is contained in:
parent
905f8854db
commit
badd982904
3 changed files with 322 additions and 0 deletions
114
src/modules/ocr/composition.ts
Normal file
114
src/modules/ocr/composition.ts
Normal file
|
|
@ -0,0 +1,114 @@
|
||||||
|
import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js";
|
||||||
|
|
||||||
|
export interface CandidateLineInput {
|
||||||
|
lineId: string;
|
||||||
|
text: string;
|
||||||
|
confidence: number;
|
||||||
|
bbox: [number, number, number, number];
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface CandidatePageInput {
|
||||||
|
page: number;
|
||||||
|
method: "native" | "ocr" | "blank";
|
||||||
|
nativeText: string;
|
||||||
|
rawOcrText: string;
|
||||||
|
lines: CandidateLineInput[];
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface CandidateLine extends CandidateLineInput {
|
||||||
|
lineSha256: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface CandidatePage {
|
||||||
|
page: number;
|
||||||
|
method: CandidatePageInput["method"];
|
||||||
|
nativeText: string;
|
||||||
|
rawOcrText: string;
|
||||||
|
lines: CandidateLine[];
|
||||||
|
candidateText: string;
|
||||||
|
candidateTextSha256: string;
|
||||||
|
risks: string[];
|
||||||
|
}
|
||||||
|
|
||||||
|
const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g;
|
||||||
|
const CONTEXT_WORD = /(?:codigo|error|regla|sqlstate|estado|identificador)\s*[:#-]?\s*$/iu;
|
||||||
|
|
||||||
|
export function prioritizeRiskTokens(text: string, comparisonText = ""): string[] {
|
||||||
|
const matches = [...text.matchAll(RISK_TOKEN)];
|
||||||
|
const counts = new Map<string, number>();
|
||||||
|
for (const match of matches) counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
|
||||||
|
|
||||||
|
const unique = new Map<string, { token: string; position: number; score: number }>();
|
||||||
|
for (const match of matches) {
|
||||||
|
const token = match[0];
|
||||||
|
if (unique.has(token)) continue;
|
||||||
|
const position = match.index ?? 0;
|
||||||
|
const differs = !comparisonText.includes(token);
|
||||||
|
const mixedCase = /[A-Z]/u.test(token) && /[a-z]/u.test(token);
|
||||||
|
const ambiguous = /[Oo0Ii1lSs5]/u.test(token);
|
||||||
|
const contextAdjacent = CONTEXT_WORD.test(text.slice(Math.max(0, position - 40), position));
|
||||||
|
unique.set(token, {
|
||||||
|
token,
|
||||||
|
position,
|
||||||
|
score: Number(differs) * 2 + Number(contextAdjacent) * 2 + Number(mixedCase) + Number(ambiguous) + Number(counts.get(token) === 1)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return [...unique.values()]
|
||||||
|
.sort((left, right) => right.score - left.score || left.position - right.position)
|
||||||
|
.map(({ token }) => token);
|
||||||
|
}
|
||||||
|
|
||||||
|
export function composeCandidate(inputPages: CandidatePageInput[]): {
|
||||||
|
text: string;
|
||||||
|
textSha256: string;
|
||||||
|
candidateSha256: string;
|
||||||
|
pages: CandidatePage[];
|
||||||
|
} {
|
||||||
|
const pageNumbers = new Set<number>();
|
||||||
|
const pages = [...inputPages]
|
||||||
|
.sort((left, right) => left.page - right.page)
|
||||||
|
.map((input): CandidatePage => {
|
||||||
|
if (!Number.isInteger(input.page) || input.page < 1 || pageNumbers.has(input.page)) {
|
||||||
|
throw new Error("Candidate pages must be unique positive one-based integers");
|
||||||
|
}
|
||||||
|
pageNumbers.add(input.page);
|
||||||
|
const lines = input.method === "ocr"
|
||||||
|
? input.lines
|
||||||
|
.map((line, originalIndex) => ({ line, originalIndex }))
|
||||||
|
.sort((left, right) => left.line.bbox[1] - right.line.bbox[1]
|
||||||
|
|| left.line.bbox[0] - right.line.bbox[0]
|
||||||
|
|| left.originalIndex - right.originalIndex)
|
||||||
|
.map(({ line }) => ({
|
||||||
|
...line,
|
||||||
|
bbox: [line.bbox[0], line.bbox[1], line.bbox[2], line.bbox[3]] as [number, number, number, number],
|
||||||
|
lineSha256: sha256Hex(line.text)
|
||||||
|
}))
|
||||||
|
: [];
|
||||||
|
const candidateText = input.method === "native"
|
||||||
|
? input.nativeText
|
||||||
|
: input.method === "ocr"
|
||||||
|
? lines.map(({ text }) => text).join("\n")
|
||||||
|
: "";
|
||||||
|
return {
|
||||||
|
page: input.page,
|
||||||
|
method: input.method,
|
||||||
|
nativeText: input.nativeText,
|
||||||
|
rawOcrText: input.rawOcrText,
|
||||||
|
lines,
|
||||||
|
candidateText,
|
||||||
|
candidateTextSha256: sha256Hex(candidateText),
|
||||||
|
risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText)
|
||||||
|
};
|
||||||
|
});
|
||||||
|
|
||||||
|
const nonBlank = pages.filter(({ candidateText }) => candidateText.length > 0);
|
||||||
|
const text = nonBlank.map(({ page, candidateText }, index) => index === 0
|
||||||
|
? candidateText
|
||||||
|
: `--- Page ${page} ---\n\n${candidateText}`).join("\n\n");
|
||||||
|
return {
|
||||||
|
text,
|
||||||
|
textSha256: sha256Hex(text),
|
||||||
|
candidateSha256: sha256Hex(canonicalJson(pages)),
|
||||||
|
pages
|
||||||
|
};
|
||||||
|
}
|
||||||
70
src/modules/ocr/detection.ts
Normal file
70
src/modules/ocr/detection.ts
Normal file
|
|
@ -0,0 +1,70 @@
|
||||||
|
import path from "node:path";
|
||||||
|
|
||||||
|
export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const;
|
||||||
|
|
||||||
|
export interface NativeTextMetrics {
|
||||||
|
nonWhitespaceCharacters: number;
|
||||||
|
alphanumericCharacters: number;
|
||||||
|
wordCount: number;
|
||||||
|
replacementControlRatio: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface OcrQualityMetrics {
|
||||||
|
nonWhitespaceCharacters: number;
|
||||||
|
medianConfidence: number;
|
||||||
|
p10Confidence: number;
|
||||||
|
lowConfidenceLineRatio: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u;
|
||||||
|
|
||||||
|
export function computeNativeMetrics(text: string): NativeTextMetrics {
|
||||||
|
const normalized = text.normalize("NFC");
|
||||||
|
const characters = Array.from(normalized);
|
||||||
|
const nonWhitespaceCharacters = characters.filter((character) => /\S/u.test(character) && !CONTROL_CHARACTER.test(character)).length;
|
||||||
|
const replacementOrControl = characters.filter((character) => character === "\uFFFD" || CONTROL_CHARACTER.test(character)).length;
|
||||||
|
return {
|
||||||
|
nonWhitespaceCharacters,
|
||||||
|
alphanumericCharacters: characters.filter((character) => /[A-Za-z0-9]/u.test(character)).length,
|
||||||
|
wordCount: normalized.trim() ? normalized.trim().split(/\s+/u).length : 0,
|
||||||
|
replacementControlRatio: nonWhitespaceCharacters === 0 ? 0 : replacementOrControl / nonWhitespaceCharacters
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean {
|
||||||
|
return metrics.nonWhitespaceCharacters >= 120
|
||||||
|
&& metrics.alphanumericCharacters >= 80
|
||||||
|
&& metrics.wordCount >= 20
|
||||||
|
&& metrics.replacementControlRatio <= 0.01;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function selectPdfPagesForOcr(
|
||||||
|
filePath: string,
|
||||||
|
pages: ReadonlyArray<{ page: number; text: string }>
|
||||||
|
): number[] {
|
||||||
|
if (path.extname(filePath).toLowerCase() !== ".pdf") return [];
|
||||||
|
|
||||||
|
const selected = new Set<number>();
|
||||||
|
for (const { page, text } of pages) {
|
||||||
|
if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers");
|
||||||
|
if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page);
|
||||||
|
}
|
||||||
|
return [...selected].sort((left, right) => left - right);
|
||||||
|
}
|
||||||
|
|
||||||
|
export function classifyOcrPage(input: {
|
||||||
|
inkCoverage: number;
|
||||||
|
metrics: OcrQualityMetrics;
|
||||||
|
}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } {
|
||||||
|
const { inkCoverage, metrics } = input;
|
||||||
|
if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" };
|
||||||
|
if (
|
||||||
|
metrics.nonWhitespaceCharacters >= 40
|
||||||
|
&& metrics.medianConfidence >= 0.8
|
||||||
|
&& metrics.p10Confidence >= 0.5
|
||||||
|
&& metrics.lowConfidenceLineRatio <= 0.2
|
||||||
|
) {
|
||||||
|
return { method: "ocr" };
|
||||||
|
}
|
||||||
|
return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" };
|
||||||
|
}
|
||||||
138
tests/ocr/detection.test.ts
Normal file
138
tests/ocr/detection.test.ts
Normal file
|
|
@ -0,0 +1,138 @@
|
||||||
|
import assert from "node:assert/strict";
|
||||||
|
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||||
|
import os from "node:os";
|
||||||
|
import path from "node:path";
|
||||||
|
import test from "node:test";
|
||||||
|
import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js";
|
||||||
|
import {
|
||||||
|
DETECTION_POLICY_VERSION,
|
||||||
|
classifyOcrPage,
|
||||||
|
computeNativeMetrics,
|
||||||
|
isNativeTextSufficient,
|
||||||
|
selectPdfPagesForOcr
|
||||||
|
} from "../../src/modules/ocr/detection.js";
|
||||||
|
import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js";
|
||||||
|
import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js";
|
||||||
|
|
||||||
|
function buildPdf(pageTexts: string[]): Buffer {
|
||||||
|
const fontId = 3 + pageTexts.length * 2;
|
||||||
|
const objects = [
|
||||||
|
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||||
|
`<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>`
|
||||||
|
];
|
||||||
|
for (const [index, text] of pageTexts.entries()) {
|
||||||
|
const pageId = 3 + index * 2;
|
||||||
|
const contentId = pageId + 1;
|
||||||
|
const escaped = text.replace(/([\\()])/g, "\\$1");
|
||||||
|
const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : "";
|
||||||
|
objects.push(
|
||||||
|
`<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`,
|
||||||
|
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`
|
||||||
|
);
|
||||||
|
}
|
||||||
|
objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>");
|
||||||
|
|
||||||
|
let pdf = "%PDF-1.4\n";
|
||||||
|
const offsets = [0];
|
||||||
|
objects.forEach((object, index) => {
|
||||||
|
offsets.push(Buffer.byteLength(pdf));
|
||||||
|
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||||
|
});
|
||||||
|
const xref = Buffer.byteLength(pdf);
|
||||||
|
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||||
|
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
|
||||||
|
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
|
||||||
|
return Buffer.from(pdf);
|
||||||
|
}
|
||||||
|
|
||||||
|
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
|
||||||
|
|
||||||
|
test("native detection applies every pdf-detection-v1 boundary per page", () => {
|
||||||
|
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
|
||||||
|
|
||||||
|
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1");
|
||||||
|
assert.deepEqual(
|
||||||
|
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
|
||||||
|
.map(isNativeTextSufficient),
|
||||||
|
[true, false, false, false, false]
|
||||||
|
);
|
||||||
|
assert.deepEqual(computeNativeMetrics("Árbol 12\nword<72>\u0001"), {
|
||||||
|
nonWhitespaceCharacters: 12,
|
||||||
|
alphanumericCharacters: 10,
|
||||||
|
wordCount: 3,
|
||||||
|
replacementControlRatio: 2 / 12
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
|
||||||
|
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
|
||||||
|
const filePath = path.join(directory, "mixed.PDF");
|
||||||
|
try {
|
||||||
|
await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"]));
|
||||||
|
const pages = await parsePdfPages(filePath);
|
||||||
|
|
||||||
|
assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]);
|
||||||
|
for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) {
|
||||||
|
assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []);
|
||||||
|
}
|
||||||
|
assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]);
|
||||||
|
} finally {
|
||||||
|
await rm(directory, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test("OCR blank and quality gates fail closed at exact thresholds", () => {
|
||||||
|
const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 };
|
||||||
|
|
||||||
|
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" });
|
||||||
|
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" });
|
||||||
|
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" });
|
||||||
|
for (const metrics of [
|
||||||
|
{ ...passing, nonWhitespaceCharacters: 39 },
|
||||||
|
{ ...passing, medianConfidence: 0.799 },
|
||||||
|
{ ...passing, p10Confidence: 0.499 },
|
||||||
|
{ ...passing, lowConfidenceLineRatio: 0.201 }
|
||||||
|
]) {
|
||||||
|
assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => {
|
||||||
|
const input = [
|
||||||
|
{
|
||||||
|
page: 3,
|
||||||
|
method: "ocr" as const,
|
||||||
|
nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06",
|
||||||
|
rawOcrText: "raw service text",
|
||||||
|
lines: [
|
||||||
|
{ lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] },
|
||||||
|
{ lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] },
|
||||||
|
{ lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] }
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{ page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] },
|
||||||
|
{ page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] }
|
||||||
|
];
|
||||||
|
const first = composeCandidate(input);
|
||||||
|
const second = composeCandidate(structuredClone(input));
|
||||||
|
|
||||||
|
assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7");
|
||||||
|
assert.equal(first.textSha256, sha256Hex(first.text));
|
||||||
|
assert.deepEqual(first, second);
|
||||||
|
assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [
|
||||||
|
{ page: 1, candidateText: "Native first page.", risks: [] },
|
||||||
|
{ page: 2, candidateText: "", risks: [] },
|
||||||
|
{ page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] }
|
||||||
|
]);
|
||||||
|
assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages)));
|
||||||
|
assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a"));
|
||||||
|
assert.equal(first.pages[0]?.rawOcrText, "ignored OCR");
|
||||||
|
assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => {
|
||||||
|
assert.deepEqual(
|
||||||
|
prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"),
|
||||||
|
["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"]
|
||||||
|
);
|
||||||
|
});
|
||||||
Loading…
Add table
Reference in a new issue