feat(ocr): detect and compose OCR pages
This commit is contained in:
parent
905f8854db
commit
badd982904
3 changed files with 322 additions and 0 deletions
114
src/modules/ocr/composition.ts
Normal file
114
src/modules/ocr/composition.ts
Normal file
|
|
@ -0,0 +1,114 @@
|
|||
import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
export interface CandidateLineInput {
|
||||
lineId: string;
|
||||
text: string;
|
||||
confidence: number;
|
||||
bbox: [number, number, number, number];
|
||||
}
|
||||
|
||||
export interface CandidatePageInput {
|
||||
page: number;
|
||||
method: "native" | "ocr" | "blank";
|
||||
nativeText: string;
|
||||
rawOcrText: string;
|
||||
lines: CandidateLineInput[];
|
||||
}
|
||||
|
||||
export interface CandidateLine extends CandidateLineInput {
|
||||
lineSha256: string;
|
||||
}
|
||||
|
||||
export interface CandidatePage {
|
||||
page: number;
|
||||
method: CandidatePageInput["method"];
|
||||
nativeText: string;
|
||||
rawOcrText: string;
|
||||
lines: CandidateLine[];
|
||||
candidateText: string;
|
||||
candidateTextSha256: string;
|
||||
risks: string[];
|
||||
}
|
||||
|
||||
const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g;
|
||||
const CONTEXT_WORD = /(?:codigo|error|regla|sqlstate|estado|identificador)\s*[:#-]?\s*$/iu;
|
||||
|
||||
export function prioritizeRiskTokens(text: string, comparisonText = ""): string[] {
|
||||
const matches = [...text.matchAll(RISK_TOKEN)];
|
||||
const counts = new Map<string, number>();
|
||||
for (const match of matches) counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
|
||||
|
||||
const unique = new Map<string, { token: string; position: number; score: number }>();
|
||||
for (const match of matches) {
|
||||
const token = match[0];
|
||||
if (unique.has(token)) continue;
|
||||
const position = match.index ?? 0;
|
||||
const differs = !comparisonText.includes(token);
|
||||
const mixedCase = /[A-Z]/u.test(token) && /[a-z]/u.test(token);
|
||||
const ambiguous = /[Oo0Ii1lSs5]/u.test(token);
|
||||
const contextAdjacent = CONTEXT_WORD.test(text.slice(Math.max(0, position - 40), position));
|
||||
unique.set(token, {
|
||||
token,
|
||||
position,
|
||||
score: Number(differs) * 2 + Number(contextAdjacent) * 2 + Number(mixedCase) + Number(ambiguous) + Number(counts.get(token) === 1)
|
||||
});
|
||||
}
|
||||
return [...unique.values()]
|
||||
.sort((left, right) => right.score - left.score || left.position - right.position)
|
||||
.map(({ token }) => token);
|
||||
}
|
||||
|
||||
export function composeCandidate(inputPages: CandidatePageInput[]): {
|
||||
text: string;
|
||||
textSha256: string;
|
||||
candidateSha256: string;
|
||||
pages: CandidatePage[];
|
||||
} {
|
||||
const pageNumbers = new Set<number>();
|
||||
const pages = [...inputPages]
|
||||
.sort((left, right) => left.page - right.page)
|
||||
.map((input): CandidatePage => {
|
||||
if (!Number.isInteger(input.page) || input.page < 1 || pageNumbers.has(input.page)) {
|
||||
throw new Error("Candidate pages must be unique positive one-based integers");
|
||||
}
|
||||
pageNumbers.add(input.page);
|
||||
const lines = input.method === "ocr"
|
||||
? input.lines
|
||||
.map((line, originalIndex) => ({ line, originalIndex }))
|
||||
.sort((left, right) => left.line.bbox[1] - right.line.bbox[1]
|
||||
|| left.line.bbox[0] - right.line.bbox[0]
|
||||
|| left.originalIndex - right.originalIndex)
|
||||
.map(({ line }) => ({
|
||||
...line,
|
||||
bbox: [line.bbox[0], line.bbox[1], line.bbox[2], line.bbox[3]] as [number, number, number, number],
|
||||
lineSha256: sha256Hex(line.text)
|
||||
}))
|
||||
: [];
|
||||
const candidateText = input.method === "native"
|
||||
? input.nativeText
|
||||
: input.method === "ocr"
|
||||
? lines.map(({ text }) => text).join("\n")
|
||||
: "";
|
||||
return {
|
||||
page: input.page,
|
||||
method: input.method,
|
||||
nativeText: input.nativeText,
|
||||
rawOcrText: input.rawOcrText,
|
||||
lines,
|
||||
candidateText,
|
||||
candidateTextSha256: sha256Hex(candidateText),
|
||||
risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText)
|
||||
};
|
||||
});
|
||||
|
||||
const nonBlank = pages.filter(({ candidateText }) => candidateText.length > 0);
|
||||
const text = nonBlank.map(({ page, candidateText }, index) => index === 0
|
||||
? candidateText
|
||||
: `--- Page ${page} ---\n\n${candidateText}`).join("\n\n");
|
||||
return {
|
||||
text,
|
||||
textSha256: sha256Hex(text),
|
||||
candidateSha256: sha256Hex(canonicalJson(pages)),
|
||||
pages
|
||||
};
|
||||
}
|
||||
70
src/modules/ocr/detection.ts
Normal file
70
src/modules/ocr/detection.ts
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
import path from "node:path";
|
||||
|
||||
export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const;
|
||||
|
||||
export interface NativeTextMetrics {
|
||||
nonWhitespaceCharacters: number;
|
||||
alphanumericCharacters: number;
|
||||
wordCount: number;
|
||||
replacementControlRatio: number;
|
||||
}
|
||||
|
||||
export interface OcrQualityMetrics {
|
||||
nonWhitespaceCharacters: number;
|
||||
medianConfidence: number;
|
||||
p10Confidence: number;
|
||||
lowConfidenceLineRatio: number;
|
||||
}
|
||||
|
||||
const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u;
|
||||
|
||||
export function computeNativeMetrics(text: string): NativeTextMetrics {
|
||||
const normalized = text.normalize("NFC");
|
||||
const characters = Array.from(normalized);
|
||||
const nonWhitespaceCharacters = characters.filter((character) => /\S/u.test(character) && !CONTROL_CHARACTER.test(character)).length;
|
||||
const replacementOrControl = characters.filter((character) => character === "\uFFFD" || CONTROL_CHARACTER.test(character)).length;
|
||||
return {
|
||||
nonWhitespaceCharacters,
|
||||
alphanumericCharacters: characters.filter((character) => /[A-Za-z0-9]/u.test(character)).length,
|
||||
wordCount: normalized.trim() ? normalized.trim().split(/\s+/u).length : 0,
|
||||
replacementControlRatio: nonWhitespaceCharacters === 0 ? 0 : replacementOrControl / nonWhitespaceCharacters
|
||||
};
|
||||
}
|
||||
|
||||
export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean {
|
||||
return metrics.nonWhitespaceCharacters >= 120
|
||||
&& metrics.alphanumericCharacters >= 80
|
||||
&& metrics.wordCount >= 20
|
||||
&& metrics.replacementControlRatio <= 0.01;
|
||||
}
|
||||
|
||||
export function selectPdfPagesForOcr(
|
||||
filePath: string,
|
||||
pages: ReadonlyArray<{ page: number; text: string }>
|
||||
): number[] {
|
||||
if (path.extname(filePath).toLowerCase() !== ".pdf") return [];
|
||||
|
||||
const selected = new Set<number>();
|
||||
for (const { page, text } of pages) {
|
||||
if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers");
|
||||
if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page);
|
||||
}
|
||||
return [...selected].sort((left, right) => left - right);
|
||||
}
|
||||
|
||||
export function classifyOcrPage(input: {
|
||||
inkCoverage: number;
|
||||
metrics: OcrQualityMetrics;
|
||||
}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } {
|
||||
const { inkCoverage, metrics } = input;
|
||||
if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" };
|
||||
if (
|
||||
metrics.nonWhitespaceCharacters >= 40
|
||||
&& metrics.medianConfidence >= 0.8
|
||||
&& metrics.p10Confidence >= 0.5
|
||||
&& metrics.lowConfidenceLineRatio <= 0.2
|
||||
) {
|
||||
return { method: "ocr" };
|
||||
}
|
||||
return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" };
|
||||
}
|
||||
138
tests/ocr/detection.test.ts
Normal file
138
tests/ocr/detection.test.ts
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js";
|
||||
import {
|
||||
DETECTION_POLICY_VERSION,
|
||||
classifyOcrPage,
|
||||
computeNativeMetrics,
|
||||
isNativeTextSufficient,
|
||||
selectPdfPagesForOcr
|
||||
} from "../../src/modules/ocr/detection.js";
|
||||
import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js";
|
||||
import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
function buildPdf(pageTexts: string[]): Buffer {
|
||||
const fontId = 3 + pageTexts.length * 2;
|
||||
const objects = [
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
`<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>`
|
||||
];
|
||||
for (const [index, text] of pageTexts.entries()) {
|
||||
const pageId = 3 + index * 2;
|
||||
const contentId = pageId + 1;
|
||||
const escaped = text.replace(/([\\()])/g, "\\$1");
|
||||
const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : "";
|
||||
objects.push(
|
||||
`<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`,
|
||||
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`
|
||||
);
|
||||
}
|
||||
objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>");
|
||||
|
||||
let pdf = "%PDF-1.4\n";
|
||||
const offsets = [0];
|
||||
objects.forEach((object, index) => {
|
||||
offsets.push(Buffer.byteLength(pdf));
|
||||
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||
});
|
||||
const xref = Buffer.byteLength(pdf);
|
||||
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
|
||||
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
|
||||
return Buffer.from(pdf);
|
||||
}
|
||||
|
||||
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
|
||||
|
||||
test("native detection applies every pdf-detection-v1 boundary per page", () => {
|
||||
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
|
||||
|
||||
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1");
|
||||
assert.deepEqual(
|
||||
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
|
||||
.map(isNativeTextSufficient),
|
||||
[true, false, false, false, false]
|
||||
);
|
||||
assert.deepEqual(computeNativeMetrics("Árbol 12\nword<72>\u0001"), {
|
||||
nonWhitespaceCharacters: 12,
|
||||
alphanumericCharacters: 10,
|
||||
wordCount: 3,
|
||||
replacementControlRatio: 2 / 12
|
||||
});
|
||||
});
|
||||
|
||||
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
|
||||
const filePath = path.join(directory, "mixed.PDF");
|
||||
try {
|
||||
await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"]));
|
||||
const pages = await parsePdfPages(filePath);
|
||||
|
||||
assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]);
|
||||
for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) {
|
||||
assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []);
|
||||
}
|
||||
assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]);
|
||||
} finally {
|
||||
await rm(directory, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("OCR blank and quality gates fail closed at exact thresholds", () => {
|
||||
const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 };
|
||||
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" });
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" });
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" });
|
||||
for (const metrics of [
|
||||
{ ...passing, nonWhitespaceCharacters: 39 },
|
||||
{ ...passing, medianConfidence: 0.799 },
|
||||
{ ...passing, p10Confidence: 0.499 },
|
||||
{ ...passing, lowConfidenceLineRatio: 0.201 }
|
||||
]) {
|
||||
assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked");
|
||||
}
|
||||
});
|
||||
|
||||
test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => {
|
||||
const input = [
|
||||
{
|
||||
page: 3,
|
||||
method: "ocr" as const,
|
||||
nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06",
|
||||
rawOcrText: "raw service text",
|
||||
lines: [
|
||||
{ lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] },
|
||||
{ lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] },
|
||||
{ lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] }
|
||||
]
|
||||
},
|
||||
{ page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] },
|
||||
{ page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] }
|
||||
];
|
||||
const first = composeCandidate(input);
|
||||
const second = composeCandidate(structuredClone(input));
|
||||
|
||||
assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7");
|
||||
assert.equal(first.textSha256, sha256Hex(first.text));
|
||||
assert.deepEqual(first, second);
|
||||
assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [
|
||||
{ page: 1, candidateText: "Native first page.", risks: [] },
|
||||
{ page: 2, candidateText: "", risks: [] },
|
||||
{ page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] }
|
||||
]);
|
||||
assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages)));
|
||||
assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a"));
|
||||
assert.equal(first.pages[0]?.rawOcrText, "ignored OCR");
|
||||
assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06");
|
||||
});
|
||||
|
||||
test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => {
|
||||
assert.deepEqual(
|
||||
prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"),
|
||||
["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"]
|
||||
);
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue