fix(ocr): detect raster-heavy pdf pages
This commit is contained in:
parent
fd35e6a6f1
commit
76ebb35213
8 changed files with 82 additions and 11 deletions
|
|
@ -746,6 +746,7 @@ Despues de extraccion nativa calcular:
|
|||
- `A`: caracteres alfanumericos;
|
||||
- `W`: palabras o tokens;
|
||||
- `R`: proporcion de caracteres de reemplazo/control;
|
||||
- `rasterCoverage`: page-area fraction painted by PDF.js raster operators, derived from transforms and capped at 1; intrinsic pixel dimensions and image count are not routing signals.
|
||||
- `inkCoverage`: proporcion no blanca calculada por el servicio OCR sobre un render de baja resolucion.
|
||||
|
||||
La extraccion nativa debe usar un callback por pagina probado con fixture. Si `pdf-parse` no conserva paginas de forma fiable, se sustituye por una libreria Node con licencia permisiva antes de continuar; no se simulan paginas cortando el texto agregado.
|
||||
|
|
@ -768,9 +769,10 @@ inkCoverage < 0.015 AND nonWhitespaceCharactersOCR < 10
|
|||
Reglas:
|
||||
|
||||
- `N == 0` siempre solicita inspeccion OCR; el servicio decide si esta vacia.
|
||||
- Text-rich pages also request OCR when `rasterCoverage >= 0.05`; smaller raster marks such as logos do not force OCR.
|
||||
- Una pagina con `inkCoverage >= 0.015` no puede clasificarse como vacia aunque el OCR no encuentre texto: falla el quality gate.
|
||||
- Nunca decidir por promedio del documento.
|
||||
- Umbrales configurables y registrados bajo `detectionPolicyVersion: "pdf-detection-v1"`.
|
||||
- Umbrales configurables y registrados bajo `detectionPolicyVersion: "pdf-detection-v2"`.
|
||||
|
||||
Composicion del candidato:
|
||||
|
||||
|
|
|
|||
|
|
@ -3,13 +3,22 @@
|
|||
**Proyecto:** Workspace de tools IA para empresas
|
||||
**Modulo:** RAG
|
||||
**Ultima actualizacion:** 2026-09-16
|
||||
**Ultima modificacion por:** Subagente Actualizacion Operativa OCR
|
||||
**Ultima modificacion por:** Subagente Deteccion PDF Mixto
|
||||
**Estado:** Activo
|
||||
|
||||
---
|
||||
|
||||
## Registro de sesion
|
||||
|
||||
### 2026-09-16 - Subagente Deteccion PDF Mixto
|
||||
**Agent:** Subagente Deteccion PDF Mixto · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5640de36ffekplsg06rgaNmX9` (subagent of `ses_29bdbd003ffeLrLjUlFgnp08Y7`)
|
||||
**Responsibility:** Implement only Unit 16 raster-aware PDF detection and routing without production calls, candidate mutation, or task 7.4 completion.
|
||||
**Work:** Added painted page-area telemetry from PDF.js operators, `pdf-detection-v2`, 5% routing with logo exclusion, fingerprint binding, focused tests, and the minimal canonical OCR contract update.
|
||||
**Validation:** Strict-TDD RED 8/11; GREEN focused 11/11 and canonical 82/82; check/build/whitespace passed; the read-only real-PDF harness selected pages 1–25 and confirmed all four exact codes remain absent from native extraction.
|
||||
**Files:** Parser, detection, ingest fingerprint, focused tests, OCR contract, OpenSpec apply progress, and this history. Task 7.4 remains pending; no commit or push.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-16 - Subagente Actualizacion Operativa OCR
|
||||
**Agent:** Subagente Actualizacion Operativa OCR · **Model:** ollama/glm-5.3:cloud · **Session:** `ses_f5662ce40ffepxlOg2W0v9zrYv` (subagente de `ses_29bdbd003ffeLrLjUlFgnp08Y7`)
|
||||
**Responsibility:** Corregir el estado operativo verificado no secreto de EasyPanel/RAG/OCR y fijar el orden seguro de despliegue, sin commit ni push (el agente principal los entrega despues).
|
||||
|
|
|
|||
|
|
@ -471,3 +471,18 @@ None — the migration follows the proposal, specifications, design, and closed
|
|||
| Focused/regression | Focused 4/4; multipart runtime 1/1; canonical Node 81/81; check/build/whitespace clean. |
|
||||
| Runtime harness | Ephemeral localhost Express multipart uploads reused `Errores Junio 2026 - OCR verificado.md` across two physical PDF names, preserved `fallback.pdf` when omitted, rejected blank input `400`, kept activation false, and removed temporary files. |
|
||||
| Rollback boundary | Revert only Unit 14c deltas in `src/{app,api/openapi}.ts`, its two tests, and this progress/history metadata; Units 14a–14b, active content, and corpus remain intact. |
|
||||
|
||||
## Unit 16: PDF Raster-Aware Routing
|
||||
- Added PDF.js painted-area telemetry, `pdf-detection-v2`, a 5% raster threshold, and fingerprint binding; task 7.4 remains pending.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Safety Net | RED | GREEN / TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|
|
||||
| Unit 16 | Parser/detection 10/10 | 8/11; v1, missing telemetry, missing routing | 11/11; exact 5% routes and 4.99% logo stays native | Matrix transform readability; 11/11 remained green |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused/canonical | Focused 11/11; canonical Node 82/82; check/build/whitespace clean. |
|
||||
| Runtime harness | Real 25-page PDF reported raster coverage 1 on every page, selected pages 1–25, and native text still lacked CBG04a/FAT07/DSAU08/NSAV06. |
|
||||
| Rollback boundary | Revert Unit 16 parser/detection/ingest/test/contract deltas and this metadata; preserve Units 1–15, candidates, and active content. |
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ import {
|
|||
import { listFilesRecursively } from "../../shared/utils/files.js";
|
||||
import { env } from "../../config/env.js";
|
||||
import { CatalogError, type CatalogDocumentInput, type CatalogRepository } from "../catalog/repository.js";
|
||||
import { selectPdfPagesForOcr } from "../ocr/detection.js";
|
||||
import { DETECTION_POLICY_VERSION, selectPdfPagesForOcr } from "../ocr/detection.js";
|
||||
import { stageOcrArtifacts } from "../ocr/artifacts.js";
|
||||
|
||||
type PreparedDocument = CatalogDocumentInput & {
|
||||
|
|
@ -388,7 +388,7 @@ export class IngestService {
|
|||
): Promise<OcrAccepted> {
|
||||
const processingFingerprint = buildProcessingFingerprint({
|
||||
parserVersion: "native-pages-v1+ocr-v1",
|
||||
detectionPolicyVersion: "pdf-detection-v1",
|
||||
detectionPolicyVersion: DETECTION_POLICY_VERSION,
|
||||
normalizationPolicy: "bom-crlf-trim-final-lf-v1",
|
||||
chunking: { code: codeChunkingPolicy, documental: documentalChunkingPolicy },
|
||||
embeddingProvider: this.embeddingProvider.providerName,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import path from "node:path";
|
||||
|
||||
export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const;
|
||||
export const DETECTION_POLICY_VERSION = "pdf-detection-v2" as const;
|
||||
export const MIN_RASTER_COVERAGE = 0.05;
|
||||
|
||||
export interface NativeTextMetrics {
|
||||
nonWhitespaceCharacters: number;
|
||||
|
|
@ -40,14 +41,14 @@ export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean {
|
|||
|
||||
export function selectPdfPagesForOcr(
|
||||
filePath: string,
|
||||
pages: ReadonlyArray<{ page: number; text: string }>
|
||||
pages: ReadonlyArray<{ page: number; text: string; rasterCoverage?: number }>
|
||||
): number[] {
|
||||
if (path.extname(filePath).toLowerCase() !== ".pdf") return [];
|
||||
|
||||
const selected = new Set<number>();
|
||||
for (const { page, text } of pages) {
|
||||
for (const { page, text, rasterCoverage = 0 } of pages) {
|
||||
if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers");
|
||||
if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page);
|
||||
if (!isNativeTextSufficient(computeNativeMetrics(text)) || rasterCoverage >= MIN_RASTER_COVERAGE) selected.add(page);
|
||||
}
|
||||
return [...selected].sort((left, right) => left - right);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -15,13 +15,47 @@ export interface ParsedDocument {
|
|||
export interface ParsedPdfPage {
|
||||
page: number;
|
||||
text: string;
|
||||
rasterCoverage?: number;
|
||||
textSha256: string;
|
||||
}
|
||||
|
||||
interface PdfPageData {
|
||||
view: [number, number, number, number];
|
||||
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||
items: Array<{ str?: string }>;
|
||||
}>;
|
||||
getOperatorList(): Promise<{ fnArray: number[]; argsArray: unknown[][] }>;
|
||||
}
|
||||
|
||||
type Matrix = [number, number, number, number, number, number];
|
||||
|
||||
function multiply(left: Matrix, right: Matrix): Matrix {
|
||||
return [
|
||||
left[0] * right[0] + left[2] * right[1],
|
||||
left[1] * right[0] + left[3] * right[1],
|
||||
left[0] * right[2] + left[2] * right[3],
|
||||
left[1] * right[2] + left[3] * right[3],
|
||||
left[0] * right[4] + left[2] * right[5] + left[4],
|
||||
left[1] * right[4] + left[3] * right[5] + left[5]
|
||||
];
|
||||
}
|
||||
|
||||
function computeRasterCoverage(page: PdfPageData, operators: Awaited<ReturnType<PdfPageData["getOperatorList"]>>): number {
|
||||
let current: Matrix = [1, 0, 0, 1, 0, 0];
|
||||
const stack: Matrix[] = [];
|
||||
let paintedArea = 0;
|
||||
const determinant = (matrix: Matrix) => Math.abs(matrix[0] * matrix[3] - matrix[1] * matrix[2]);
|
||||
for (const [index, operator] of operators.fnArray.entries()) {
|
||||
const args = operators.argsArray[index]!;
|
||||
if (operator === 10) stack.push([...current]);
|
||||
else if (operator === 11) current = stack.pop() ?? current;
|
||||
else if (operator === 12) current = multiply(current, args as Matrix);
|
||||
else if ([82, 85, 86].includes(operator)) paintedArea += determinant(current);
|
||||
else if (operator === 88) paintedArea += determinant(current) * Math.abs(Number(args[1]) * Number(args[2])) * ((args[3] as ArrayLike<number>).length / 2);
|
||||
else if (operator === 87) paintedArea += (args[1] as Array<{ transform: Matrix }>).reduce((area, entry) => area + determinant(multiply(current, entry.transform)), 0);
|
||||
}
|
||||
const [x1, y1, x2, y2] = page.view;
|
||||
return Math.min(1, paintedArea / ((x2 - x1) * (y2 - y1)));
|
||||
}
|
||||
|
||||
const documentalExtensions = [".md", ".txt", ".pdf"] as const;
|
||||
|
|
@ -75,8 +109,9 @@ export async function parsePdfPages(filePath: string | URL): Promise<ParsedPdfPa
|
|||
normalizeWhitespace: false,
|
||||
disableCombineTextItems: false
|
||||
});
|
||||
const operatorList = await pageData.getOperatorList();
|
||||
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||
pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) });
|
||||
pages.push({ page: pages.length + 1, text, rasterCoverage: computeRasterCoverage(pageData, operatorList), textSha256: sha256Hex(text) });
|
||||
return text;
|
||||
}
|
||||
});
|
||||
|
|
|
|||
|
|
@ -47,10 +47,10 @@ function buildPdf(pageTexts: string[]): Buffer {
|
|||
|
||||
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
|
||||
|
||||
test("native detection applies every pdf-detection-v1 boundary per page", () => {
|
||||
test("native detection applies every pdf-detection-v2 boundary per page", () => {
|
||||
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
|
||||
|
||||
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1");
|
||||
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v2");
|
||||
assert.deepEqual(
|
||||
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
|
||||
.map(isNativeTextSufficient),
|
||||
|
|
@ -64,6 +64,12 @@ test("native detection applies every pdf-detection-v1 boundary per page", () =>
|
|||
});
|
||||
});
|
||||
|
||||
test("text-rich pages route at the raster threshold while small logos stay native", () => {
|
||||
const page = { page: 1, text: sufficientText };
|
||||
assert.deepEqual(selectPdfPagesForOcr("visual.pdf", [{ ...page, rasterCoverage: 0.05 }]), [1]);
|
||||
assert.deepEqual(selectPdfPagesForOcr("logo.pdf", [{ ...page, rasterCoverage: 0.0499 }]), []);
|
||||
});
|
||||
|
||||
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
|
||||
const filePath = path.join(directory, "mixed.PDF");
|
||||
|
|
|
|||
|
|
@ -18,16 +18,19 @@ test("parsePdfPages extracts ordered one-based pages with stable native text has
|
|||
{
|
||||
page: 1,
|
||||
text: "Native page one.",
|
||||
rasterCoverage: 0,
|
||||
textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4"
|
||||
},
|
||||
{
|
||||
page: 2,
|
||||
text: "",
|
||||
rasterCoverage: 0,
|
||||
textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
|
||||
},
|
||||
{
|
||||
page: 3,
|
||||
text: "Native page three.",
|
||||
rasterCoverage: 0,
|
||||
textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e"
|
||||
}
|
||||
]);
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue