From 76ebb352132c97065d82f7868966f8b4ca95ae48 Mon Sep 17 00:00:00 2001 From: Paco POR-CORREO Date: Wed, 16 Sep 2026 12:44:40 +0200 Subject: [PATCH] fix(ocr): detect raster-heavy pdf pages --- docs/CONTRATO_CICLO_VIDA_Y_OCR.md | 4 +- docs/HISTORIAL_SESIONES.md | 11 +++++- .../ocr-ingest-integration/apply-progress.md | 15 ++++++++ src/modules/ingest/service.ts | 4 +- src/modules/ocr/detection.ts | 9 +++-- src/modules/parsers/parser-registry.ts | 37 ++++++++++++++++++- tests/ocr/detection.test.ts | 10 ++++- tests/parsers/pdf-pages.test.ts | 3 ++ 8 files changed, 82 insertions(+), 11 deletions(-) diff --git a/docs/CONTRATO_CICLO_VIDA_Y_OCR.md b/docs/CONTRATO_CICLO_VIDA_Y_OCR.md index 8938eee..249d994 100644 --- a/docs/CONTRATO_CICLO_VIDA_Y_OCR.md +++ b/docs/CONTRATO_CICLO_VIDA_Y_OCR.md @@ -746,6 +746,7 @@ Despues de extraccion nativa calcular: - `A`: caracteres alfanumericos; - `W`: palabras o tokens; - `R`: proporcion de caracteres de reemplazo/control; +- `rasterCoverage`: page-area fraction painted by PDF.js raster operators, derived from transforms and capped at 1; intrinsic pixel dimensions and image count are not routing signals. - `inkCoverage`: proporcion no blanca calculada por el servicio OCR sobre un render de baja resolucion. La extraccion nativa debe usar un callback por pagina probado con fixture. Si `pdf-parse` no conserva paginas de forma fiable, se sustituye por una libreria Node con licencia permisiva antes de continuar; no se simulan paginas cortando el texto agregado. @@ -768,9 +769,10 @@ inkCoverage < 0.015 AND nonWhitespaceCharactersOCR < 10 Reglas: - `N == 0` siempre solicita inspeccion OCR; el servicio decide si esta vacia. +- Text-rich pages also request OCR when `rasterCoverage >= 0.05`; smaller raster marks such as logos do not force OCR. - Una pagina con `inkCoverage >= 0.015` no puede clasificarse como vacia aunque el OCR no encuentre texto: falla el quality gate. - Nunca decidir por promedio del documento. -- Umbrales configurables y registrados bajo `detectionPolicyVersion: "pdf-detection-v1"`. +- Umbrales configurables y registrados bajo `detectionPolicyVersion: "pdf-detection-v2"`. Composicion del candidato: diff --git a/docs/HISTORIAL_SESIONES.md b/docs/HISTORIAL_SESIONES.md index 6747beb..40f7ecb 100644 --- a/docs/HISTORIAL_SESIONES.md +++ b/docs/HISTORIAL_SESIONES.md @@ -3,13 +3,22 @@ **Proyecto:** Workspace de tools IA para empresas **Modulo:** RAG **Ultima actualizacion:** 2026-09-16 -**Ultima modificacion por:** Subagente Actualizacion Operativa OCR +**Ultima modificacion por:** Subagente Deteccion PDF Mixto **Estado:** Activo --- ## Registro de sesion +### 2026-09-16 - Subagente Deteccion PDF Mixto +**Agent:** Subagente Deteccion PDF Mixto · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5640de36ffekplsg06rgaNmX9` (subagent of `ses_29bdbd003ffeLrLjUlFgnp08Y7`) +**Responsibility:** Implement only Unit 16 raster-aware PDF detection and routing without production calls, candidate mutation, or task 7.4 completion. +**Work:** Added painted page-area telemetry from PDF.js operators, `pdf-detection-v2`, 5% routing with logo exclusion, fingerprint binding, focused tests, and the minimal canonical OCR contract update. +**Validation:** Strict-TDD RED 8/11; GREEN focused 11/11 and canonical 82/82; check/build/whitespace passed; the read-only real-PDF harness selected pages 1–25 and confirmed all four exact codes remain absent from native extraction. +**Files:** Parser, detection, ingest fingerprint, focused tests, OCR contract, OpenSpec apply progress, and this history. Task 7.4 remains pending; no commit or push. + +--- + ### 2026-09-16 - Subagente Actualizacion Operativa OCR **Agent:** Subagente Actualizacion Operativa OCR · **Model:** ollama/glm-5.3:cloud · **Session:** `ses_f5662ce40ffepxlOg2W0v9zrYv` (subagente de `ses_29bdbd003ffeLrLjUlFgnp08Y7`) **Responsibility:** Corregir el estado operativo verificado no secreto de EasyPanel/RAG/OCR y fijar el orden seguro de despliegue, sin commit ni push (el agente principal los entrega despues). diff --git a/openspec/changes/ocr-ingest-integration/apply-progress.md b/openspec/changes/ocr-ingest-integration/apply-progress.md index ac19e6e..ea3c2d7 100644 --- a/openspec/changes/ocr-ingest-integration/apply-progress.md +++ b/openspec/changes/ocr-ingest-integration/apply-progress.md @@ -471,3 +471,18 @@ None — the migration follows the proposal, specifications, design, and closed | Focused/regression | Focused 4/4; multipart runtime 1/1; canonical Node 81/81; check/build/whitespace clean. | | Runtime harness | Ephemeral localhost Express multipart uploads reused `Errores Junio 2026 - OCR verificado.md` across two physical PDF names, preserved `fallback.pdf` when omitted, rejected blank input `400`, kept activation false, and removed temporary files. | | Rollback boundary | Revert only Unit 14c deltas in `src/{app,api/openapi}.ts`, its two tests, and this progress/history metadata; Units 14a–14b, active content, and corpus remain intact. | + +## Unit 16: PDF Raster-Aware Routing +- Added PDF.js painted-area telemetry, `pdf-detection-v2`, a 5% raster threshold, and fingerprint binding; task 7.4 remains pending. + +### TDD Cycle Evidence +| Task | Safety Net | RED | GREEN / TRIANGULATE | REFACTOR | +|---|---|---|---|---| +| Unit 16 | Parser/detection 10/10 | 8/11; v1, missing telemetry, missing routing | 11/11; exact 5% routes and 4.99% logo stays native | Matrix transform readability; 11/11 remained green | + +### Work Unit Evidence +| Evidence | Result | +|---|---| +| Focused/canonical | Focused 11/11; canonical Node 82/82; check/build/whitespace clean. | +| Runtime harness | Real 25-page PDF reported raster coverage 1 on every page, selected pages 1–25, and native text still lacked CBG04a/FAT07/DSAU08/NSAV06. | +| Rollback boundary | Revert Unit 16 parser/detection/ingest/test/contract deltas and this metadata; preserve Units 1–15, candidates, and active content. | diff --git a/src/modules/ingest/service.ts b/src/modules/ingest/service.ts index 9ff965a..28cc71b 100644 --- a/src/modules/ingest/service.ts +++ b/src/modules/ingest/service.ts @@ -22,7 +22,7 @@ import { import { listFilesRecursively } from "../../shared/utils/files.js"; import { env } from "../../config/env.js"; import { CatalogError, type CatalogDocumentInput, type CatalogRepository } from "../catalog/repository.js"; -import { selectPdfPagesForOcr } from "../ocr/detection.js"; +import { DETECTION_POLICY_VERSION, selectPdfPagesForOcr } from "../ocr/detection.js"; import { stageOcrArtifacts } from "../ocr/artifacts.js"; type PreparedDocument = CatalogDocumentInput & { @@ -388,7 +388,7 @@ export class IngestService { ): Promise { const processingFingerprint = buildProcessingFingerprint({ parserVersion: "native-pages-v1+ocr-v1", - detectionPolicyVersion: "pdf-detection-v1", + detectionPolicyVersion: DETECTION_POLICY_VERSION, normalizationPolicy: "bom-crlf-trim-final-lf-v1", chunking: { code: codeChunkingPolicy, documental: documentalChunkingPolicy }, embeddingProvider: this.embeddingProvider.providerName, diff --git a/src/modules/ocr/detection.ts b/src/modules/ocr/detection.ts index 8a77edc..0a31793 100644 --- a/src/modules/ocr/detection.ts +++ b/src/modules/ocr/detection.ts @@ -1,6 +1,7 @@ import path from "node:path"; -export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const; +export const DETECTION_POLICY_VERSION = "pdf-detection-v2" as const; +export const MIN_RASTER_COVERAGE = 0.05; export interface NativeTextMetrics { nonWhitespaceCharacters: number; @@ -40,14 +41,14 @@ export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean { export function selectPdfPagesForOcr( filePath: string, - pages: ReadonlyArray<{ page: number; text: string }> + pages: ReadonlyArray<{ page: number; text: string; rasterCoverage?: number }> ): number[] { if (path.extname(filePath).toLowerCase() !== ".pdf") return []; const selected = new Set(); - for (const { page, text } of pages) { + for (const { page, text, rasterCoverage = 0 } of pages) { if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers"); - if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page); + if (!isNativeTextSufficient(computeNativeMetrics(text)) || rasterCoverage >= MIN_RASTER_COVERAGE) selected.add(page); } return [...selected].sort((left, right) => left - right); } diff --git a/src/modules/parsers/parser-registry.ts b/src/modules/parsers/parser-registry.ts index c4127d0..e19f5bd 100644 --- a/src/modules/parsers/parser-registry.ts +++ b/src/modules/parsers/parser-registry.ts @@ -15,13 +15,47 @@ export interface ParsedDocument { export interface ParsedPdfPage { page: number; text: string; + rasterCoverage?: number; textSha256: string; } interface PdfPageData { + view: [number, number, number, number]; getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{ items: Array<{ str?: string }>; }>; + getOperatorList(): Promise<{ fnArray: number[]; argsArray: unknown[][] }>; +} + +type Matrix = [number, number, number, number, number, number]; + +function multiply(left: Matrix, right: Matrix): Matrix { + return [ + left[0] * right[0] + left[2] * right[1], + left[1] * right[0] + left[3] * right[1], + left[0] * right[2] + left[2] * right[3], + left[1] * right[2] + left[3] * right[3], + left[0] * right[4] + left[2] * right[5] + left[4], + left[1] * right[4] + left[3] * right[5] + left[5] + ]; +} + +function computeRasterCoverage(page: PdfPageData, operators: Awaited>): number { + let current: Matrix = [1, 0, 0, 1, 0, 0]; + const stack: Matrix[] = []; + let paintedArea = 0; + const determinant = (matrix: Matrix) => Math.abs(matrix[0] * matrix[3] - matrix[1] * matrix[2]); + for (const [index, operator] of operators.fnArray.entries()) { + const args = operators.argsArray[index]!; + if (operator === 10) stack.push([...current]); + else if (operator === 11) current = stack.pop() ?? current; + else if (operator === 12) current = multiply(current, args as Matrix); + else if ([82, 85, 86].includes(operator)) paintedArea += determinant(current); + else if (operator === 88) paintedArea += determinant(current) * Math.abs(Number(args[1]) * Number(args[2])) * ((args[3] as ArrayLike).length / 2); + else if (operator === 87) paintedArea += (args[1] as Array<{ transform: Matrix }>).reduce((area, entry) => area + determinant(multiply(current, entry.transform)), 0); + } + const [x1, y1, x2, y2] = page.view; + return Math.min(1, paintedArea / ((x2 - x1) * (y2 - y1))); } const documentalExtensions = [".md", ".txt", ".pdf"] as const; @@ -75,8 +109,9 @@ export async function parsePdfPages(filePath: string | URL): Promise item.str ?? []).join(" ").trim(); - pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) }); + pages.push({ page: pages.length + 1, text, rasterCoverage: computeRasterCoverage(pageData, operatorList), textSha256: sha256Hex(text) }); return text; } }); diff --git a/tests/ocr/detection.test.ts b/tests/ocr/detection.test.ts index 43ff965..fd2b185 100644 --- a/tests/ocr/detection.test.ts +++ b/tests/ocr/detection.test.ts @@ -47,10 +47,10 @@ function buildPdf(pageTexts: string[]): Buffer { const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" "); -test("native detection applies every pdf-detection-v1 boundary per page", () => { +test("native detection applies every pdf-detection-v2 boundary per page", () => { const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 }; - assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1"); + assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v2"); assert.deepEqual( [boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }] .map(isNativeTextSufficient), @@ -64,6 +64,12 @@ test("native detection applies every pdf-detection-v1 boundary per page", () => }); }); +test("text-rich pages route at the raster threshold while small logos stay native", () => { + const page = { page: 1, text: sufficientText }; + assert.deepEqual(selectPdfPagesForOcr("visual.pdf", [{ ...page, rasterCoverage: 0.05 }]), [1]); + assert.deepEqual(selectPdfPagesForOcr("logo.pdf", [{ ...page, rasterCoverage: 0.0499 }]), []); +}); + test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => { const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-")); const filePath = path.join(directory, "mixed.PDF"); diff --git a/tests/parsers/pdf-pages.test.ts b/tests/parsers/pdf-pages.test.ts index db889f3..5771e3e 100644 --- a/tests/parsers/pdf-pages.test.ts +++ b/tests/parsers/pdf-pages.test.ts @@ -18,16 +18,19 @@ test("parsePdfPages extracts ordered one-based pages with stable native text has { page: 1, text: "Native page one.", + rasterCoverage: 0, textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4" }, { page: 2, text: "", + rasterCoverage: 0, textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" }, { page: 3, text: "Native page three.", + rasterCoverage: 0, textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e" } ]);