From 0654ce58eec7781ee7f742bd325c74d480f5320a Mon Sep 17 00:00:00 2001 From: Paco POR-CORREO Date: Wed, 23 Sep 2026 13:00:27 +0200 Subject: [PATCH] feat(ocr): harden quality diagnostics and release contract --- AGENTS.md | 26 ++ Dockerfile | 12 +- TEMPLATE_VERSION | 1 + VERSION | 1 + docs/ACTUALIZACION_PLANTILLA.md | 15 + docs/CONTEXTO_PROYECTO.md | 39 +++ docs/CONTRATO_CICLO_VIDA_Y_OCR.md | 306 ++++++++++++++++++++- docs/ESTRUCTURA_CARPETAS_EMPRESA.md | 16 ++ docs/HISTORIAL_SESIONES.md | 110 +++++++- docs/INDICE_DOCUMENTACION.md | 48 ++++ docs/LEARNED_SKILLS.md | 16 ++ docs/OPERATIVA.md | 21 +- docs/PENDIENTES_RAG.md | 52 +++- docs/REGISTRO_SITUACIONES.md | 83 ++++++ docs/info-git-por-correo.md | 158 +++++++++++ docs/readme.md | 34 +++ docs/sesion_actual_opencode.md | 19 ++ migrations/004_ocr_quality_diagnostics.sql | 19 ++ ocr-service/Dockerfile | 4 +- ocr-service/Dockerfile.dockerignore | 1 + ocr-service/app/jobs.py | 38 ++- ocr-service/app/main.py | 27 +- ocr-service/app/render.py | 2 +- ocr-service/app/version.py | 12 + ocr-service/tests/test_api.py | 56 +++- ocr-service/tests/test_render.py | 3 +- opencode.json | 8 + package-lock.json | 4 +- package.json | 2 +- src/api/openapi.ts | 4 +- src/app.ts | 10 +- src/config/env.ts | 3 +- src/config/version.ts | 14 + src/modules/catalog/reconciler.ts | 6 +- src/modules/catalog/repository.ts | 172 ++++++++++-- src/modules/ingest/service.ts | 7 +- src/modules/ocr/artifacts.ts | 206 ++++++++++++-- src/modules/ocr/client.ts | 99 ++++--- src/modules/ocr/composition.ts | 15 +- src/modules/ocr/detection.ts | 50 +++- src/modules/ocr/dispatcher.ts | 68 ++++- src/modules/ocr/review.ts | 52 +++- tests/catalog/migration-004.test.ts | 15 + tests/catalog/repository-ocr.test.ts | 81 +++++- tests/ocr/client.test.ts | 46 ++-- tests/ocr/contracts-deploy.test.ts | 18 +- tests/ocr/detection.test.ts | 39 ++- tests/ocr/dispatcher.test.ts | 157 ++++++++++- tests/ocr/e2e.test.ts | 2 +- tests/ocr/indexing-store.test.ts | 53 ++-- tests/ocr/quality-diagnostics.test.ts | 139 ++++++++++ tests/ocr/review.test.ts | 10 +- 52 files changed, 2166 insertions(+), 233 deletions(-) create mode 100644 AGENTS.md create mode 100644 TEMPLATE_VERSION create mode 100644 VERSION create mode 100644 docs/ACTUALIZACION_PLANTILLA.md create mode 100644 docs/CONTEXTO_PROYECTO.md create mode 100644 docs/ESTRUCTURA_CARPETAS_EMPRESA.md create mode 100644 docs/INDICE_DOCUMENTACION.md create mode 100644 docs/LEARNED_SKILLS.md create mode 100644 docs/REGISTRO_SITUACIONES.md create mode 100644 docs/info-git-por-correo.md create mode 100644 docs/readme.md create mode 100644 docs/sesion_actual_opencode.md create mode 100644 migrations/004_ocr_quality_diagnostics.sql create mode 100644 ocr-service/app/version.py create mode 100644 opencode.json create mode 100644 src/config/version.ts create mode 100644 tests/catalog/migration-004.test.ts create mode 100644 tests/ocr/quality-diagnostics.test.ts diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..2f5aded --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,26 @@ +# RAG - Punto de entrada para agentes + +El protocolo general de trabajo, identidad y registro vive en `docs/readme.md`. + +## Reglas del modulo RAG + +- Este worktree corresponde exclusivamente al modulo `RAG/`. +- El unico proyecto canonico de Engram para este repositorio es `rag-service`. +- Antes de realizar trabajo RAG, confirma que OpenCode se inicio desde la raiz de este worktree y que `mem_current_project` resuelve `rag-service`. +- Todas las lecturas y escrituras de Engram para RAG deben indicar explicitamente `project: "rag-service"`. +- Si la deteccion no devuelve `rag-service`, no escribas memoria bajo otro proyecto. Deten el flujo e informa que la sesion debe reiniciarse desde `RAG/`. +- `all_projects=true` queda reservado para recuperaciones o consolidaciones autorizadas expresamente por el usuario. +- El contrato y el backlog canonicos del OCR viven en `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` y `docs/PENDIENTES_RAG.md`. +- Antes de tareas con servicios, credenciales, despliegues o ejecuciones recurrentes, consulta `docs/OPERATIVA.md`. +- Respeta la aprobacion explicita del usuario antes de modificar archivos, ejecutar comandos relevantes o interactuar con infraestructura. +- Registra los cambios relevantes en `docs/HISTORIAL_SESIONES.md` antes de cerrar el bloque de trabajo. + +## Documentacion principal + +- `docs/readme.md` - Protocolo de agentes. +- `docs/CONTEXTO_PROYECTO.md` - Ficha y estado del modulo. +- `docs/HISTORIAL_SESIONES.md` - Continuidad de sesiones. +- `docs/INDICE_DOCUMENTACION.md` - Mapa de la documentacion. +- `docs/OPERATIVA.md` - Hechos operativos sin secretos. + +La version de esta plantilla se controla en `TEMPLATE_VERSION`. diff --git a/Dockerfile b/Dockerfile index f896442..093d5bf 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,8 +1,10 @@ FROM node:22-bookworm-slim AS build -ARG RAG_VERSION=0.2.0 +ARG RAG_VERSION ARG BUILD_REVISION=unknown WORKDIR /app -COPY package.json package-lock.json tsconfig.json ./ +COPY VERSION package.json package-lock.json tsconfig.json ./ +RUN test "$RAG_VERSION" = "$(tr -d '\r\n' < VERSION)" \ + && node -e 'process.exit(require("./package.json").version === require("node:fs").readFileSync("VERSION", "utf8").trim() ? 0 : 1)' RUN npm ci COPY src ./src COPY migrations ./migrations @@ -10,13 +12,15 @@ COPY public ./public RUN npm run build FROM node:22-bookworm-slim AS runtime -ARG RAG_VERSION=0.2.0 +ARG RAG_VERSION ARG BUILD_REVISION=unknown WORKDIR /app ENV NODE_ENV=production OCR_ARTIFACT_ROOT=/data/ingestions RAG_VERSION=${RAG_VERSION} BUILD_REVISION=${BUILD_REVISION} LABEL org.opencontainers.image.version=${RAG_VERSION} \ org.opencontainers.image.revision=${BUILD_REVISION} -COPY package.json package-lock.json ./ +COPY VERSION package.json package-lock.json ./ +RUN test "$RAG_VERSION" = "$(tr -d '\r\n' < VERSION)" \ + && node -e 'process.exit(require("./package.json").version === require("node:fs").readFileSync("VERSION", "utf8").trim() ? 0 : 1)' RUN npm ci --omit=dev COPY --from=build /app/dist ./dist COPY --from=build /app/migrations ./migrations diff --git a/TEMPLATE_VERSION b/TEMPLATE_VERSION new file mode 100644 index 0000000..c04c650 --- /dev/null +++ b/TEMPLATE_VERSION @@ -0,0 +1 @@ +1.2.7 diff --git a/VERSION b/VERSION new file mode 100644 index 0000000..0c62199 --- /dev/null +++ b/VERSION @@ -0,0 +1 @@ +0.2.1 diff --git a/docs/ACTUALIZACION_PLANTILLA.md b/docs/ACTUALIZACION_PLANTILLA.md new file mode 100644 index 0000000..d5a4853 --- /dev/null +++ b/docs/ACTUALIZACION_PLANTILLA.md @@ -0,0 +1,15 @@ +# Actualizacion de plantilla documental + +## Principio general + +Actualizar la plantilla es una migracion controlada, no una copia completa encima del workspace. + +1. Lee `TEMPLATE_VERSION` del workspace y la version maestra. +2. Consulta en el changelog maestro solo los bloques posteriores a la version local. +3. Aplica los bloques en orden y compara solo los archivos que indiquen. +4. Conserva siempre historial, contexto, backlog, bitacoras y documentacion especifica del modulo. +5. Actualiza `TEMPLATE_VERSION` solo despues de validar y registrar la migracion. + +## Workspaces no versionados + +Si existe `docs/` pero no `TEMPLATE_VERSION`, trata el workspace como no versionado. Revisa diferencias y solicita aprobacion antes de aplicar una migracion inicial controlada. diff --git a/docs/CONTEXTO_PROYECTO.md b/docs/CONTEXTO_PROYECTO.md new file mode 100644 index 0000000..feca674 --- /dev/null +++ b/docs/CONTEXTO_PROYECTO.md @@ -0,0 +1,39 @@ +# Ficha rapida del proyecto + +## Proyecto: RAG Service + +### Descripcion + +Servicio RAG para ingesta, versionado y recuperacion de conocimiento. Incluye un servicio OCR privado reutilizable: RAG lo consume para procesar documentos con imagenes, pero OCR no es exclusivo del modulo. + +### Stack tecnologico + +- **Backend:** Node.js y TypeScript. +- **OCR:** Python, FastAPI y PaddleOCR. +- **Datos:** PostgreSQL, Qdrant y SQLite para la cola OCR. +- **Infraestructura:** Docker y EasyPanel. + +### Estado actual + +- [x] Activo +- [x] En desarrollo +- [x] En produccion + +**Ultima sesion:** 2026-09-22 +**Ultimo agente:** Agente RAG 3 + +### Ubicacion del proyecto + +`/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG` + +### Documentacion clave + +- `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` - Contrato canonico del ciclo de vida y OCR. +- `docs/PENDIENTES_RAG.md` - Backlog priorizado. +- `docs/OPERATIVA.md` - Operacion sin secretos. +- `docs/HISTORIAL_SESIONES.md` - Continuidad de sesiones. + +### Notas rapidas + +- `rag-service` es el unico proyecto Engram canonico; la sesion debe iniciarse desde la raiz de este worktree. +- R4 de OCR `0.2.1` esta completada localmente: `004` se valido en PostgreSQL Docker aislado y detenido. R5 es la unica fase que puede desplegar o tocar produccion, con autorizacion explicita. diff --git a/docs/CONTRATO_CICLO_VIDA_Y_OCR.md b/docs/CONTRATO_CICLO_VIDA_Y_OCR.md index 0bed216..ce1ade6 100644 --- a/docs/CONTRATO_CICLO_VIDA_Y_OCR.md +++ b/docs/CONTRATO_CICLO_VIDA_Y_OCR.md @@ -2,7 +2,7 @@ **Proyecto:** Workspace de tools IA para empresas **Modulo:** RAG -**Ultima actualizacion:** 2026-09-21 +**Ultima actualizacion:** 2026-09-22 **Ultima modificacion por:** Agente RAG 2 **Estado:** Punto 2 validado; pipeline OCR desplegado; aceptacion productiva bloqueada hasta completar el hardening de runtime definido en este contrato @@ -785,7 +785,9 @@ Composicion del candidato: - ordenar por `page_number` ascendente y separar paginas no vacias con `\n\n--- Page ---\n\n`; - no concatenar texto nativo insuficiente con OCR porque duplica y mezcla variantes. -## Contrato interno del servicio OCR +## Contrato interno del servicio OCR `ocr-v1` + +Esta seccion describe el contrato base entregado en `0.2.0`. Para `0.2.1`, las subsecciones `Identidad y compatibilidad ocr-v2`, `Calidad bloqueante frente a advertencias` y `Diagnostico durable` del anexo correctivo sustituyen los payloads de identidad, respuestas de estado/resultado/imagen y quality gate aqui definidos. Los campos antiguos no se consideran suficientes para una solicitud `ocr-v2`. ### Crear trabajo @@ -897,7 +899,9 @@ GET /health/ready Errores: `400` solicitud invalida, `401/403` autenticacion, `409` conflicto de idempotencia, `413` limite excedido, `422` PDF corrupto/cifrado/no soportado, `429` cola llena y `500/503` fallo de motor. -## Quality gate OCR +## Quality gate OCR `ocr-v1` + +Esta puerta documenta el comportamiento de `0.2.0`. Queda sustituida exclusivamente para trabajos `ocr-v2` por la tabla y la persistencia del anexo correctivo. En `ocr-v2`, `p10Confidence < 0.50` aislado es una advertencia y no ejecuta la regla de fallo de esta seccion. Una pagina OCR produce candidato mecanicamente revisable si cumple: @@ -1307,6 +1311,8 @@ Cada punto se entrega como bloque independiente: 8. rollback por Git y redeploy si falla; 9. cerrar el punto en `PENDIENTES_RAG.md` solo con evidencia. +Esta es la regla general de las fases anteriores. Para la correccion `0.2.1`, el anexo siguiente la sustituye de forma explicita: R1-R4 se implementan, prueban y documentan sin desplegar ni ejecutar pruebas productivas parciales. Solo R5 puede desplegar, y debe hacerlo como una unica operacion controlada que incluye RAG y OCR, seguida de pruebas conjuntas de ambos servicios. `OCR_INGEST_ENABLED` permanece en `true`: no se desactiva preventivamente ni se despliega por etapas salvo que un fallo real y su evidencia justifiquen aislar un servicio. La entrega no se considera completada hasta que ambos servicios coincidan en version y revision. + ## Exclusiones explicitas - No corregir texto OCR con un LLM en esta fase. @@ -1318,3 +1324,297 @@ Cada punto se entrega como bloque independiente: - No conservar un camino legacy indefinido despues de migrar. - No iniciar busqueda hibrida; corresponde al pendiente 4. - No presentar el servicio OCR en un dominio publico. + +# Anexo correctivo: hallazgos productivos v6-v7 + +## Decision y alcance + +La aceptacion productiva de FacturaTech queda pausada. Antes de crear otra candidata se entrega una correccion conjunta RAG/OCR `0.2.1` que resuelve los defectos observados en v6 y v7 sin aprobar, indexar ni activar contenido. + +Este anexo es el contrato canonico de la correccion. No autoriza su implementacion: el usuario debe revisar este diseno y dar un visto bueno explicito antes de modificar codigo. + +La correccion incluye: + +- identidad idempotente versionada para no reutilizar trabajos de contratos incompatibles; +- recuperacion segura de respuestas idempotentes `queued`, `running`, `succeeded` y `failed`; +- separacion entre fallo de calidad bloqueante y advertencia que requiere revision humana; +- diagnostico durable y estructurado de las paginas que no superen calidad; +- validacion local, independiente y productiva antes de aprobar o activar. + +Quedan fuera: + +- corregir automaticamente texto o codigos con reglas, expresiones regulares o LLM; +- relajar silenciosamente todos los umbrales de calidad; +- reutilizar trabajos legacy que no garanticen PNG persistidos; +- recuperar, aprobar o activar v5, v6 o v7; +- borrar las versiones fallidas de PostgreSQL, que permanecen como evidencia auditable. + +## Evidencia que obliga a corregir + +| Intento | Resultado | Hallazgo | +|---|---|---| +| v6 `b9281626-088a-46e5-9f66-b4a3bbe2159f` | Fallo inmediato, `0/25` | El mismo PDF, paginas y `configVersion=ocr-v1` reutilizaron la clave de un trabajo terminal anterior. OCR devolvio `202` con la fila existente, cuyo estado ya no era `queued`; RAG rechazo el acknowledgement como fallo generico de integridad. | +| v7 `c9756e59-f91d-4b57-8f31-e7b5b2b09558` | OCR `25/25`, despues `OCR_QUALITY_BLOCKED` | Las paginas 2, 20 y 21 fallaron solo por `p10Confidence < 0.5`. Sus medianas fueron `0.983-0.999`, el ratio de lineas de baja confianza `0.122-0.158` y el contenido era legible. | + +La evidencia durable de v7 esta completa: original, paginas nativas, resultado OCR y 25 PNG unicos; todos los hashes y firmas PNG son validos y las imagenes ocupan `12.556.918` bytes. Las 34 entradas esperadas estan presentes. Al considerar que una entrada puede continuar en la pagina siguiente, la coincidencia minima por tokens es `88,9 %` para mensaje y `100 %` para causa y solucion. + +El OCR produjo tres ambiguedades que deben llegar a revision humana, no provocar correccion automatica: + +| Esperado | OCR v7 | Pagina | +|---|---|---| +| `FAT07` | `FATo7` | 8 | +| `NSAV06` | `NSAvo6` | 15 | +| `DSAU04` | `DSAUo4` | 16 | + +`CBG04a` y `DSAU08` se recuperaron exactamente. La version activa `3fc78163-9cfb-4979-985c-1520a63327b0` permanecio intacta durante ambos intentos. + +### Limite de la evidencia historica + +Las cifras anteriores describen la inspeccion productiva ya realizada, pero no sustituyen evidencia reproducible dentro del repositorio. El documento manual de contraste vive fuera de este repositorio en `Empresa/Clientes/FacturaTech/Fidi/Info-para-Agente-Comercial/Errores Junio 2026 - OCR verificado.md`; por tanto, cualquier afirmacion que dependa de su contenido se etiqueta como verificacion manual externa y no obtiene PASS por si sola. + +R4 debe aportar fixtures o generadores autocontenidos para probar identidad, recuperacion, calidad e integridad sin depender del documento del cliente. R5 debe registrar en el historial los IDs, estados, recuentos, hashes, versiones, revisiones y digests necesarios para auditar la ejecucion; la comprobacion semantica de las 34 entradas y los codigos se mantiene separada como evidencia manual externa, sin copiar contenido del cliente al repositorio. + +## Causas raiz + +### C1. La identidad remota no representa el contrato de artefactos + +La clave `:ocr-v1:` identifica motor, paginas y configuracion, pero no diferencia el pipeline anterior del pipeline endurecido que persiste PNG de revision. Un trabajo `succeeded` antiguo puede ser valido como texto y, al mismo tiempo, incompatible con el contrato actual de evidencias. + +Ademas, OCR devuelve la fila existente para una clave idempotente, cualquiera que sea su estado, mientras el cliente RAG exige que todo acknowledgement tenga `status=queued`. El servidor y el cliente discrepan en un caso normal de idempotencia. + +### C2. Un percentil aislado bloquea paginas revisables + +La politica actual exige simultaneamente: + +- `nonWhitespaceCharacters >= 40`; +- `medianConfidence >= 0.8`; +- `p10Confidence >= 0.5`; +- `lowConfidenceLineRatio <= 0.2`. + +En v7, `p10Confidence` bloqueo tres paginas aunque los otros tres indicadores pasaron y las entradas eran legibles. El percentil bajo reacciona a botones, iconos y fragmentos visuales con baja confianza; no demuestra que toda la pagina sea inutilizable. + +### C3. El fallo pierde el diagnostico que permitiria revisarlo + +La composicion se detiene en la primera pagina bloqueada y propaga solo `OCR_QUALITY_BLOCKED`. El estado publico no identifica pagina, predicados incumplidos ni metricas. Tampoco publica una candidata revisable, aunque el resultado y las imagenes ya sean durables. + +## Contrato corregido + +### Identidad y compatibilidad `ocr-v2` + +1. RAG y OCR usan una unica constante por runtime con valor `ocr-v2` para `configVersion`. +2. La clave remota sigue el formato `::`; el cambio a `ocr-v2` crea un espacio de identidad nuevo y evita reutilizar trabajos `ocr-v1`. +3. `ocr-v2` significa, como minimo: resultado OCR validado, PNG privados persistidos durante el render, manifiesto de imagenes y endpoint de lectura sin rerender. +4. RAG rechaza cualquier acknowledgement, estado, resultado o imagen cuya identidad no coincida exactamente con hash, paginas y `ocr-v2`. +5. No se agrega fallback a `ocr-v1`. Un despliegue mixto falla cerrado y no crea contenido activo. + +La identidad canonica de una solicitud es el JSON canonico formado por `idempotencyKey`, `documentSha256`, `requestedPages` ordenadas y `configVersion`; su SHA-256 se denomina `requestIdentitySha256`. El contrato exige los siguientes campos: + +| Respuesta | Identidad obligatoria | +|---|---| +| Acknowledgement | `jobId`, `status`, `idempotencyKey`, `documentSha256`, `requestedPages`, `configVersion`, `requestIdentitySha256`, `createdAt`. | +| Estado | `jobId`, `status`, `idempotencyKey`, `documentSha256`, `requestedPages`, `configVersion`, `requestIdentitySha256`, contadores y error seguro. | +| Resultado | `schemaVersion`, `jobId`, `documentSha256`, `requestedPages`, `configVersion`, `requestIdentitySha256`, motor y paginas. | +| Imagen | cabeceras `X-Ocr-Job-Id`, `X-Ocr-Identity-Sha256`, `X-Ocr-Config-Version`, `X-Document-Sha256`, `X-Page-Number` y `X-Content-Sha256`. | + +RAG valida todos los campos contra la identidad persistida, no solo contra el `jobId`. Un campo ausente, un orden de paginas distinto o cualquier discrepancia falla cerrado con `OCR_RESPONSE_INTEGRITY_FAILED` antes de publicar artefactos locales. + +### Reenvio idempotente y estados terminales + +`POST /v1/jobs` puede devolver el trabajo existente con `status` igual a `queued`, `running`, `succeeded` o `failed`, siempre ligado a la misma identidad completa. + +RAG actua asi: + +| Estado devuelto | Conducta | +|---|---| +| `queued` o `running` | Persiste `remoteJobId`, consulta el estado y continua el polling acotado existente. | +| `succeeded` | Persiste `remoteJobId`, valida el resultado y transfiere las imagenes; no repite PaddleOCR. | +| `failed` | Persiste `remoteJobId` y propaga el codigo terminal seguro del trabajo; no lo presenta como corrupcion de integridad. | +| Identidad o estado invalido | Falla cerrado con `OCR_RESPONSE_INTEGRITY_FAILED`, distinto de errores de red, calidad o procesamiento. | + +El borrado remoto conserva el orden actual: solo despues de publicar y releer resultado, 25 comprobantes de imagen, candidata compuesta y estado local durable. Si RAG falla antes, el trabajo remoto puede reutilizarse hasta su TTL. + +### Recuperacion despues de `succeeded` + +El estado `succeeded` de `rag_ocr_jobs` confirma que el resultado OCR y las imagenes requeridas quedaron transferidos y releidos, pero no implica que la version haya terminado su composicion. Durante todo el flujo OCR la version permanece en `indexing`, de acuerdo con la maquina de estados canonica; `phase` expresa `ocr_queued`, `ocr_running` o `indexing` durante la finalizacion local. El reconciliador debe buscar versiones `indexing` con todos sus trabajos en `succeeded` y completar idempotentemente los pasos que falten: + +1. releer y validar manifiesto de fuente, paginas nativas, resultados OCR y manifiestos de imagenes; +2. crear o releer `quality-report.json` a partir de esos artefactos durables; +3. persistir diagnostico por pagina y decidir `failed` o candidata revisable; +4. crear o releer `candidate-pages.json` cuando no haya bloqueos; +5. persistir los hashes finales y transicionar mediante compare-and-set a `failed` o `review_required`. + +Cada publicacion es inmutable: si el fichero ya existe debe coincidir byte a byte. Una caida despues de cualquier paso repite la reconciliacion sin ejecutar PaddleOCR, sin duplicar filas y sin borrar el trabajo remoto. Si falta un artefacto local y el trabajo remoto sigue disponible, se reanuda solo la transferencia ausente; si ya no puede recuperarse, la version falla cerrada con un error de artefacto auditable. El borrado remoto queda prohibido hasta que la version local tenga informe, diagnostico y candidata o fallo durables. + +### Calidad bloqueante frente a advertencias + +La clasificacion sigue siendo cerrada para paginas vacias o inutilizables, pero `p10Confidence` deja de ser un veto aislado. `extractionMethod` describe de donde sale el texto y `qualityOutcome` decide la puerta de calidad: + +| Condicion | `extractionMethod` | `qualityOutcome` | Diagnostico | +|---|---|---|---| +| Pagina con texto nativo suficiente | `native` | `accepted` | Sin advertencias ni bloqueos OCR. | +| `inkCoverage < 0.015` y menos de 10 caracteres OCR | `blank` | `accepted` | Pagina vacia verificada. | +| Cualquier pagina no vacia con menos de 40 caracteres OCR | `ocr` | `blocked` | `INSUFFICIENT_TEXT`. | +| `medianConfidence < 0.8` | `ocr` | `blocked` | `LOW_MEDIAN_CONFIDENCE`. | +| `lowConfidenceLineRatio > 0.2` | `ocr` | `blocked` | `EXCESS_LOW_CONFIDENCE_LINES`. | +| Solo `p10Confidence < 0.5` | `ocr` | `warning` | `LOW_P10_CONFIDENCE`; pasa obligatoriamente a revision humana. | +| Todos los umbrales OCR superados | `ocr` | `accepted` | Sin advertencia de pagina. | + +No se cambian los otros tres umbrales sin nueva evidencia. La confianza nunca garantiza exactitud de codigos. Los tokens alfanumericos ambiguos siguen priorizados como riesgos de revision y nunca se autocorrigen. + +Una pagina puede acumular mas de un bloqueo. `blockingReasons` se ordena siempre como `INSUFFICIENT_TEXT`, `LOW_MEDIAN_CONFIDENCE`, `EXCESS_LOW_CONFIDENCE_LINES`; `primaryBlockingReason` es el primer elemento. `LOW_P10_CONFIDENCE` solo se incluye en `warnings` cuando no existe ningun bloqueo. Todas las paginas del original aparecen una vez en el informe, incluidas las nativas suficientes y las vacias verificadas. + +### Diagnostico durable + +La composicion clasifica todas las paginas antes de decidir el estado final y publica atomicamente `quality-report.json` junto al resultado OCR. El informe no contiene texto ni secretos: + +```json +{ + "schemaVersion": "1", + "versionId": "uuid", + "configVersion": "ocr-v2", + "sourceManifestSha256": "sha256", + "ocrResults": [ + {"documentId": "doc:...", "artifactSha256": "sha256", "resultSha256": "sha256"} + ], + "reviewImageManifests": [ + {"documentId": "doc:...", "manifestSha256": "sha256"} + ], + "pages": [ + { + "documentId": "doc:...", + "page": 2, + "extractionMethod": "ocr", + "qualityOutcome": "warning", + "warnings": ["LOW_P10_CONFIDENCE"], + "blockingReasons": [], + "primaryBlockingReason": null, + "metrics": { + "nonWhitespaceCharacters": 399, + "medianConfidence": 0.988, + "p10Confidence": 0.339, + "lowConfidenceLineRatio": 0.158, + "inkCoverage": 0.321 + } + } + ] +} +``` + +El informe se calcula sobre el manifiesto de fuente, cada sobre durable `ocr-result.json` y cada manifiesto durable de imagenes. Los hashes de entrada forman parte del propio informe; `qualityReportSha256` se persiste en PostgreSQL y se referencia desde `candidate-pages.json` o desde `error.details`. La lectura rechaza cualquier informe cuyo `versionId`, `configVersion` o cadena de hashes no coincida con los artefactos releidos. + +Si existe al menos una pagina bloqueada: + +- no se publica `candidate-pages.json` ni se pasa a revision; +- la version queda `failed` con codigo `OCR_QUALITY_BLOCKED`; +- `GET /ingestions/:versionId` devuelve `error.details.blockedPages` con documento, pagina y razones, sin texto OCR; +- `error.details` incluye el SHA-256 del informe para ligar el diagnostico de PostgreSQL con el artefacto durable; +- el informe durable permite auditar el fallo sin volver a ejecutar OCR. + +Si solo existen advertencias: + +- se publica la candidata; +- `candidate-pages.json` incluye `qualityReportSha256` y las advertencias por pagina; +- PostgreSQL conserva las advertencias por pagina en `quality_warnings`, separadas de los tokens de riesgo; +- la version llega a `review_required`; +- la API y la vista de revision muestran primero paginas advertidas y tokens de riesgo. + +La migracion `004_ocr_quality_diagnostics.sql` anade a `rag_document_pages` `quality_outcome text NULL`, `quality_warnings jsonb NOT NULL DEFAULT '[]'`, `blocking_reasons jsonb NOT NULL DEFAULT '[]'`, `primary_blocking_reason text NULL` y `quality_report_sha256 char(64) NULL`. `quality_outcome` solo admite `accepted`, `warning` o `blocked`; las filas nuevas `ocr-v2` deben informarlo. `extractionMethod` del informe se persiste en la columna existente `extraction_method`; `qualityOutcome`, `warnings`, `blockingReasons` y `primaryBlockingReason` se mapean respectivamente a las cuatro columnas nuevas. Si el resultado es `blocked`, `blocking_reasons` contiene todas las razones en el orden canonico definido arriba y `primary_blocking_reason` coincide con la primera; si no es bloqueante, ambos quedan vacios. `blocked_reason` se conserva solo para compatibilidad historica y no es la fuente canonica de filas `ocr-v2`. + +La migracion no reinterpreta filas antiguas ni altera estados existentes. `error_detail` conserva compatibilidad como texto para errores historicos; solo `OCR_QUALITY_BLOCKED` nuevo persiste un JSON canonico acotado que el repositorio valida antes de exponer como `error.details`. El diagnostico por pagina y `quality_report_sha256` se confirman en PostgreSQL antes de marcar la version como `failed` o `review_required`. + +### Revision de FacturaTech + +Llegar a `review_required` no equivale a PASS. Antes de aprobar v8 o posterior: + +1. verificar las 25 imagenes y sus hashes; +2. confirmar las 34 entradas contra el documento manual; +3. corregir explicitamente `FATo7 -> FAT07`, `NSAvo6 -> NSAV06` y `DSAUo4 -> DSAU04` si reaparecen; +4. comprobar exactamente `CBG04a`, `FAT07`, `DSAU08` y `NSAV06` en el texto revisado; +5. presentar diferencias, hashes y version al usuario; +6. esperar una autorizacion separada para aprobar/indexar y otra confirmacion si se desea activar. + +## Unidades de implementacion + +Las unidades se ejecutan en orden. Cada una incluye pruebas y documentacion propia. R1-R4 no despliegan ningun servicio, no ejecutan pruebas productivas y no crean otra candidata. R5 es la unica unidad autorizada para desplegar y probar conjuntamente RAG y OCR, siempre tras superar R4 y recibir aprobacion explicita. + +### R1. Versionar identidad y recuperar acknowledgements terminales + +- [x] Centralizar `ocr-v2` en cliente RAG y servicio OCR y actualizar payload, acknowledgement, resultado, huella de procesamiento e idempotency key. +- [x] Aceptar los cuatro estados conocidos en el acknowledgement sin debilitar las validaciones de identidad. +- [x] Persistir `remoteJobId` antes de consultar o consumir un trabajo repetido. +- [x] Reutilizar `succeeded`, propagar `failed` y conservar polling para `queued/running`. +- [x] Incorporar y validar la identidad canonica completa en acknowledgement, estado, resultado e imagenes. +- [x] Recuperar una version local `indexing` cuyos trabajos ya esten `succeeded` hasta informe, diagnostico y candidata o fallo durables, sin repetir OCR. +- [x] Cubrir duplicados en los cuatro estados y demostrar que un trabajo `ocr-v1` nunca satisface una solicitud `ocr-v2`. + +### R2. Separar bloqueo y advertencia de calidad + +- [x] Sustituir la decision booleana por una clasificacion con `extractionMethod`, `qualityOutcome`, `warnings`, `blockingReasons` y `primaryBlockingReason`. +- [x] Convertir exclusivamente `p10Confidence < 0.5` en `LOW_P10_CONFIDENCE` no bloqueante. +- [x] Mantener como bloqueantes los demas umbrales y la deteccion de pagina vacia. +- [x] Clasificar todas las paginas antes de fallar y conservar orden estable. +- [x] Persistir todas las razones de bloqueo por pagina, una razon primaria determinista y las advertencias por separado. +- [x] Anadir regresiones con las metricas exactas de las paginas 2, 20 y 21 de v7, sin copiar contenido del cliente al repositorio. + +### R3. Persistir y exponer diagnostico seguro + +- [x] Publicar y releer `quality-report.json` con permisos `0600`, escritura atomica, identidad, hashes de los resultados OCR y hashes de los manifiestos de imagenes. +- [x] Crear y validar `004_ocr_quality_diagnostics.sql` sin modificar estados ni datos historicos. +- [x] Propagar detalles estructurados de paginas bloqueadas hasta `GET /ingestions/:versionId`. +- [x] Ligar manifiesto de fuente, resultados OCR, manifiestos de imagenes, informe, candidata y error mediante SHA-256; rechazar cualquier discrepancia de identidad. +- [x] Conservar `quality_outcome`, `quality_warnings`, todas las `blocking_reasons`, la razon primaria y `quality_report_sha256` por pagina en PostgreSQL. +- [x] Priorizar en revision paginas advertidas y tokens alfanumericos ambiguos. +- [x] Probar que diagnosticos y logs no incluyen texto OCR, rutas internas, tokens ni secretos. + +### R4. Validacion integral e independiente + +R4 se ejecuta exclusivamente en local. La validacion de `004_ocr_quality_diagnostics.sql` debe usar un contenedor PostgreSQL local aislado y reutilizable, sin instalar un servicio nativo permanente y sin cargar `.env.local`, `.env.easypanel.local`, `POSTGRES_URL` ni `DATABASE_URL` de EasyPanel. Antes de ejecutar SQL se debe confirmar que el destino es ese contenedor. El runner `npm run migrate:lifecycle` aplica todas las migraciones pendientes, no solo `004`; por eso la prueba debe controlar la secuencia `001-003`, conservar una fila historica representativa, aplicar `004` y verificar que la fila no cambia. El contenedor y su volumen se conservan para futuras pruebas, pero debe configurarse sin reinicio automatico, arrancarse solo durante la prueba y quedar detenido al terminar. Un contenedor detenido no mantiene PostgreSQL consumiendo CPU o RAM asignada; la imagen y los datos siguen ocupando disco. No se apaga ni limpia Docker globalmente y no se permite usar `docker system prune`. + +- [x] Ejecutar pruebas focalizadas Node y Python para R1-R3. +- [x] Ejecutar suites completas, `npm run check`, `npm run build`, `py_compile` y `git diff --check`. +- [x] Simular caida de RAG despues de que OCR llegue a `succeeded`; el reenvio debe reutilizar el mismo trabajo y completar la transferencia. +- [x] Simular caidas despues de resultado, imagenes, informe y diagnostico; el reconciliador debe completar la finalizacion local exactamente una vez. +- [x] Simular un trabajo remoto `failed`; el estado debe conservar su codigo y no convertirse en error de integridad. +- [x] Validar con un fixture autocontenido de 25 paginas advertencias p10, 25 PNG y composicion revisable. +- [x] Validar por separado una pagina realmente bloqueada y comprobar informe, detalles API y ausencia de candidata. +- [x] Verificar la cadena de hashes completa y usar evidencia autocontenida para todas las pruebas automatizadas; etiquetar por separado cualquier comprobacion manual externa. +- [x] Encargar una revision independiente que no corrija silenciosamente defectos. +- [x] Validar `004` en el contenedor PostgreSQL local reutilizable: destino comprobado, historial `001-003`, fila historica intacta, columnas y restriccion nuevas verificadas y contenedor detenido al terminar. + +### R5. Entrega conjunta `0.2.1` y nueva aceptacion + +- [ ] Antes de aplicar migraciones en produccion, identificar explicitamente base y entorno, revisar `rag_schema_migrations` y confirmar que solo estan pendientes las migraciones esperadas. La validacion local de R4 no autoriza esta operacion. +- [x] Crear una unica fuente versionada en la raiz, `VERSION`, con `0.2.1`; RAG, OCR, pruebas y proceso de build deben consumirla y fallar si aparece otra version declarada. +- [ ] Publicar RAG/OCR desde el mismo commit. `BUILD_REVISION` se carga manualmente con ese commit mientras no exista automatizacion aprobada; los digests de imagen siguen siendo la identidad exacta autoritativa. +- [ ] Ejecutar una unica operacion conjunta de despliegue con `OCR_INGEST_ENABLED=true`. EasyPanel puede reconstruir cada servicio por separado; no se ejecuta la aceptacion intencional de FacturaTech y no se considera desplegada la entrega hasta que ambos terminen. No introducir una desactivacion preventiva ni un despliegue por etapas sin un fallo real que lo justifique. +- [ ] Cuando ambos despliegues terminen, comprobar que health de RAG y OCR devuelve `version=0.2.1`, la misma revision esperada y que las etiquetas OCI de ambas imagenes coinciden; una diferencia bloquea la aceptacion. +- [ ] Verificar OCR live/ready, cola vacia, almacenamiento, RAG, PostgreSQL, Qdrant y reconciliador como una unica bateria posterior al despliegue conjunto. +- [ ] Cerrar o conservar auditadamente v7 sin aprobarla, indexarla ni activarla. +- [ ] Crear una candidata nueva con `activate=false` y supervisarla hasta `review_required` o fallo. +- [ ] Ejecutar la revision de FacturaTech definida en este anexo. +- [ ] Registrar evidencia autocontenida de IDs, estados, recuentos, hashes, versiones, revisiones y digests; marcar como manual externa la comparacion con el documento del cliente. +- [ ] Solicitar autorizacion explicita antes de aprobar, indexar o activar. + +## Criterios de salida + +La correccion solo obtiene PASS cuando se cumplen todos estos puntos: + +1. Ningun trabajo `ocr-v1` colisiona ni se reutiliza como `ocr-v2`. +2. Un reenvio de cada estado terminal o no terminal tiene conducta determinista y cubierta por pruebas. +3. Las metricas de v7 para paginas 2, 20 y 21 producen advertencias, no bloqueo. +4. Una pagina realmente inutilizable sigue fallando cerrada con diagnostico por pagina. +5. Todas las razones de bloqueo y advertencias quedan persistidas por pagina con una razon primaria determinista. +6. La cadena de hashes liga manifiesto de fuente, resultados OCR, manifiestos de imagenes, informe de calidad y candidata o error. +7. Una caida local posterior a `succeeded` se recupera sin repetir OCR ni perder la finalizacion de la version. +8. Las 25 imagenes y todos los artefactos conservan identidad, hash, permisos y retencion correctos. +9. RAG y OCR obtienen `0.2.1` de `VERSION`, exponen metadatos coincidentes y se despliegan y prueban como una unica unidad. +10. La nueva candidata contiene las 34 entradas y alcanza `review_required` sin indexacion ni activacion. +11. Los cuatro codigos criticos son exactos despues de revision humana externa. +12. La version activa anterior permanece disponible hasta una activacion autorizada y conserva rollback verificable. + +## Rollback + +- RAG y OCR `0.2.1` forman una unidad de compatibilidad; no se revierte solo uno de los dos. +- Ante fallo, mantener detenidas las nuevas ingestas OCR, volver ambos servicios al commit previamente validado de la entrega `0.2.0` dentro de la misma operacion y verificar conjuntamente salud, versiones y digests antes de reabrir trafico. +- No borrar versiones fallidas ni artefactos antes de capturar la evidencia necesaria. +- La version activa actual no se modifica durante implementacion, validacion, despliegue ni revision. diff --git a/docs/ESTRUCTURA_CARPETAS_EMPRESA.md b/docs/ESTRUCTURA_CARPETAS_EMPRESA.md new file mode 100644 index 0000000..fd8e164 --- /dev/null +++ b/docs/ESTRUCTURA_CARPETAS_EMPRESA.md @@ -0,0 +1,16 @@ +# Estructura de carpetas de empresa + +## Regla general + +La ubicacion se decide por la naturaleza del trabajo: + +- Las instalaciones operativas usadas por otros workspaces van en `Empresa/IA/herramientas/`. +- El codigo fuente, forks, laboratorios y desarrollos propios van en `Empresa/Desarrollo/IA/`. +- La documentacion transversal vive junto al area funcional que la utiliza. +- No se prueba directamente sobre una instalacion operativa compartida. + +## Criterio para RAG + +RAG es un modulo de desarrollo propio. Su codigo y documentacion viven en este worktree; los servicios desplegados se tratan como infraestructura operativa y nunca como laboratorio de pruebas. + +Antes de crear, clonar, mover o instalar un proyecto dentro de `Empresa`, confirma con el usuario si existe riesgo operativo o ambiguedad de ubicacion. diff --git a/docs/HISTORIAL_SESIONES.md b/docs/HISTORIAL_SESIONES.md index 8c28731..b48eb13 100644 --- a/docs/HISTORIAL_SESIONES.md +++ b/docs/HISTORIAL_SESIONES.md @@ -2,14 +2,120 @@ **Proyecto:** Workspace de tools IA para empresas **Modulo:** RAG -**Ultima actualizacion:** 2026-09-21 -**Ultima modificacion por:** Agente RAG 2 +**Ultima actualizacion:** 2026-09-23 +**Ultima modificacion por:** Agente RAG 3 **Estado:** Activo --- ## Registro de sesion +### 2026-09-23 - Agente RAG 3 - Validacion local integral de R5 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-terra · **Session:** no disponible tras compactacion +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Retomada la validacion local R5 tras liberar capacidad Docker. La imagen `rag-ocr:version-check` construyo dependencias y modelos Paddle correctamente con `OCR_VERSION=0.2.1` y `BUILD_REVISION=local-version-check`. Se actualizaron dos aserciones Python para que contrasten la revision efectiva del entorno y para exigir `VERSION` dentro del contexto Docker OCR. +**Validation:** Imagen OCR con etiquetas OCI `0.2.1 local-version-check`; contenedor real `healthy`; `/health/live` y `/health/ready` confirmaron modelo cargado, worker/sweeper operativos, cola vacia y la misma version/revision. Python en imagen: 26 pruebas correctas. Un build con `OCR_VERSION=0.2.2` fallo antes de instalar dependencias. Node: `npm test` 116/116, `npm run check`, `npm run build` y contrato de despliegue 3/3 correctos. No se desplego, no se accedio a EasyPanel y no se modificaron datos productivos. +**Rollback:** Revertir las pruebas y documentacion de esta entrada junto con la entrega `0.2.1`; no existe cambio remoto que revertir. +**Files:** `ocr-service/tests/test_api.py`, `ocr-service/tests/test_render.py`, `docs/PENDIENTES_RAG.md`, `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`, `docs/OPERATIVA.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 3 - Fuente unica de version para R5 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-terra · **Session:** no disponible tras compactacion +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Creado `VERSION` con `0.2.1` como fuente de version comun. RAG y OCR lo leen al iniciar y rechazan `RAG_VERSION` u `OCR_VERSION` divergentes; OpenAPI, health y metadatos de paquete consumen ese valor. Ambos Dockerfiles exigen el build argument coincidente, publican las etiquetas OCI y el contexto OCR incluye `VERSION` expresamente. +**Validation:** `npm test` (116 pruebas), `npm run check`, `npm run build`, prueba de contrato de despliegue, `py_compile`, rechazo Python de version OCR divergente, `git diff --check` y build de imagen RAG con etiquetas `0.2.1 local-version-check` correctos. La imagen OCR llego a validar `VERSION`, pero no completo la instalacion de dependencias: Docker agoto el espacio de su particion raiz (`1,1 GB` libres) en la inicializacion de PaddleX. No se limpio Docker ni se modifico infraestructura. +**Rollback:** Revertir los ficheros de version y Dockerfiles de esta entrada como una unidad; no desplegar solo RAG u OCR. +**Files:** `VERSION`, `src/config/{version,env}.ts`, `src/api/openapi.ts`, `ocr-service/app/{version,main}.py`, `Dockerfile`, `ocr-service/Dockerfile`, `ocr-service/Dockerfile.dockerignore`, `package.json`, `package-lock.json`, pruebas de contrato y API OCR. + +--- + +### 2026-09-22 - Agente RAG 3 - Continuidad ejecutable de R5 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f358ccfe9ffeEzEVNuKo93aHyH` +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Persistida en el backlog una checklist ejecutable de R5 para retomarla en orden tras la compactacion: fuente unica `VERSION`, validacion local, commit/push comun, inspeccion productiva previa, despliegue conjunto con OCR habilitado, bateria de salud, nueva candidata no activa, revision humana y autorizaciones separadas. +**Validation:** La checklist conserva `OCR_INGEST_ENABLED=true`, reserva el modo por etapas para fallos reales y mantiene bloqueadas aprobacion, indexacion y activacion hasta autorizacion explicita. +**Rollback:** Eliminar solo la checklist de continuidad y esta entrada; no se modificaron codigo, servicios ni produccion. +**Files:** `docs/PENDIENTES_RAG.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 3 - Simplificacion operativa de R5 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f358ccfe9ffeEzEVNuKo93aHyH` +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Sustituidas las referencias operativas fijas a `0.2.0` por la version declarada en `VERSION`. Por decision del usuario, el despliegue conjunto mantiene `OCR_INGEST_ENABLED=true` y no introduce una desactivacion preventiva ni una ventana sin ingestas. El valor `false` y el despliegue por etapas quedan reservados para diagnosticar o recuperar un fallo real cuando la evidencia justifique aislar servicios. +**Validation:** Contrato, backlog y operativa expresan la misma regla; la aceptacion intencional de FacturaTech solo comienza cuando ambos despliegues terminan y coinciden en version y revision. +**Rollback:** Restaurar la regla preventiva anterior solo mediante una nueva decision explicita; no se modificaron codigo, servicios ni produccion. +**Files:** `docs/OPERATIVA.md`, `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`, `docs/PENDIENTES_RAG.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 3 - Validacion local de R4 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-terra · **Session:** `ses_f358ccfe9ffeEzEVNuKo93aHyH` +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Validada la migracion `004_ocr_quality_diagnostics.sql` exclusivamente en PostgreSQL Docker local. Se aplicaron y registraron `001-003`, se conservo una fila historica representativa y el runner `npm run migrate:lifecycle` aplico `004`. No se cargaron ficheros de entorno ni URLs de EasyPanel. +**Validation:** Historial `001-004`; cinco columnas nuevas y su restriccion comprobadas; la fila historica mantiene sus valores previos y recibe valores por defecto; segunda ejecucion idempotente del runner; `npx tsx --test tests/catalog/migration-004.test.ts` PASS. El contenedor `rag-r4-postgres` tiene reinicio `no`, volumen `rag-r4-postgres-data` y quedo detenido. +**Rollback:** No se modificaron datos ni infraestructura remota. Para repetir R4, iniciar solo `rag-r4-postgres`; no eliminar el volumen salvo que se requiera una base nueva y exista aprobacion explicita. +**Files:** `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`, `docs/PENDIENTES_RAG.md`, `docs/CONTEXTO_PROYECTO.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 3 - Consolidacion documental RAG +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-terra · **Session:** `ses_f358ccfe9ffeEzEVNuKo93aHyH` +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Auditadas y consolidadas las referencias documentales entre `IA/docs/` y `RAG/docs/`. Se traslado el documento de acceso Git exclusivo del repositorio y las incidencias RAG desde el registro global. El backlog global conserva solo visibilidad y enlace al backlog canonico. Se actualizo el contexto con el agente actual. +**Validation:** `git diff --check` sin errores; confirmado el traslado del documento Git, cuatro incidencias RAG en su registro local y su ausencia del registro global. +**Rollback:** Devolver `info-git-por-correo.md` y las cuatro entradas a `IA/docs/`; restaurar los enlaces globales anteriores. +**Files:** `docs/info-git-por-correo.md`, `docs/REGISTRO_SITUACIONES.md`, `docs/INDICE_DOCUMENTACION.md`, `docs/CONTEXTO_PROYECTO.md`, `docs/HISTORIAL_SESIONES.md`, `../docs/PENDIENTES_GENERALES.md`, `../docs/REGISTRO_SITUACIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 3 - Migracion documental a plantilla 1.2.7 +**Agent:** Agente RAG 3 · **Model:** openai/gpt-5.6-terra · **Session:** `ses_f358ccfe9ffeEzEVNuKo93aHyH` +**Role:** Desarrollo, mantenimiento y continuidad del servicio RAG y su integracion con el servicio OCR reutilizable. +**Work:** Migrado el workspace RAG no versionado a la plantilla documental `1.2.7` con aprobacion explicita. Se preservaron las instrucciones concretas del modulo en `AGENTS.md`: worktree exclusivo `RAG/`, proyecto Engram canonico `rag-service`, operaciones Engram con proyecto explicito, contrato y backlog OCR, consulta de operativa y aprobacion previa. Se integraron las instrucciones estandar de OpenCode y se anadieron los documentos de plantilla ausentes con contexto real de RAG. No se modificaron el contrato OCR, backlog, operativa, codigo, migraciones, datos, infraestructura ni produccion. +**Validation:** `mem_current_project` resuelve `rag-service`; `git diff --check` sin errores; `TEMPLATE_VERSION` creado con `1.2.7`; `opencode.json` contiene las tres instrucciones estandar sin cargar el indice ni el historial por turno. +**Rollback:** Revertir solo los archivos de plantilla creados o actualizados en esta entrada y eliminar `TEMPLATE_VERSION`; no existen efectos de runtime. +**Files:** `AGENTS.md`, `opencode.json`, `TEMPLATE_VERSION`, `docs/{ACTUALIZACION_PLANTILLA,CONTEXTO_PROYECTO,ESTRUCTURA_CARPETAS_EMPRESA,INDICE_DOCUMENTACION,LEARNED_SKILLS,REGISTRO_SITUACIONES,readme,sesion_actual_opencode}.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 2 - Enrutamiento canonico de Engram +**Agent:** Agente RAG 2 · **Model:** openai/gpt-5.6-terra · **Session:** `ses_29bdbd003ffeLrLjUlFgnp08Y7` +**Work:** Confirmado que `RAG/` es un worktree Git anidado y que la sesion se habia abierto desde `IA/`; Engram resolvia por ello el proyecto externo `desarrollo`. Se declaro `rag-service` como unico destino canonico de memoria RAG. Se anadieron reglas de enrutamiento en el workspace padre y una configuracion e instrucciones locales de OpenCode en `RAG/`: las sesiones RAG deben iniciarse desde este worktree, verificar la deteccion y pasar explicitamente `project: "rag-service"` en toda operacion de memoria. Las busquedas entre proyectos quedan limitadas a recuperaciones o consolidaciones autorizadas. +**Validation:** `git rev-parse --show-toplevel` desde `RAG/` devuelve su propia raiz. La consulta de sesiones de OpenCode desde `RAG/` no devolvio filas, lo que confirma que debe iniciarse una sesion nueva en ese directorio para validar la deteccion automatica. No se modificaron codigo, migraciones, datos, infraestructura ni produccion. +**Rollback:** Revertir solo `AGENTS.md`, `opencode.json` y las entradas de historial de este cambio; no existen efectos sobre R4 ni servicios. +**Files:** `../AGENTS.md`, `AGENTS.md`, `opencode.json`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-22 - Agente RAG 2 - Implementacion local R1-R4 de OCR `0.2.1` +**Agent:** Agente RAG 2 · **Model:** openai/gpt-5.6-terra · **Session:** `ses_29bdbd003ffeLrLjUlFgnp08Y7` +**Work:** Implementados R1-R3 y la evidencia local disponible de R4 sin desplegar: identidad canonica `ocr-v2` para acknowledgement, estado, resultado e imagenes; recuperacion idempotente de trabajo remoto ya completado; diagnosticos de calidad por pagina, informe privado inmutable y migracion `004`; advertencia `LOW_P10_CONFIDENCE`; y transferencia de revision ligada por hashes. La revision independiente detecto y se corrigieron recuperacion incompleta tras diagnosticos, fallo cerrado para integridad/404, validacion de estado del acknowledgement, propagacion de codigo terminal y la identidad semantica del informe. Tambien se impidio que el sweeper marque como huerfana una version con OCR remoto `queued` o `running`. No se aplico la migracion, no se desplego, no se creo candidata y no se modifico FacturaTech ni la version activa. +**Validation:** Node 112/112, Python OCR 25/25, `npm run check`, `npm run build`, `py_compile` y `git diff --check` limpios. La revision especializada 4R fue rechazada por `opencode_review_transport_binding_invalid`; una revision independiente general read-only se ejecuto en su lugar y sus hallazgos criticos fueron corregidos y revalidados. El fixture autocontenido de 25 paginas y las recuperaciones por frontera se validaron despues. Pendiente en R4: validar `004` en un contenedor PostgreSQL local aislado y reutilizable; queda prohibido usar la base productiva de EasyPanel. El contenedor conservara sus datos, no tendra reinicio automatico y debera quedar detenido al finalizar cada prueba. +**Rollback:** Revertir los cambios locales R1-R4 y no aplicar `004`; R5 sigue siendo la unica operacion que puede desplegar conjuntamente RAG y OCR. +**Files:** `src/modules/ocr/*`, `src/modules/catalog/{repository,reconciler}.ts`, `src/modules/ingest/service.ts`, `src/app.ts`, `ocr-service/app/*`, `migrations/004_ocr_quality_diagnostics.sql`, pruebas OCR/catalogo y documentacion de este contrato. + +--- + +### 2026-09-22 - Agente RAG 2 - Cierre documental de auditoria `0.2.1` +**Agent:** Agente RAG 2 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_29bdbd003ffeLrLjUlFgnp08Y7` +**Work:** Corregidos contrato y backlog tras la auditoria independiente que retuvo la aprobacion por seis huecos. El contrato ahora exige recuperacion idempotente de versiones locales `indexing` despues de trabajos OCR `succeeded`; identidad canonica exacta en acknowledgement, estado, resultado e imagenes; persistencia por pagina de advertencias, todas las razones de bloqueo y una razon primaria; cadena de hashes entre manifiesto de fuente, resultados OCR, manifiestos de imagenes, informe y candidata o error; evidencia automatizada autocontenida con la comprobacion del documento del cliente etiquetada como manual externa; y sustitucion explicita de la antigua regla de deploy por una unica operacion conjunta RAG/OCR en R5. Tambien se fijo `VERSION` como futura fuente versionada unica para `0.2.1`, validacion de igualdad en health y etiquetas OCI, `BUILD_REVISION` manual contrastada con el commit y digests como identidad exacta. R1-R4 no despliegan produccion. No se modifico codigo ni se autorizo implementacion o despliegue. +**Validation:** `git diff --check` limpio. La reauditoria read-only detecto primero contradicciones de estado y vocabulario, se corrigieron, y el segundo pase emitio PASS sobre los seis hallazgos; sus dos observaciones LOW de terminologia tambien quedaron resueltas. Tras la compactacion se verificaron contrato, backlog e historial y se corrigieron el estado de Fase 3C y las fechas documentales que aun reflejaban la auditoria como pendiente. La implementacion sigue bloqueada hasta aprobacion explicita del usuario. +**Rollback:** Revertir solo esta correccion documental; no se modificaron servicios, datos, candidatas ni la version activa. +**Files:** `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`, `docs/PENDIENTES_RAG.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-21 - Agente RAG 2 - Diseno correctivo tras aceptacion v6-v7 +**Agent:** Agente RAG 2 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_29bdbd003ffeLrLjUlFgnp08Y7` +**Work:** Ejecutada, con aprobacion del usuario, la aceptacion no activa de FacturaTech sobre RAG/OCR `0.2.0`. v6 `b9281626-088a-46e5-9f66-b4a3bbe2159f` fallo en `0/25` por reutilizacion de una identidad remota `ocr-v1` terminal incompatible. Tras confirmar por SSH que el sweeper habia eliminado automaticamente fila y artefactos antiguos, v7 `c9756e59-f91d-4b57-8f31-e7b5b2b09558` creo un trabajo nuevo y completo `25/25` en un intento, pero la candidata quedo `failed` por `OCR_QUALITY_BLOCKED`: solo las paginas 2, 20 y 21 incumplieron `p10Confidence >= 0.5`, con medianas altas, ratios bajos y contenido legible. Verificadas 25 imagenes unicas, firmas y hashes correctos, `12.556.918` bytes, original, paginas nativas y resultado OCR durables. Las 34 entradas estan presentes; la revision debe corregir `FATo7`, `NSAvo6` y `DSAUo4`. La version activa `3fc78163-9cfb-4979-985c-1520a63327b0` permanece intacta. Por indicacion del usuario no se inicio una reparacion directa: se anadio al contrato un anexo correctivo para `0.2.1` con causas, decisiones, invariantes, tareas R1-R5, criterios de salida y rollback; el backlog queda bloqueado hasta su aprobacion explicita. +**Validation:** Produccion: v7 OCR `25/25`, trabajo remoto `succeeded`, 25/25 PNG con hash y firma validos, 34/34 mensajes con recuerdo por tokens >= `88,9 %`, causas y soluciones >= `100 %` considerando continuacion de pagina; fuente activa sin cambios. Documentacion: `git diff --check` limpio y revision final de contrato, backlog e historial completada. No se modifico codigo, no se aprobo, indexo ni activo contenido. +**Rollback:** Revertir solo el anexo correctivo, el estado de Fase 3C y esta entrada documental. Las versiones productivas fallidas se conservan como evidencia; no se ejecutaron cambios de codigo ni despliegues. +**Files:** `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`, `docs/PENDIENTES_RAG.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + ### 2026-09-21 - Agente RAG 2 - Preparacion de la version 0.2.0 **Agent:** Agente RAG 2 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_29bdbd003ffeLrLjUlFgnp08Y7` **Work:** Preparada la entrega conjunta RAG/OCR `0.2.0` tras superar D1-D4 la validacion independiente. Ambos servicios muestran version y revision de commit en sus respuestas de salud y etiquetas de imagen. Actualizados el README privado del OCR, la ayuda del RAG, el contrato, la operativa y el backlog. Por decision del usuario, el proximo despliegue instala RAG y OCR juntos con OCR activado; el despliegue separado queda solo como diagnostico si aparece un fallo ambiguo. Produccion continua en `0.1.0` hasta el despliegue manual en EasyPanel. diff --git a/docs/INDICE_DOCUMENTACION.md b/docs/INDICE_DOCUMENTACION.md new file mode 100644 index 0000000..dece52c --- /dev/null +++ b/docs/INDICE_DOCUMENTACION.md @@ -0,0 +1,48 @@ +# Indice de documentacion - RAG Service + +**Ultima actualizacion:** 2026-09-22 +**Ultima modificacion por:** Agente RAG 3 +**Version:** 1.0 + +Este indice es un mapa de consulta. Leelo antes de crear, buscar o actualizar documentacion; no es necesario cargarlo en cada sesion. + +## Documentacion de continuidad + +| Documento | Proposito | +| --- | --- | +| `CONTEXTO_PROYECTO.md` | Ficha rapida y estado del modulo. | +| `HISTORIAL_SESIONES.md` | Registro de sesiones, cambios y decisiones. | +| `PENDIENTES_RAG.md` | Backlog priorizado y orden de trabajo. | +| `CONTRATO_CICLO_VIDA_Y_OCR.md` | Contrato canonico de lifecycle y OCR. | +| `OPERATIVA.md` | Hechos operativos sin secretos. | +| `REGISTRO_SITUACIONES.md` | Incidencias y evidencia concreta. | + +## Documentacion tecnica + +| Documento | Proposito | +| --- | --- | +| `API_RAG.md` | Contratos y uso de la API RAG. | +| `INGESTA.md` | Flujo de ingesta. | +| `PROCESADO.md` | Procesamiento documental. | +| `SALIDA.md` | Generacion y salida de respuestas. | +| `SISTEMA_RAG_BASE.md` | Arquitectura base del servicio. | +| `STACK_TECNICO_V1.md` | Tecnologias y decisiones de stack. | +| `PLAYGROUND.md` | Interfaz de prueba y revision. | +| `LOGS_EVALUACION.md` | Registro de evaluacion. | +| `info-git-por-correo.md` | Acceso y flujo de Git del repositorio RAG. | + +## Documentacion de plantilla + +| Documento | Proposito | +| --- | --- | +| `readme.md` | Protocolo de agentes. | +| `sesion_actual_opencode.md` | Recuperacion de la sesion OpenCode. | +| `LEARNED_SKILLS.md` | Aprendizajes reutilizables validados. | +| `ESTRUCTURA_CARPETAS_EMPRESA.md` | Norma de ubicacion dentro de Empresa. | +| `ACTUALIZACION_PLANTILLA.md` | Procedimiento de actualizacion de plantilla. | +| `TEMPLATE_VERSION` | Version aplicada de la plantilla. | + +## Regla de mantenimiento + +- Todo documento nuevo, movido o eliminado en `docs/` requiere actualizar este indice en la misma intervencion. +- Los documentos de contrato, backlog, operativa e historial son fuentes canonicas y no deben duplicarse. diff --git a/docs/LEARNED_SKILLS.md b/docs/LEARNED_SKILLS.md new file mode 100644 index 0000000..2a55781 --- /dev/null +++ b/docs/LEARNED_SKILLS.md @@ -0,0 +1,16 @@ +# Habilidades Aprendidas (Skills) + +**Ultima actualizacion:** 2026-09-22 +**Version:** 1.0 + +Este documento conserva procedimientos reutilizables validados para este workspace. No sustituye `REGISTRO_SITUACIONES.md` para incidencias ni `OPERATIVA.md` para hechos de entorno. + +## Criterio de uso + +- Registra soluciones, decisiones operativas y acciones posteriores a actualizaciones que puedan reutilizarse. +- Distingue entre lo automatico tras actualizar y lo que requiere accion explicita. +- Registra evidencia suficiente para ejecutar el procedimiento sin reinterpretar un changelog. + +## Aprendizajes validados + +- Las sesiones RAG deben iniciarse desde `IA/RAG` y usar `rag-service` explicitamente en Engram para evitar deriva al workspace padre. diff --git a/docs/OPERATIVA.md b/docs/OPERATIVA.md index 7b50ada..ce104f9 100644 --- a/docs/OPERATIVA.md +++ b/docs/OPERATIVA.md @@ -1,8 +1,8 @@ # Operativa del servicio RAG **Modulo:** RAG -**Ultima actualizacion:** 2026-09-21 -**Version:** 1.3 +**Ultima actualizacion:** 2026-09-22 +**Version:** 1.4 --- @@ -24,18 +24,19 @@ Este documento registra los hechos operativos del servicio RAG: la configuracion - La candidata OCR heredada de FacturaTech v4 fue cerrada como `failed` el 2026-09-20 mediante recuperacion administrativa auditada tras confirmar `OCR_ARTIFACT_UNAVAILABLE`. No fue indexada ni activada; la fuente conserva su version activa y no tiene candidatas OCR bloqueantes. - Las imagenes en ejecucion tienen digest, pero ambas etiquetas OCI `org.opencontainers.image.revision` valen `unknown`: EasyPanel no esta pasando `BUILD_REVISION` durante el build. La identidad de revision verificable sigue pendiente. -## Proxima version preparada +## Version de despliegue -- RAG y OCR `0.2.0` estan preparados localmente; produccion sigue en `0.1.0` hasta ejecutar el despliegue. +- La version correcta de cada despliegue es la declarada en el fichero raiz `VERSION` del commit que se publica (para la entrega actual, `0.2.1`). No reutilizar numeros escritos en entradas historicas de este documento. - Ambas imagenes publican `org.opencontainers.image.version` y `org.opencontainers.image.revision`; las respuestas de salud muestran `version` y `revision`. -- EasyPanel debe construir ambas imagenes con `RAG_VERSION=0.2.0` u `OCR_VERSION=0.2.0` y `BUILD_REVISION=`. -- Por decision del usuario, RAG y OCR se despliegan juntos con `OCR_INGEST_ENABLED=true`. El despliegue por etapas queda reservado para diagnosticar un fallo cuyo origen no sea claro. +- EasyPanel debe construir ambas imagenes con la version de `VERSION` y con `BUILD_REVISION=`. +- Por decision del usuario, RAG y OCR se despliegan juntos con `OCR_INGEST_ENABLED=true`; no se desactiva preventivamente ni se introduce una ventana sin ingestas por protocolo. +- `OCR_INGEST_ENABLED=false` y el despliegue por etapas se reservan para responder a un fallo real cuando sus sintomas y la evidencia indiquen que aislar un servicio ayudara al diagnostico o a la recuperacion. ## Lista rapida en EasyPanel 1. Rotar las credenciales expuestas (ver seccion siguiente); la rotacion sigue pendiente. 2. Comprobar que las variables coinciden con la tabla de configuracion actual. -3. Mantener `OCR_INGEST_ENABLED=true` para el flujo OCR ya habilitado. +3. Mantener `OCR_INGEST_ENABLED=true`; cambiarlo a `false` solo ante un fallo real que justifique aislar el flujo OCR. 4. Pulsar `Deploy` en EasyPanel despues de cada cambio principal. 5. Verificar `GET /health` del RAG y las rutas internas de salud del OCR tras cada deploy. 6. Registrar el digest de cada imagen y confirmar que `version` y `revision` coinciden con la entrega desplegada. @@ -133,19 +134,19 @@ El OCR es un servicio privado e independiente. El RAG solo lo llama si `OCR_INGE ### Verificacion posterior al despliegue (cuando cambian ambos servicios) -1. Publicar el codigo en Git `main` y desplegar juntos OCR y RAG `0.2.0` desde EasyPanel con `OCR_INGEST_ENABLED=true`. +1. Publicar el codigo en Git `main` y desplegar juntos OCR y RAG con la version indicada en `VERSION` y `OCR_INGEST_ENABLED=true`. 2. Verificar el OCR internamente: `GET /health/live` y `GET /health/ready` deben responder con el servicio preparado. 3. Verificar `GET /health` del RAG: PostgreSQL, Qdrant y reconciliador deben estar correctos. 4. Comprobar una ruta OCR protegida sin token. Un `401 Lifecycle admin token is required` confirma que OCR esta habilitado y que la autenticacion administrativa permanece protegida; no es un error que requiera correccion. 5. Contrastar `OCR_INGEST_ENABLED=true` en el entorno persistido de EasyPanel y en el proceso del contenedor si la UI no coincide con el comportamiento efectivo. 6. Consultar con token administrativo una version inexistente en revision y recuperacion: ambas deben devolver un error estructurado `404 OCR_CANDIDATE_NOT_FOUND` sin modificar datos. 7. Confirmar que `rag_schema_migrations` contiene `003_ocr_recovery_audit.sql` y que existe `rag_ocr_recovery_audit`. -8. Revisar los digests y confirmar que las etiquetas OCI y las respuestas de salud muestran `0.2.0` y el commit desplegado, no `unknown`. +8. Revisar los digests y confirmar que las etiquetas OCI y las respuestas de salud muestran `0.2.1` y el commit desplegado, no `unknown`. 9. Tras completar estas comprobaciones, crear una candidata FacturaTech no activa. No aprobar, indexar ni activar hasta presentar la evidencia al usuario y recibir autorizacion explicita. ### Rollback de emergencia -- Ante cualquier fallo: poner `OCR_INGEST_ENABLED=false`, pulsar `Deploy` en el RAG y conservar la version activa actual del corpus. +- Ante un fallo cuyos sintomas indiquen que conviene aislar OCR: poner `OCR_INGEST_ENABLED=false`, desplegar el RAG y conservar la version activa actual del corpus. No aplicar este paso automaticamente a fallos no relacionados. - El sistema es fail-closed: un fallo del OCR deja intacta la version activa anterior; no hay activacion parcial. - Las paginas con OCR no se activan solas; requieren revision humana obligatoria (estado `review_required`). - Desactivar el OCR no borra el corpus activo ni exige reingesta. diff --git a/docs/PENDIENTES_RAG.md b/docs/PENDIENTES_RAG.md index 8650a04..a00a317 100644 --- a/docs/PENDIENTES_RAG.md +++ b/docs/PENDIENTES_RAG.md @@ -1,6 +1,6 @@ # Pendientes priorizados del RAG -**Ultima actualizacion:** 2026-09-21 +**Ultima actualizacion:** 2026-09-23 **Responsable de la priorizacion:** Usuario **Estado:** Activo @@ -35,7 +35,7 @@ Esta secuencia tiene prioridad sobre la aceptacion productiva pendiente de Factu ### Fase 3. Hardening OCR y aceptacion de FacturaTech -**Estado:** D1-D4 implementados y validados de forma independiente; queda pendiente desplegar la version `0.2.0` y ejecutar la aceptacion productiva sin activar contenido automaticamente. +**Estado:** `0.2.0` desplegada y D1-D4 verificados en produccion. La aceptacion v6-v7 encontro dos defectos correctivos; queda bloqueada hasta entregar `0.2.1` segun el anexo del contrato. La candidata v5 `5f2317c6-7a8a-4e08-a614-f8189602ebb8` completo 25/25 paginas OCR, pero fallo despues cuando RAG solicito 25 imagenes de revision en paralelo. El rerender concurrente provoco un `SIGSEGV` de PDFium/FreeType, salida `139` y reinicio del contenedor; no hubo OOM. La candidata quedo fallida y la version activa no cambio. @@ -50,15 +50,51 @@ La candidata v5 `5f2317c6-7a8a-4e08-a614-f8189602ebb8` completo 25/25 paginas OC #### Fase 3B. Aceptacion productiva -1. Desplegar juntos OCR y RAG `0.2.0`, con OCR habilitado desde el inicio y sin activar contenido. -2. Crear una nueva candidata no activada para el documento. -3. Verificar la evidencia durable, las 25 imagenes y sus 34 entradas. -4. Comprobar `CBG04a`, `FAT07`, `DSAU08` y `NSAV06`. -5. Requerir aprobacion humana explicita antes de indexar o activar. +**Estado:** Ejecutada y no superada. v6 fallo por colision idempotente con un trabajo terminal `ocr-v1`. Tras la limpieza automatica, v7 proceso 25/25 paginas y persistio 25 PNG validos, pero un falso negativo de `p10Confidence` la cerro como `OCR_QUALITY_BLOCKED`. Las 34 entradas estan presentes; `FAT07`, `NSAV06` y `DSAU04` requieren correccion humana por confusion `0/o`. La version activa no cambio. + +1. Completado: desplegados juntos OCR y RAG `0.2.0` con revision verificable. +2. Completado: candidatas v6 y v7 creadas con `activate=false`; ninguna fue indexada ni activada. +3. Completado: evidencia durable de v7, 25 imagenes y 34 entradas verificadas. +4. No superado: la puerta de calidad impidio llegar a revision y dos de los cuatro codigos criticos no son exactos antes de correccion humana. +5. Completado: version activa anterior confirmada intacta. + +#### Fase 3C. Correccion post-aceptacion `0.2.1` + +**Estado:** R1-R4 completados y verificados localmente. R5 centraliza `0.2.1` en `VERSION`; RAG, OCR, health, OpenAPI, etiquetas OCI y builds rechazan valores declarados divergentes. La validacion local completa ya paso: imagen OCR construida con PaddleOCR/PaddlePaddle y modelos baked, health real correcto, suite Python en imagen y suite Node correctas. El fixture autocontenido de 25 paginas valida advertencias p10, 25 PNG y candidata revisable; las caidas en cada frontera de persistencia se recuperan exactamente una vez y los diagnosticos administrativos no exponen errores internos. La migracion `004` se valido con historial `001-003` y una fila historica en PostgreSQL Docker local aislado; el contenedor reutilizable quedo detenido. R5 permanece bloqueada hasta autorizacion explicita de despliegue conjunto. Contrato completo en `CONTRATO_CICLO_VIDA_Y_OCR.md`, anexo "Hallazgos productivos v6-v7". + +1. Completado localmente: R1, identidad `ocr-v2` completa, acknowledgement de los cuatro estados, recuperacion posterior a `succeeded` sin reenviar OCR y rechazo fail-closed de respuestas incompatibles. +2. Completado localmente: R2, `p10Confidence` bajo aislado como advertencia, bloqueos reales preservados y diagnostico determinista por pagina. +3. Completado localmente: R3, migracion `004` preparada sin aplicar, cadena de hashes durable, `quality-report.json` privado e inmutable y diagnostico seguro. +4. Completado localmente: R4, suites, revision independiente, fixture autocontenido de 25 paginas, simulaciones de recuperacion tras cada frontera de persistencia y hardening de errores administrativos. `004` se aplico mediante el runner en PostgreSQL Docker local aislado despues de controlar `001-003`; se verificaron la fila historica, columnas, restriccion, historial `001-004` e idempotencia. El contenedor reutilizable queda detenido; nunca se uso EasyPanel ni produccion. +5. Pendiente: R5, consumir `0.2.1` desde un unico `VERSION`, validar igualdad de health y etiquetas OCI, desplegar y probar RAG/OCR como una unica operacion y repetir aceptacion no activa con revision humana. + +#### Checklist ejecutable de R5 + +Retomar esta lista en orden despues de la compactacion. No iniciar acciones productivas antes de completar la preparacion local y obtener la autorizacion correspondiente. + +- [x] Crear `VERSION` en la raiz con `0.2.1` y convertirlo en la unica fuente de version para RAG, OCR, health, pruebas y builds; cualquier version duplicada o divergente debe fallar. +- [x] Ejecutar la validacion local completa de RAG, OCR e imagenes y actualizar la documentacion afectada. +- [ ] Revisar el worktree, crear un unico commit de entrega y hacer push a `main`; RAG y OCR deben publicarse desde ese mismo commit. +- [ ] Antes del despliegue, identificar explicitamente la base y el entorno productivos, revisar `rag_schema_migrations` y confirmar que solo estan pendientes las migraciones esperadas. +- [ ] Configurar en ambos servicios `BUILD_REVISION=` y mantener `OCR_INGEST_ENABLED=true`. +- [ ] Desplegar RAG y OCR desde el mismo commit como una unica operacion en EasyPanel; no considerar completada la entrega hasta que ambos terminen. +- [ ] Verificar conjuntamente `version=0.2.1`, revision, etiquetas OCI, digests, migracion `004`, health, cola OCR, almacenamiento, PostgreSQL, Qdrant y reconciliador. +- [ ] Confirmar el estado auditado de v7 sin aprobarla, indexarla ni activarla. +- [ ] Crear una candidata nueva de FacturaTech con `activate=false` y supervisarla hasta `review_required` o fallo. +- [ ] Ejecutar la revision humana de las 34 entradas y los cuatro codigos criticos; registrar IDs, estados, recuentos, hashes, versiones, revisiones y digests. +- [ ] Solicitar autorizacion separada antes de aprobar o indexar y una confirmacion adicional antes de activar. + +Si aparece un fallo real cuyos sintomas justifiquen aislar servicios, valorar entonces `OCR_INGEST_ENABLED=false` y el despliegue por etapas. No aplicar ese modo preventivamente. + +**Regla de despliegue:** R1-R4 no despliegan ni prueban produccion. Solo R5 puede desplegar; RAG y OCR se despliegan juntos con `OCR_INGEST_ENABLED=true` y la entrega no se considera completada hasta que ambos coincidan en version/revision y superen juntos la bateria de salud. No se desactiva OCR ni se despliega por etapas de forma preventiva: ese modo queda reservado para un fallo real cuyos sintomas justifiquen aislar un servicio. `BUILD_REVISION` sigue manual para esta entrega y se contrasta con el commit; los digests son la identidad exacta. No se introduce CI, registry ni automatizacion de EasyPanel sin aprobacion separada. + +**Regla PostgreSQL de R4:** no usar la base configurada en EasyPanel ni cargar los ficheros locales que contienen sus secretos. Usar un contenedor local aislado y reutilizable, pasar una URL local solo al proceso de prueba y comprobar la secuencia de migraciones. El contenedor conserva su volumen para futuras pruebas, no tiene reinicio automatico y debe quedar detenido al terminar para que PostgreSQL no consuma CPU ni RAM activa. No apagar ni limpiar Docker globalmente. El runner aplica todas las migraciones pendientes. La aplicacion de `004` en produccion pertenece a R5 y necesita autorizacion explicita independiente. + +**Evidencia:** las pruebas automatizadas deben ser autocontenidas. La comparacion con el documento manual de FacturaTech es evidencia externa y se registra como tal, separada del PASS automatizado. R5 debe dejar IDs, estados, recuentos, hashes, versiones, revisiones y digests suficientes para auditar la aceptacion sin depender de memoria conversacional. El SDD `ocr-ingest-integration` se archiva con 29/30 tareas completas y 7.4 incompleta. La continuidad del hardening y de la aceptacion se controla mediante el contrato ODD canónico, sin declarar retrospectivamente superada la aceptacion fallida. -**Salida:** el runtime OCR queda estabilizado y la ingesta OCR de FacturaTech se acepta y activa de forma segura solo con autorizacion explicita. +**Salida:** `0.2.1` supera los criterios del anexo, FacturaTech alcanza `review_required` con evidencia completa y solo se aprueba o activa con autorizacion explicita. ## 1. Documentacion y descubrimiento de la API diff --git a/docs/REGISTRO_SITUACIONES.md b/docs/REGISTRO_SITUACIONES.md new file mode 100644 index 0000000..89bbcb4 --- /dev/null +++ b/docs/REGISTRO_SITUACIONES.md @@ -0,0 +1,83 @@ +# Registro de situaciones detectadas + +Este documento registra errores, incidencias, bloqueos y hallazgos concretos. No sustituye `HISTORIAL_SESIONES.md`, que resume las sesiones de trabajo. + +## Como registrar una entrada + +1. Usa fecha y hora local en formato `YYYY-MM-DD HH:mm`. +2. Incluye evidencia verificable cuando exista. +3. Indica el estado: `pendiente`, `en analisis`, `resuelto` o `descartado`. +4. No borres entradas anteriores; actualizalas o anade una nota posterior. + +## Plantilla de entrada + +```markdown +### Entrada (YYYY-MM-DD HH:mm) + +- Contexto: +- Situacion observada: +- Resultado esperado: +- Resultado real: +- Evidencia: +- Accion aplicada: +- Validacion posterior: +- Estado: pendiente | en analisis | resuelto | descartado +``` + +## Entradas + +### Entrada (2026-09-17 12:00) + +- Contexto: Diagnostico remoto de solo lectura del error OCR FacturaTech v4. +- Flujo probado: Automatizacion SSH hacia VPS2 a partir de la fuente canonica de acceso. +- Situacion observada: Los delimitadores Markdown de la contrasena causaban extracciones incorrectas recurrentes; un intento PTY que reenvio la entrada estandar expuso el valor en la salida local y no autentico. +- Resultado esperado: Extraer y usar la credencial sin delimitadores ni exponerla. +- Resultado real: Se eliminaron los backticks de la fuente canonica; la linea de contrasena se valido sin delimitadores. +- Evidencia: `Servidores/VPS2/Vps2_despliegue_apps/instrucciones_montado_y_despliegue_apps_vps2_easypanel.md`; Engram #3197, #3292 y #3598. +- Hipotesis de causa: El formato Markdown introducia ambiguedad para extractores automatizados; `script` no es apto para recibir una contrasena por entrada estandar porque puede hacer eco antes de que SSH desactive el eco. +- Accion aplicada: Eliminados los delimitadores Markdown con aprobacion explicita del usuario. Queda prohibido reenviar la contrasena por entrada estandar a una pseudo-terminal. +- Validacion posterior: La linea de contrasena existe y no contiene backticks; acceso remoto pendiente mediante un mecanismo PTY que no use entrada estandar. +- Estado: en analisis +- Notas: El usuario difiere la rotacion de credenciales hasta el cierre de desarrollo. No registrar ni volver a mostrar el valor. + +### Entrada (2026-09-17 12:10) + +- Contexto: Revision productiva de la candidata OCR FacturaTech v4 `ce1b6462-7617-4721-aa3a-8e8216a584ed`. +- Flujo probado: `GET /ingestions/:versionId/review`, comprobacion remota de solo lectura del volumen `/data/ingestions` y recuperacion administrativa autenticada aprobada por el usuario. +- Situacion observada: La version estaba en `review_required`, pero su evidencia durable estaba incompleta. +- Resultado esperado: Preservar manifest, paginas nativas, resultado OCR, imagenes y `candidate-pages.json` antes de pasar a revision. +- Resultado real: Solo existen `manifest.json` y el PDF original; faltan las paginas nativas, resultados OCR, imagenes de revision y `candidate-pages.json`. +- Evidencia: `ENOENT` para `/data/ingestions/ce1b6462-7617-4721-aa3a-8e8216a584ed/candidate-pages.json`; inspeccion de solo lectura del contenedor RAG y Engram #3600. +- Hipotesis de causa: Confirmada. v4 se creo seis horas antes de desplegar el lector durable de revision. El flujo anterior marcaba `review_required` sin persistir paginas nativas, resultado OCR, imagenes ni candidata compuesta. +- Accion aplicada: Tras confirmar el contrato seguro `409 OCR_ARTIFACT_UNAVAILABLE` con accion `use_admin_recovery` y recibir aprobacion explicita del usuario, `POST /ingestions/:versionId/recover` cerro exclusivamente v4 como `failed` con resultado `closed_failed`. +- Validacion posterior: PostgreSQL confirma que v4 no es activa, conserva la version activa previa de la fuente, contiene un registro de auditoria y ya no hay candidatas OCR bloqueantes. +- Estado: resuelto +- Notas: No se aprobo, rechazo, indexo ni activo contenido. + +### Entrada (2026-09-11 23:00) + +- Contexto: Inspeccion SSH de PostgreSQL para preparar la base del RAG. +- Flujo probado: Conexion mediante `SSH_ASKPASS` usando la fuente canonica de credenciales de VPS2. +- Situacion observada: Las primeras conexiones fallaron porque el agente incluyo los backticks Markdown de la contrasena al extraerla del documento. +- Resultado esperado: Conectar usando solo el valor de la contrasena, sin delimitadores de formato. +- Resultado real: La conexion funciono al eliminar los backticks. La consulta posterior a PostgreSQL no pudo usar el rol `postgres` porque ese rol no existe en la instancia. +- Evidencia: `Servidores/VPS2/Vps2_despliegue_apps/instrucciones_montado_y_despliegue_apps_vps2_easypanel.md`; salida SSH; error `FATAL: role "postgres" does not exist`. +- Hipotesis de causa: La instancia usa el usuario administrativo configurado por EasyPanel, no necesariamente el rol estandar `postgres`. +- Accion aplicada: Anadida una nota permanente en la fuente canonica indicando que no se deben usar los backticks. +- Validacion posterior: SSH funciona; queda identificar el usuario administrativo real sin asumir `postgres`. +- Estado: en analisis +- Notas: No repetir intentos con `postgres` hasta consultar la configuracion de la instancia sin exponer contrasenas. + +### Entrada (2026-09-11 20:27) + +- Contexto: Deploy del punto 2 del ciclo de vida del conocimiento del RAG. +- Flujo probado: Inicio del servicio y consulta `GET https://rag.por-correo.com/health`. +- Situacion observada: El RAG se desplego y escuchaba, pero PostgreSQL aparecia como no disponible. +- Resultado esperado: PostgreSQL preparado, conectado y disponible para el catalogo de fuentes y versiones. +- Resultado real: Qdrant y el reconciliador estaban operativos, pero `postgres.ok=false` por falta de configuracion de conexion en EasyPanel. +- Evidencia: Respuesta de `/health`; `src/config/env.ts`; `src/modules/catalog/client.ts`; `migrations/001_knowledge_lifecycle.sql`; commit `551cfe8`. +- Hipotesis de causa: Faltaba crear o confirmar la base de datos, configurar `POSTGRES_URL`/`POSTGRES_SSL` y completar la preparacion operativa de PostgreSQL. +- Accion aplicada: Se detuvieron la migracion legacy, la activacion de enforcement y el inicio de OCR. Se creo un prerrequisito bloqueante en `docs/PENDIENTES_RAG.md`. +- Validacion posterior: Esta incidencia historica fue superada; la validacion local de `004` sigue pendiente como R4. +- Estado: resuelto +- Notas: No compartir credenciales. diff --git a/docs/info-git-por-correo.md b/docs/info-git-por-correo.md new file mode 100644 index 0000000..3d2ffa7 --- /dev/null +++ b/docs/info-git-por-correo.md @@ -0,0 +1,158 @@ +# Informacion de acceso a Git - Forgejo + +**Proyecto:** RAG Service +**Ultima actualizacion:** 2026-09-22 +**Ultima modificacion por:** Agente RAG 3 +**Estado:** Activo + +--- + +## Proposito + +Este documento deja la informacion de acceso al repositorio Git de Forgejo para que cualquier agente pueda hacer commits y push cuando el usuario lo pida. + +--- + +## Repositorio RAG + +### URL remota + +```text +ssh://git@git.por-correo.com:2222/paco/rag-service.git +``` + +### Datos de conexion + +| Campo | Valor | +|-------|-------| +| Usuario | `git` | +| Host | `git.por-correo.com` | +| Puerto SSH | `2222` | +| Metodo de autenticacion | Clave publica SSH | +| Rama principal | `main` | + +### Comandos habituales + +```bash +# Ver estado del repositorio +git status --short --branch + +# Añadir archivos al staging +git add + +# Hacer commit +git commit -m "" + +# Subir cambios al remoto +git push origin main + +# Verificar despues del push +git status --short --branch +``` + +### Verificacion antes de commitear + +Antes de hacer commit y push, verificar: + +```bash +# Ver que archivos se van a incluir +git status --short + +# Ver el diff de los cambios +git diff --stat +``` + +--- + +## Reglas de uso para agentes + +### Lo que SI se debe hacer + +1. **Verificar el estado antes de commitear** + - Revisar que solo se incluyen archivos relevantes para la tarea + - No mezclar cambios de diferentes tareas en un mismo commit + +2. **Usar mensajes de commit descriptivos** + - Explicar el "por que" del cambio, no solo el "que" + - Ser conciso pero informativo + +3. **Validar antes de push** + - Compilar el proyecto si aplica (`npm run build`, `npm run check`) + - Verificar que no hay errores de tipo o compilacion + +4. **Documentar cambios relevantes** + - Actualizar documentacion si se introduce funcionalidad nueva + - Registrar cambios en `HISTORIAL_SESIONES.md` si corresponde + +### Lo que NO se debe hacer + +1. **No tocar configuracion global de Git** + ```bash + # NO hacer esto + git config --global user.name "..." + git config --global user.email "..." + ``` + +2. **No subir archivos sensibles** + - `.env.local` + - `.env.easypanel.local` + - credenciales + - claves privadas + - secretos de cualquier tipo + +3. **No hacer force push a main** + ```bash + # NUNCA hacer esto + git push --force origin main + ``` + +4. **No commitear sin validacion** + - No subir codigo que no compila + - No subir cambios sin verificar `git status` + +--- + +## Documentacion relacionada + +- `RAG/docs/METODOLOGIA_ITERACION_Y_REDEPLOY.md` - Flujo de trabajo para iterar y desplegar +- `RAG/docs/DESPLIEGUE_EASYPANEL.md` - Informacion de despliegue en EasyPanel + +--- + +## Referencia rapida para agentes + +Cuando el usuario pida hacer commit y push: + +1. **Verificar cambios** + ```bash + git status --short + ``` + +2. **Añadir archivos relevantes** + ```bash + git add + ``` + +3. **Hacer commit con mensaje descriptivo** + ```bash + git commit -m "Descripcion clara del cambio" + ``` + +4. **Subir al remoto** + ```bash + git push origin main + ``` + +5. **Confirmar al usuario** + - Indicar que los cambios estan subidos + - Proporcionar el hash del commit + - Avisar si ya puede hacer deploy en EasyPanel (si aplica) + +--- + +## Notas importantes + +- La autenticacion se hace por clave publica SSH registrada en Forgejo +- No es necesario configurar usuario/email porque ya esta en la configuracion local del repo +- El puerto SSH es `2222`, no el estandar `22` +- El repo esta en el host `git.por-correo.com` diff --git a/docs/readme.md b/docs/readme.md new file mode 100644 index 0000000..b206223 --- /dev/null +++ b/docs/readme.md @@ -0,0 +1,34 @@ +# README para Agentes + +## Proyecto: RAG Service + +## Reglas comunes + +- Respeta la aprobacion explicita del usuario antes de crear, modificar o eliminar ficheros, ejecutar comandos relevantes o realizar cambios significativos. +- Si falta informacion critica o aparece un bloqueo, detente, informalo y espera instrucciones. +- Toda la documentacion del modulo vive en `docs/`. +- El unico readme valido del modulo es `docs/readme.md`. + +## Si eres agente principal + +1. Recupera tu `session_id` con `docs/sesion_actual_opencode.md` cuando el runtime no lo proporcione. +2. Consulta `docs/HISTORIAL_SESIONES.md` para recuperar identidad y continuidad. +3. Si no estas registrado, solicita nombre y rol antes de iniciar trabajo de proyecto. +4. Antes de crear o buscar documentacion, consulta `docs/INDICE_DOCUMENTACION.md`. +5. Antes de trabajar con servicios, credenciales, despliegues o ejecuciones recurrentes, consulta `docs/OPERATIVA.md`. +6. Registra cada bloque relevante de trabajo en `docs/HISTORIAL_SESIONES.md` antes de cerrarlo. + +## Si eres subagente + +- Tu identidad y responsabilidad las dicta el agente invocador. +- No ejecutes el flujo de identidad ni la comprobacion de plantilla. +- No leas el indice ni el historial salvo que la tarea lo requiera. +- Si necesitas tu `session_id`, usa la receta `parent_id` de `docs/sesion_actual_opencode.md`. +- Registra tu trabajo en el historial con el nombre `Subagente [tarea]` y tu propio `session_id`. + +## Reglas especificas de RAG + +- Confirma que la sesion esta en la raiz `RAG/` y que Engram resuelve `rag-service` antes de cualquier trabajo del modulo. +- Indica siempre `project: "rag-service"` en las operaciones Engram de RAG. +- El contrato OCR y el backlog canonicos son `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` y `docs/PENDIENTES_RAG.md`. +- R1-R4 se ejecutan exclusivamente en local. R5 es la unica fase que puede desplegar RAG y OCR como una operacion conjunta. diff --git a/docs/sesion_actual_opencode.md b/docs/sesion_actual_opencode.md new file mode 100644 index 0000000..c631a21 --- /dev/null +++ b/docs/sesion_actual_opencode.md @@ -0,0 +1,19 @@ +# Sesion actual de OpenCode + +## Comando canonico + +```bash +opencode db "SELECT id,title,directory,datetime(time_updated/1000,'unixepoch','localtime') AS updated FROM session WHERE directory='$(pwd)' ORDER BY time_updated DESC LIMIT 1" --format tsv +``` + +Ejecutalo desde la raiz del worktree. Si no hay filas, responde `SIN_SESION_EN_ESTE_WORKSPACE`; no inventes datos. + +## Identificacion de subagentes + +El agente invocador debe incluir su `session_id` en el prompt. El subagente recupera el suyo con: + +```bash +opencode db "SELECT id,title FROM session WHERE parent_id='[SESSION_ID_DEL_INVOCADOR]' ORDER BY time_created DESC LIMIT 1" --format tsv +``` + +Si hay varios hijos, desambigua por el titulo de la tarea. Si el id coincide con el del invocador, informa el fallo y no inventes una identidad. diff --git a/migrations/004_ocr_quality_diagnostics.sql b/migrations/004_ocr_quality_diagnostics.sql new file mode 100644 index 0000000..ae7eb01 --- /dev/null +++ b/migrations/004_ocr_quality_diagnostics.sql @@ -0,0 +1,19 @@ +ALTER TABLE rag_document_pages + ADD COLUMN IF NOT EXISTS quality_outcome text NULL, + ADD COLUMN IF NOT EXISTS quality_warnings jsonb NOT NULL DEFAULT '[]'::jsonb, + ADD COLUMN IF NOT EXISTS blocking_reasons jsonb NOT NULL DEFAULT '[]'::jsonb, + ADD COLUMN IF NOT EXISTS primary_blocking_reason text NULL, + ADD COLUMN IF NOT EXISTS quality_report_sha256 char(64) NULL; + +DO $$ +BEGIN + IF NOT EXISTS ( + SELECT 1 FROM pg_constraint + WHERE conrelid = 'rag_document_pages'::regclass + AND conname = 'rag_document_pages_quality_outcome_check' + ) THEN + ALTER TABLE rag_document_pages + ADD CONSTRAINT rag_document_pages_quality_outcome_check + CHECK (quality_outcome IN ('accepted', 'warning', 'blocked')); + END IF; +END $$; diff --git a/ocr-service/Dockerfile b/ocr-service/Dockerfile index 21208d0..5c1e723 100644 --- a/ocr-service/Dockerfile +++ b/ocr-service/Dockerfile @@ -1,6 +1,8 @@ FROM python:3.11.13-slim-bookworm -ARG OCR_VERSION=0.2.0 +ARG OCR_VERSION ARG BUILD_REVISION=unknown +COPY VERSION /srv/VERSION +RUN test "$OCR_VERSION" = "$(tr -d '\r\n' < /srv/VERSION)" LABEL resource.cpu.max="3" \ resource.memory.max="5GiB" diff --git a/ocr-service/Dockerfile.dockerignore b/ocr-service/Dockerfile.dockerignore index fc11f50..f9b4a53 100644 --- a/ocr-service/Dockerfile.dockerignore +++ b/ocr-service/Dockerfile.dockerignore @@ -1,4 +1,5 @@ ** +!VERSION !ocr-service/ !ocr-service/app/ !ocr-service/app/** diff --git a/ocr-service/app/jobs.py b/ocr-service/app/jobs.py index 8193de1..5bd45c5 100644 --- a/ocr-service/app/jobs.py +++ b/ocr-service/app/jobs.py @@ -64,13 +64,13 @@ class JobQueue: self.connection.execute( "CREATE TABLE IF NOT EXISTS jobs (" "job_id TEXT PRIMARY KEY, idempotency_key TEXT UNIQUE, payload_hash TEXT NOT NULL, " - "document_sha256 TEXT NOT NULL, pages TEXT NOT NULL, status TEXT NOT NULL, created_at TEXT NOT NULL, " + "document_sha256 TEXT NOT NULL, pages TEXT NOT NULL, config_version TEXT NOT NULL DEFAULT 'ocr-v2', status TEXT NOT NULL, created_at TEXT NOT NULL, " "input_path TEXT, result TEXT, error TEXT, attempts INTEGER NOT NULL DEFAULT 0, " "recovery_attempts INTEGER NOT NULL DEFAULT 0, started_at TEXT, completed_at TEXT, lease_expires_at TEXT)" ) columns = {row[1] for row in self.connection.execute("PRAGMA table_info(jobs)")} for name, kind in ( - ("input_path", "TEXT"), ("result", "TEXT"), ("error", "TEXT"), + ("input_path", "TEXT"), ("result", "TEXT"), ("error", "TEXT"), ("config_version", "TEXT NOT NULL DEFAULT 'ocr-v2'"), ("attempts", "INTEGER NOT NULL DEFAULT 0"), ("recovery_attempts", "INTEGER NOT NULL DEFAULT 0"), ("started_at", "TEXT"), ("completed_at", "TEXT"), ("lease_expires_at", "TEXT"), ): @@ -176,11 +176,24 @@ class JobQueue: def contains(self, key: str) -> bool: return self.connection.execute("SELECT 1 FROM jobs WHERE idempotency_key=?", (key,)).fetchone() is not None + @staticmethod + def _identity(row: sqlite3.Row) -> dict[str, Any]: + pages = json.loads(row["pages"]) + config_version = row["config_version"] + canonical = json.dumps({ + "configVersion": config_version, + "documentSha256": row["document_sha256"], + "idempotencyKey": row["idempotency_key"], + "requestedPages": pages, + }, sort_keys=True, separators=(",", ":")).encode() + return {"idempotencyKey": row["idempotency_key"], "documentSha256": row["document_sha256"], + "requestedPages": pages, "configVersion": config_version, + "requestIdentitySha256": hashlib.sha256(canonical).hexdigest()} + @staticmethod def ack(row: sqlite3.Row) -> dict[str, Any]: return { - "jobId": row["job_id"], "status": row["status"], "documentSha256": row["document_sha256"], - "requestedPages": json.loads(row["pages"]), "configVersion": "ocr-v1", "createdAt": row["created_at"], + "jobId": row["job_id"], "status": row["status"], **JobQueue._identity(row), "createdAt": row["created_at"], } def submit(self, key: str, request: dict[str, Any], pdf: bytes) -> tuple[dict[str, Any], bool]: @@ -199,9 +212,9 @@ class JobQueue: try: input_path = self._write_artifact(job_id, pdf) self.connection.execute( - "INSERT INTO jobs (job_id,idempotency_key,payload_hash,document_sha256,pages,status,created_at,input_path) " - "VALUES (?,?,?,?,?,?,?,?)", - (job_id, key, payload_hash, request["documentSha256"], json.dumps(request["pages"]), "queued", self._iso(self.now()), input_path), + "INSERT INTO jobs (job_id,idempotency_key,payload_hash,document_sha256,pages,config_version,status,created_at,input_path) " + "VALUES (?,?,?,?,?,?,?,?,?)", + (job_id, key, payload_hash, request["documentSha256"], json.dumps(request["pages"]), request["configVersion"], "queued", self._iso(self.now()), input_path), ) self.connection.commit() except Exception: @@ -293,6 +306,8 @@ class JobQueue: row["job_id"], row["document_sha256"], input_path.read_bytes(), json.loads(row["pages"]), engine, persist_review_image=lambda page, png: self._persist_review_image(row["job_id"], page, png), ) + result.update(self._identity(row)) + result["engine"]["configVersion"] = row["config_version"] update = ("succeeded", json.dumps(result, sort_keys=True, separators=(",", ":")), None) except StoragePressureError: self._remove_review_images(row["job_id"]) @@ -350,15 +365,16 @@ class JobQueue: if not row: return None total_pages = len(json.loads(row["pages"])) - return {"jobId": job_id, "status": row["status"], "completedPages": total_pages if row["status"] == "succeeded" else 0, + return {"jobId": job_id, "status": row["status"], **self._identity(row), + "completedPages": total_pages if row["status"] == "succeeded" else 0, "totalPages": total_pages, "error": json.loads(row["error"]) if row["error"] else None} def result(self, job_id: str) -> tuple[str, dict[str, Any] | None] | None: row = self.connection.execute("SELECT status,result FROM jobs WHERE job_id=?", (job_id,)).fetchone() return (row["status"], json.loads(row["result"]) if row["result"] else None) if row else None - def review_image(self, job_id: str, page: int) -> tuple[str, bytes] | None: - row = self.connection.execute("SELECT document_sha256,pages,status FROM jobs WHERE job_id=?", (job_id,)).fetchone() + def review_image(self, job_id: str, page: int) -> tuple[dict[str, Any], bytes] | None: + row = self.connection.execute("SELECT * FROM jobs WHERE job_id=?", (job_id,)).fetchone() if not row: return None if row["status"] != "succeeded": @@ -366,7 +382,7 @@ class JobQueue: if page not in json.loads(row["pages"]): raise ValueError("INVALID_PAGE") try: - return row["document_sha256"], self._review_image_path(job_id, page).read_bytes() + return self._identity(row), self._review_image_path(job_id, page).read_bytes() except FileNotFoundError as error: raise RuntimeError("RESULT_NOT_READY") from error diff --git a/ocr-service/app/main.py b/ocr-service/app/main.py index 7766b3c..d07b6e1 100644 --- a/ocr-service/app/main.py +++ b/ocr-service/app/main.py @@ -12,6 +12,7 @@ from fastapi import Depends, FastAPI, File, Form, Header, HTTPException, Respons from .engine import OcrEngine from .jobs import JobActiveError, JobQueue, QueueFullError, StoragePressureError from .models import load_runtime_engine +from .version import release_version MAX_UPLOAD_BYTES = 50 * 1024 * 1024 @@ -22,7 +23,7 @@ ALLOWED_CONFIG = { "engine": "paddleocr", "engineVersion": "3.4.0", "runtimeVersion": "3.2.2", - "configVersion": "ocr-v1", + "configVersion": "ocr-v2", "returnLayout": True, } REQUEST_FIELDS = {"documentSha256", "pages", *ALLOWED_CONFIG} @@ -37,6 +38,16 @@ def page_hash(pages: list[int]) -> str: return hashlib.sha256(value).hexdigest() +def request_identity(idempotency_key: str, document_sha256: str, pages: list[int], config_version: str) -> str: + canonical = json.dumps({ + "configVersion": config_version, + "documentSha256": document_sha256, + "idempotencyKey": idempotency_key, + "requestedPages": pages, + }, sort_keys=True, separators=(",", ":")).encode() + return hashlib.sha256(canonical).hexdigest() + + def utc_now() -> datetime: return datetime.now(timezone.utc) @@ -49,6 +60,7 @@ def create_app( engine: OcrEngine | None = None, now: Callable[[], datetime] = utc_now, ) -> FastAPI: + version = release_version() queue = JobQueue(db_path, now) @asynccontextmanager @@ -69,7 +81,7 @@ def create_app( @application.get("/health/live") def live() -> dict[str, str]: - return {"status": "ok", "service": "ocr", "version": os.getenv("OCR_VERSION", "0.2.0"), + return {"status": "ok", "service": "ocr", "version": version, "revision": os.getenv("BUILD_REVISION", "unknown")} @application.get("/health/ready") @@ -80,7 +92,7 @@ def create_app( if not ready_state: response.status_code = 503 return {"ready": ready_state, "engineLoaded": engine_ready, "storageAvailable": storage_available, - **state, "service": "ocr", "version": os.getenv("OCR_VERSION", "0.2.0"), + **state, "service": "ocr", "version": version, "revision": os.getenv("BUILD_REVISION", "unknown")} @application.post("/v1/jobs", status_code=202, dependencies=[Depends(authorize)]) @@ -111,7 +123,7 @@ def create_app( fail(422, "INTEGRITY_MISMATCH", "PDF bytes do not match documentSha256") if file.content_type != "application/pdf" or not content.startswith(b"%PDF-") or b"/Encrypt" in content: fail(422, "UNSUPPORTED_PDF", "PDF is corrupt, encrypted, or unsupported") - expected_key = f'{digest}:ocr-v1:{page_hash(pages)}' + expected_key = f'{digest}:ocr-v2:{page_hash(pages)}' if idempotency_key != expected_key and not queue.contains(idempotency_key or ""): fail(400, "INVALID_IDEMPOTENCY_KEY", "Idempotency-Key does not match request identity") try: @@ -150,9 +162,12 @@ def create_app( fail(422, "INVALID_PAGE", "OCR review image does not exist") if stored is None: fail(404, "JOB_NOT_FOUND", "OCR job does not exist") - document_sha256, png = stored + identity, png = stored return Response(png, media_type="image/png", headers={ - "X-Document-Sha256": document_sha256, + "X-Ocr-Job-Id": job_id, + "X-Ocr-Identity-Sha256": identity["requestIdentitySha256"], + "X-Ocr-Config-Version": identity["configVersion"], + "X-Document-Sha256": identity["documentSha256"], "X-Content-Sha256": hashlib.sha256(png).hexdigest(), "X-Page-Number": str(page), }) diff --git a/ocr-service/app/render.py b/ocr-service/app/render.py index 132b953..9aa2aa5 100644 --- a/ocr-service/app/render.py +++ b/ocr-service/app/render.py @@ -126,7 +126,7 @@ def process_pdf( "version": "3.4.0", "runtime": "paddlepaddle-3.2.2", "device": "cpu", - "configVersion": "ocr-v1", + "configVersion": "ocr-v2", "dpi": RENDER_DPI, }, "pages": results, diff --git a/ocr-service/app/version.py b/ocr-service/app/version.py new file mode 100644 index 0000000..733ddf0 --- /dev/null +++ b/ocr-service/app/version.py @@ -0,0 +1,12 @@ +import os +from pathlib import Path + + +def release_version() -> str: + value = (Path(__file__).parents[2] / "VERSION").read_text(encoding="utf-8").strip() + if not value or any(part == "" or not part.isdecimal() for part in value.split(".")) or len(value.split(".")) != 3: + raise RuntimeError("VERSION must contain a semantic version") + declared = os.getenv("OCR_VERSION") + if declared is not None and declared != value: + raise RuntimeError(f"OCR_VERSION must match VERSION ({value})") + return value diff --git a/ocr-service/tests/test_api.py b/ocr-service/tests/test_api.py index eef67ee..777553e 100644 --- a/ocr-service/tests/test_api.py +++ b/ocr-service/tests/test_api.py @@ -1,5 +1,6 @@ import hashlib import json +import os import sqlite3 import sys import time @@ -14,6 +15,7 @@ sys.path.insert(0, str(Path(__file__).parents[1])) from app.main import create_app from app.engine import EngineLine from app.jobs import JobQueue +from app.version import release_version TOKEN = "unit-5-test-token" @@ -35,14 +37,14 @@ def request_for(pdf: bytes = PDF, pages: list[int] | None = None) -> dict: "engine": "paddleocr", "engineVersion": "3.4.0", "runtimeVersion": "3.2.2", - "configVersion": "ocr-v1", + "configVersion": "ocr-v2", "returnLayout": True, } def key_for(request: dict) -> str: pages = json.dumps(request["pages"], separators=(",", ":")).encode() - return f'{request["documentSha256"]}:ocr-v1:{hashlib.sha256(pages).hexdigest()}' + return f'{request["documentSha256"]}:ocr-v2:{hashlib.sha256(pages).hexdigest()}' def submit(client: TestClient, request: dict, pdf: bytes = PDF, key: str | None = None): @@ -73,9 +75,10 @@ def wait_for_status(client: TestClient, job_id: str, expected: str) -> dict: def test_auth_rejects_missing_and_wrong_bearer_and_health_exposes_no_secret(client: TestClient): + revision = os.getenv("BUILD_REVISION", "unknown") live = client.get("/health/live") ready = client.get("/health/ready") - assert live.json() == {"status": "ok", "service": "ocr", "version": "0.2.0", "revision": "unknown"} + assert live.json() == {"status": "ok", "service": "ocr", "version": release_version(), "revision": revision} assert ready.status_code == 503 assert ready.json() == { "ready": False, "engineLoaded": True, "storageAvailable": True, @@ -84,7 +87,7 @@ def test_auth_rejects_missing_and_wrong_bearer_and_health_exposes_no_secret(clie "concurrency": 1, "workerState": "stopped", "workerOperational": False, "sweeperOperational": False, "recoveryAttempts": 0, "lastSweepAt": None, - "service": "ocr", "version": "0.2.0", "revision": "unknown", + "service": "ocr", "version": release_version(), "revision": revision, } for authorization in (None, "Bearer wrong-token"): headers = {"Idempotency-Key": key_for(request_for())} @@ -104,6 +107,12 @@ def test_auth_rejects_missing_and_wrong_bearer_and_health_exposes_no_secret(clie assert unavailable.json()["ready"] is False +def test_declared_ocr_version_must_match_version_file(monkeypatch: pytest.MonkeyPatch, tmp_path: Path): + monkeypatch.setenv("OCR_VERSION", "incorrect") + with pytest.raises(RuntimeError, match="OCR_VERSION must match VERSION"): + create_app(TOKEN, tmp_path / "jobs.db", engine_ready=True) + + def test_auth_submission_is_idempotent_and_conflicting_payload_is_terminal(client: TestClient): request = request_for() first = submit(client, request) @@ -121,6 +130,28 @@ def test_auth_submission_is_idempotent_and_conflicting_payload_is_terminal(clien } +def test_auth_duplicates_preserve_each_known_remote_state_and_reject_ocr_v1(tmp_path: Path): + db_path = tmp_path / "states.db" + client = TestClient(create_app(TOKEN, db_path, engine_ready=True)) + request = request_for() + accepted = submit(client, request) + job_id = accepted.json()["jobId"] + + for state in ("queued", "running", "succeeded", "failed"): + with sqlite3.connect(db_path) as connection: + connection.execute("UPDATE jobs SET status=? WHERE job_id=?", (state, job_id)) + repeated = submit(client, request) + assert repeated.status_code == 202 + assert repeated.json()["jobId"] == job_id + assert repeated.json()["status"] == state + + legacy = request_for() + legacy["configVersion"] = "ocr-v1" + rejected = submit(client, legacy, key=key_for(legacy).replace(":ocr-v2:", ":ocr-v1:")) + assert rejected.status_code == 400 + assert rejected.json()["detail"]["code"] == "CONFIG_NOT_ALLOWED" + + @pytest.mark.parametrize( ("field", "value"), [("languages", ["en"]), ("dpi", 300), ("engine", "tesseract"), ("returnLayout", False)], @@ -159,7 +190,11 @@ def test_auth_queue_pressure_is_retryable_and_status_and_delete_are_authenticate assert pressure.headers["Retry-After"] == "2" job_id = accepted[0].json()["jobId"] status = client.get(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"}) - assert status.json() == {"jobId": job_id, "status": "queued", "completedPages": 0, "totalPages": 2, "error": None} + assert {key: status.json()[key] for key in ("jobId", "status", "completedPages", "totalPages", "error")} == { + "jobId": job_id, "status": "queued", "completedPages": 0, "totalPages": 2, "error": None, + } + assert status.json()["configVersion"] == "ocr-v2" + assert status.json()["requestIdentitySha256"] assert client.delete(f"/v1/jobs/{job_id}").status_code == 401 assert client.delete(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"}).status_code == 204 assert client.delete(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"}).status_code == 204 @@ -195,9 +230,15 @@ def test_auth_accepted_job_executes_and_exposes_integrity_bound_result(tmp_path: job_id = accepted.json()["jobId"] status = wait_for_status(client, job_id, "succeeded") result = client.get(f"/v1/jobs/{job_id}/result", headers={"Authorization": f"Bearer {TOKEN}"}) - assert status == {"jobId": job_id, "status": "succeeded", "completedPages": 1, "totalPages": 1, "error": None} + assert {key: status[key] for key in ("jobId", "status", "completedPages", "totalPages", "error")} == { + "jobId": job_id, "status": "succeeded", "completedPages": 1, "totalPages": 1, "error": None, + } + assert status["configVersion"] == "ocr-v2" + assert status["requestIdentitySha256"] assert result.status_code == 200 assert (result.json()["jobId"], result.json()["documentSha256"]) == (job_id, hashlib.sha256(pdf).hexdigest()) + assert result.json()["configVersion"] == "ocr-v2" + assert result.json()["idempotencyKey"] == key_for(request_for(pdf, [1])) assert [(page["page"], page["text"]) for page in result.json()["pages"]] == [(1, "Factura FAT07")] ready = client.get("/health/ready") assert ready.status_code == 200 @@ -209,6 +250,9 @@ def test_auth_accepted_job_executes_and_exposes_integrity_bound_result(tmp_path: assert image.status_code == 200 assert image.headers["content-type"] == "image/png" assert image.headers["x-document-sha256"] == hashlib.sha256(pdf).hexdigest() + assert image.headers["x-ocr-job-id"] == job_id + assert image.headers["x-ocr-config-version"] == "ocr-v2" + assert image.headers["x-ocr-identity-sha256"] == result.json()["requestIdentitySha256"] assert image.headers["x-content-sha256"] == hashlib.sha256(image.content).hexdigest() assert image.content.startswith(b"\x89PNG\r\n\x1a\n") assert client.get(f"/v1/jobs/{job_id}/pages/1/image").status_code == 401 diff --git a/ocr-service/tests/test_render.py b/ocr-service/tests/test_render.py index 1644aa1..fb5fe3a 100644 --- a/ocr-service/tests/test_render.py +++ b/ocr-service/tests/test_render.py @@ -98,7 +98,7 @@ def test_render_builds_repeatable_result_schema_with_deterministic_engine() -> N "version": "3.4.0", "runtime": "paddlepaddle-3.2.2", "device": "cpu", - "configVersion": "ocr-v1", + "configVersion": "ocr-v2", "dpi": 200, } assert first["pages"][0] == { @@ -174,6 +174,7 @@ def test_render_container_pins_cpu_runtime_models_and_single_worker() -> None: assert "PaddleOcrEngine" in model_loader assert dockerignore == [ "**", + "!VERSION", "!ocr-service/", "!ocr-service/app/", "!ocr-service/app/**", diff --git a/opencode.json b/opencode.json new file mode 100644 index 0000000..e4be5d5 --- /dev/null +++ b/opencode.json @@ -0,0 +1,8 @@ +{ + "$schema": "https://opencode.ai/config.json", + "instructions": [ + "docs/sesion_actual_opencode.md", + "docs/CONTEXTO_PROYECTO.md", + "docs/readme.md" + ] +} diff --git a/package-lock.json b/package-lock.json index 1ccf798..f890cf3 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "rag-service", - "version": "0.2.0", + "version": "0.2.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "rag-service", - "version": "0.2.0", + "version": "0.2.1", "dependencies": { "@qdrant/js-client-rest": "^1.15.0", "adm-zip": "^0.6.1", diff --git a/package.json b/package.json index b665f28..6ed8a96 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "rag-service", - "version": "0.2.0", + "version": "0.2.1", "private": true, "type": "module", "scripts": { diff --git a/src/api/openapi.ts b/src/api/openapi.ts index c539fab..bb5a54b 100644 --- a/src/api/openapi.ts +++ b/src/api/openapi.ts @@ -1,3 +1,5 @@ +import { releaseVersion } from "../config/version.js"; + const ref = (name: string) => ({ $ref: `#/components/schemas/${name}` }); const jsonContent = (schema: Record, example?: unknown) => ({ @@ -18,7 +20,7 @@ export const openApiDocument = { openapi: "3.1.1", info: { title: "RAG Service API", - version: "0.2.0", + version: releaseVersion, description: "HTTP API for ingesting, retrieving, answering from, and evaluating scoped RAG knowledge." }, jsonSchemaDialect: "https://json-schema.org/draft/2020-12/schema", diff --git a/src/app.ts b/src/app.ts index 2ba60ee..c13cf62 100644 --- a/src/app.ts +++ b/src/app.ts @@ -80,7 +80,13 @@ export function createApp(options: AppOptions = {}) { rootDirectory: ocr.artifactRoot, versionId: job.versionId, documentId: job.documentId, - loadImage: async (page) => ocrClient!.getReviewImage(result.jobId, page, result.documentSha256) + loadImage: async (page) => ocrClient!.getReviewImage(result.jobId, page, { + idempotencyKey: job.remoteIdempotencyKey, + documentSha256: result.documentSha256, + requestedPages: result.requestedPages, + configVersion: result.configVersion, + requestIdentitySha256: result.requestIdentitySha256 + }) }); return durable.result; }, async (versionId) => { @@ -110,7 +116,7 @@ export function createApp(options: AppOptions = {}) { const statusCode = error instanceof CatalogError ? error.statusCode : upstreamUnavailable ? 503 : 500; const body = { ok: false, - error: error instanceof Error ? error.message : fallback, + error: error instanceof CatalogError ? error.message : fallback, code: error instanceof CatalogError ? error.code : undefined }; res.status(statusCode).json(body); diff --git a/src/config/env.ts b/src/config/env.ts index c3628f6..cceac7a 100644 --- a/src/config/env.ts +++ b/src/config/env.ts @@ -1,4 +1,5 @@ import { config as loadEnv } from "dotenv"; +import { resolveReleaseVersion } from "./version.js"; if (process.env.NODE_ENV !== "test") { loadEnv(); @@ -33,7 +34,7 @@ function booleanEnv(name: string, fallback: boolean): boolean { export const env = { nodeEnv: process.env.NODE_ENV ?? "development", - ragVersion: process.env.RAG_VERSION ?? "0.2.0", + ragVersion: resolveReleaseVersion("RAG_VERSION"), buildRevision: process.env.BUILD_REVISION ?? "unknown", port: Number(process.env.PORT ?? 3000), qdrantUrl: requireEnv("QDRANT_URL", "http://localhost:6333"), diff --git a/src/config/version.ts b/src/config/version.ts new file mode 100644 index 0000000..f19542d --- /dev/null +++ b/src/config/version.ts @@ -0,0 +1,14 @@ +import { readFileSync } from "node:fs"; + +export const releaseVersion = readFileSync(new URL("../../VERSION", import.meta.url), "utf8").trim(); + +if (!/^\d+\.\d+\.\d+$/u.test(releaseVersion)) { + throw new Error("VERSION must contain a semantic version"); +} + +export function resolveReleaseVersion(name: string, declared = process.env[name]): string { + if (declared !== undefined && declared !== releaseVersion) { + throw new Error(`${name} must match VERSION (${releaseVersion})`); + } + return releaseVersion; +} diff --git a/src/modules/catalog/reconciler.ts b/src/modules/catalog/reconciler.ts index bcc7f14..c46cc94 100644 --- a/src/modules/catalog/reconciler.ts +++ b/src/modules/catalog/reconciler.ts @@ -24,7 +24,7 @@ export class KnowledgeLifecycleReconciler { constructor( private readonly catalog: CatalogRepository | undefined, private readonly vectorStore: VectorStoreClient, - private readonly ocrDispatcher?: Pick, + private readonly ocrDispatcher?: Pick, private readonly ocrRetention?: Pick ) {} @@ -58,6 +58,9 @@ export class KnowledgeLifecycleReconciler { const invariantViolations = await catalog.validateActiveInvariant(); const ocrLeasesRecovered = await this.ocrDispatcher?.recoverExpiredLeases() ?? 0; await this.ocrDispatcher?.dispatchAvailable(); + const ocrCandidatesRecovered = this.ocrDispatcher?.recoverCompletedCandidates + ? await this.ocrDispatcher.recoverCompletedCandidates() + : 0; const retention = await this.ocrRetention?.runOnce(); const orphanRecovery = await this.recoverOrphanedVersions(catalog); @@ -82,6 +85,7 @@ export class KnowledgeLifecycleReconciler { orphanedVersionsRecovered: orphanRecovery.ready, orphanedVersionsFailed: orphanRecovery.failed, ocrLeasesRecovered, + ocrCandidatesRecovered, ocrRetentionDeleted: (retention?.deleted ?? 0) + (retention?.resumed ?? 0), invariantViolations }; diff --git a/src/modules/catalog/repository.ts b/src/modules/catalog/repository.ts index 831c334..1a84f11 100644 --- a/src/modules/catalog/repository.ts +++ b/src/modules/catalog/repository.ts @@ -1,11 +1,12 @@ import type { AvailableScope, ChunkMode, IngestSourceInput, SourceType, SourceVersionState, RetrieveScope } from "../../shared/types/rag.js"; -import { sha256Hex } from "../../shared/utils/ids.js"; +import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js"; import { withCatalogAdvisoryLock, withCatalogAdvisorySharedLocks, withCatalogTryAdvisoryLock, withTransaction, type PgPool, type PgPoolClient } from "./client.js"; import { CatalogError } from "./errors.js"; import { normalizeExpectedActiveVersion } from "./lifecycle.js"; import type { OcrResult } from "../ocr/client.js"; import type { OcrRetentionCandidate } from "../ocr/retention.js"; import type { OcrReviewContext } from "../ocr/review.js"; +import type { OcrBlockingReason, OcrQualityOutcome, OcrQualityWarning } from "../ocr/detection.js"; export interface CatalogDocumentInput { documentId: string; @@ -31,7 +32,7 @@ export interface OcrJobInput { documentId: string; remoteIdempotencyKey: string; requestedPages: number[]; - configVersion: "ocr-v1"; + configVersion: "ocr-v2"; } export interface CreateVersionInput { @@ -406,10 +407,9 @@ export class CatalogRepository { async listOcrVersionsAwaitingCandidate(): Promise { const result = await this.pool.query<{ version_id: string }>( `SELECT DISTINCT v.version_id FROM rag_source_versions v - WHERE v.state = 'indexing' AND EXISTS (SELECT 1 FROM rag_ocr_jobs j WHERE j.version_id = v.version_id) - AND NOT EXISTS (SELECT 1 FROM rag_ocr_jobs j WHERE j.version_id = v.version_id AND j.state <> 'succeeded') - AND EXISTS (SELECT 1 FROM rag_document_pages p WHERE p.version_id = v.version_id AND p.candidate_text_hash IS NULL) - ORDER BY v.version_id` + WHERE v.state = 'indexing' AND EXISTS (SELECT 1 FROM rag_ocr_jobs j WHERE j.version_id = v.version_id) + AND NOT EXISTS (SELECT 1 FROM rag_ocr_jobs j WHERE j.version_id = v.version_id AND j.state <> 'succeeded') + ORDER BY v.version_id` ); return result.rows.map(({ version_id }) => version_id); } @@ -466,20 +466,32 @@ export class CatalogRepository { async persistOcrCandidate(versionId: string, pages: Array<{ documentId: string; page: number; method: "native" | "ocr" | "blank"; nativeTextSha256: string; ocrTextSha256: string | null; candidateTextSha256: string; metrics: Record; risks: string[]; + qualityOutcome: OcrQualityOutcome; warnings: OcrQualityWarning[]; blockingReasons: OcrBlockingReason[]; + primaryBlockingReason: OcrBlockingReason | null; qualityReportSha256: string; }>): Promise { + if (pages.length === 0 || new Set(pages.map(({ qualityReportSha256 }) => qualityReportSha256)).size !== 1) { + throw new CatalogError("OCR quality report identity is invalid", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); + } + for (const page of pages) assertQualityDiagnostic(page); await withTransaction(this.pool, async (client) => { for (const page of pages) { const result = await client.query( - `UPDATE rag_document_pages SET extraction_method = $4, candidate_text_hash = $5, risk_tokens = $6::jsonb - WHERE version_id = $1 AND document_id = $2 AND page_number = $3 AND extraction_method = $7 - AND native_text_hash = $8 AND ocr_text_hash IS NOT DISTINCT FROM $9::char(64) - AND metrics = $10::jsonb AND blocked_reason IS NULL`, + `UPDATE rag_document_pages SET extraction_method = $4, + candidate_text_hash = CASE WHEN $11 = 'blocked' THEN NULL ELSE $5 END, risk_tokens = $6::jsonb, + quality_outcome = $11, quality_warnings = $12::jsonb, blocking_reasons = $13::jsonb, + primary_blocking_reason = $14, quality_report_sha256 = $15 + WHERE version_id = $1 AND document_id = $2 AND page_number = $3 AND extraction_method = $7 + AND native_text_hash = $8 AND ocr_text_hash IS NOT DISTINCT FROM $9::char(64) + AND metrics = $10::jsonb AND blocked_reason IS NULL`, [versionId, page.documentId, page.page, page.method, page.candidateTextSha256, JSON.stringify(page.risks), - page.ocrTextSha256 === null ? "native" : "ocr", page.nativeTextSha256, page.ocrTextSha256, JSON.stringify(page.metrics)] + page.ocrTextSha256 === null ? "native" : "ocr", page.nativeTextSha256, page.ocrTextSha256, JSON.stringify(page.metrics), + page.qualityOutcome, JSON.stringify(page.warnings), JSON.stringify(page.blockingReasons), page.primaryBlockingReason, + page.qualityReportSha256] ); if (result.rowCount !== 1) throw new CatalogError("OCR candidate lifecycle evidence does not match durable artifacts", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); } }); + if (pages.some(({ qualityOutcome }) => qualityOutcome === "blocked")) throw new Error("OCR_QUALITY_BLOCKED"); } async failOcrJob(jobId: string, code: string, detail: string): Promise { @@ -516,12 +528,20 @@ export class CatalogRepository { page_number: string | number; extraction_method: "native" | "ocr" | "blank"; blocked_reason: string | null; + quality_outcome: OcrQualityOutcome | null; + quality_warnings: OcrQualityWarning[]; + blocking_reasons: OcrBlockingReason[]; + primary_blocking_reason: OcrBlockingReason | null; + quality_report_sha256: string | null; }>( - "SELECT document_id, page_number, extraction_method, blocked_reason FROM rag_document_pages WHERE version_id = $1 ORDER BY document_id, page_number", + `SELECT document_id, page_number, extraction_method, blocked_reason, quality_outcome, quality_warnings, + blocking_reasons, primary_blocking_reason, quality_report_sha256 + FROM rag_document_pages WHERE version_id = $1 ORDER BY document_id, page_number`, [versionId] ); const row = version.rows[0]; const phase = ingestionPhase(row.state, documents.rows.map(({ job_state }) => job_state)); + const qualityError = row.error_code === "OCR_QUALITY_BLOCKED" ? parseQualityBlockedDetail(row.error_detail) : undefined; return { sourceId: row.source_id, versionId, @@ -548,7 +568,7 @@ export class CatalogRepository { pages: documentPages.map((page) => ({ page: Number(page.page_number), method: page.extraction_method, - state: page.blocked_reason + state: page.quality_outcome === "blocked" || page.blocked_reason ? "failed" : page.extraction_method === "native" ? "native_complete" @@ -559,11 +579,21 @@ export class CatalogRepository { : document.job_state === "running" ? "ocr_running" : "ocr_queued", + qualityOutcome: page.quality_outcome, + warnings: page.quality_warnings ?? [], + blockingReasons: page.blocking_reasons ?? [], + primaryBlockingReason: page.primary_blocking_reason, + ...(page.quality_report_sha256 ? { qualityReportSha256: page.quality_report_sha256 } : {}), ...(page.blocked_reason ? { errorCode: page.blocked_reason } : {}) })) }; }), - error: row.error_code ? { code: row.error_code, message: row.error_detail ?? row.error_code, retryable: false } : null, + error: row.error_code ? { + code: row.error_code, + message: row.error_code === "OCR_QUALITY_BLOCKED" ? "OCR quality validation blocked one or more pages" : row.error_detail ?? row.error_code, + retryable: false, + ...(qualityError ? { details: qualityError } : {}) + } : null, statusUrl: `/ingestions/${versionId}`, reviewUrl: row.state === "review_required" ? `/ingestions/${versionId}/review` : null }; @@ -580,15 +610,28 @@ export class CatalogRepository { const pages = await this.pool.query<{ document_id: string; page_number: string | number; native_text_hash: string; ocr_text_hash: string | null; candidate_text_hash: string | null; metrics: Record; risk_tokens: string[]; - }>(`SELECT document_id, page_number, native_text_hash, ocr_text_hash, candidate_text_hash, metrics, risk_tokens - FROM rag_document_pages WHERE version_id = $1 ORDER BY document_id, page_number`, [versionId]); - if (pages.rows.some(({ candidate_text_hash }) => !candidate_text_hash)) throw new CatalogError("OCR candidate lifecycle evidence is incomplete", 409, "OCR_REVIEW_EVIDENCE_INCOMPLETE"); + quality_outcome: OcrQualityOutcome | null; quality_warnings: OcrQualityWarning[]; + blocking_reasons: OcrBlockingReason[]; primary_blocking_reason: OcrBlockingReason | null; quality_report_sha256: string | null; + }>(`SELECT document_id, page_number, native_text_hash, ocr_text_hash, candidate_text_hash, metrics, risk_tokens, + quality_outcome, quality_warnings, blocking_reasons, primary_blocking_reason, quality_report_sha256 + FROM rag_document_pages WHERE version_id = $1 + ORDER BY CASE WHEN quality_outcome = 'warning' THEN 0 ELSE 1 END, + CASE WHEN jsonb_array_length(risk_tokens) > 0 THEN 0 ELSE 1 END, document_id, page_number`, [versionId]); + if (pages.rows.some(({ candidate_text_hash, quality_outcome, blocking_reasons, quality_report_sha256 }) => !candidate_text_hash + || !quality_outcome || blocking_reasons.length > 0 || !quality_report_sha256)) { + throw new CatalogError("OCR candidate lifecycle evidence is incomplete", 409, "OCR_REVIEW_EVIDENCE_INCOMPLETE"); + } + if (new Set(pages.rows.map(({ quality_report_sha256 }) => quality_report_sha256)).size !== 1) { + throw new CatalogError("OCR candidate lifecycle evidence is incomplete", 409, "OCR_REVIEW_EVIDENCE_INCOMPLETE"); + } const row = version.rows[0]; return { versionId: row.version_id, sourceId: row.source_id, state: row.state, baseActiveVersionId: row.base_active_version_id, currentActiveVersionId: row.current_active_version_id, activateRequested: row.activate_requested, processingFingerprint: row.processing_fingerprint, metadataHash: row.metadata_hash, pages: pages.rows.map((page) => ({ documentId: page.document_id, page: Number(page.page_number), nativeTextSha256: page.native_text_hash, - ocrTextSha256: page.ocr_text_hash, candidateTextSha256: page.candidate_text_hash!, metrics: page.metrics, risks: page.risk_tokens })) }; + ocrTextSha256: page.ocr_text_hash, candidateTextSha256: page.candidate_text_hash!, metrics: page.metrics, risks: page.risk_tokens, + qualityOutcome: page.quality_outcome!, warnings: page.quality_warnings, blockingReasons: page.blocking_reasons, + primaryBlockingReason: page.primary_blocking_reason, qualityReportSha256: page.quality_report_sha256! })) }; } async markReviewRequired(versionId: string): Promise { @@ -605,7 +648,9 @@ export class CatalogRepository { WHERE j.version_id = d.version_id AND j.document_id = d.document_id AND j.state = 'succeeded' ) ) - AND NOT EXISTS (SELECT 1 FROM rag_document_pages WHERE version_id = $1 AND (candidate_text_hash IS NULL OR blocked_reason IS NOT NULL))`, + AND NOT EXISTS (SELECT 1 FROM rag_document_pages WHERE version_id = $1 + AND (candidate_text_hash IS NULL OR quality_outcome IS NULL OR quality_outcome = 'blocked' + OR jsonb_array_length(blocking_reasons) > 0 OR primary_blocking_reason IS NOT NULL OR quality_report_sha256 IS NULL))`, [versionId] ); if (result.rowCount !== 1) throw new CatalogError("Version is not ready for OCR review", 409, "INVALID_VERSION_STATE"); @@ -763,6 +808,10 @@ export class CatalogRepository { if (!versionId) { return; } + if (code === "OCR_QUALITY_BLOCKED") { + await this.markQualityBlocked(versionId); + return; + } const result = await this.pool.query( `UPDATE rag_source_versions SET state = 'failed', error_code = $2, error_detail = $3, retention_due_at = now() + interval '7 days' @@ -774,6 +823,38 @@ export class CatalogRepository { } } + private async markQualityBlocked(versionId: string): Promise { + await withTransaction(this.pool, async (client) => { + const diagnostics = await client.query<{ + document_id: string; page_number: string | number; blocking_reasons: OcrBlockingReason[]; + primary_blocking_reason: OcrBlockingReason | null; quality_report_sha256: string | null; + }>(`SELECT document_id, page_number, blocking_reasons, primary_blocking_reason, quality_report_sha256 + FROM rag_document_pages WHERE version_id = $1 AND quality_outcome = 'blocked' + ORDER BY document_id, page_number FOR UPDATE`, [versionId]); + if (!diagnostics.rowCount) throw new CatalogError("OCR blocking diagnostics are incomplete", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); + const qualityReportSha256 = diagnostics.rows[0]!.quality_report_sha256; + if (!qualityReportSha256 || !/^[0-9a-f]{64}$/u.test(qualityReportSha256) + || diagnostics.rows.some((row) => row.quality_report_sha256 !== qualityReportSha256 + || row.blocking_reasons.length === 0 || row.primary_blocking_reason !== row.blocking_reasons[0])) { + throw new CatalogError("OCR blocking diagnostics are incomplete", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); + } + const errorDetail = canonicalJson({ qualityReportSha256, blockedPages: diagnostics.rows.map((row) => ({ + documentId: row.document_id, page: Number(row.page_number), reasons: row.blocking_reasons, + primaryBlockingReason: row.primary_blocking_reason + })) }); + if (diagnostics.rows.length > 1000 || errorDetail.length > 262_144) { + throw new CatalogError("OCR blocking diagnostics exceed the safe status bound", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); + } + const result = await client.query( + `UPDATE rag_source_versions SET state = 'failed', error_code = 'OCR_QUALITY_BLOCKED', error_detail = $2, + retention_due_at = now() + interval '7 days' WHERE version_id = $1 AND state IN ('pending', 'indexing')`, + [versionId, errorDetail] + ); + if (result.rowCount === 1) await client.query( + "UPDATE rag_version_documents SET index_state = 'failed', error_detail = 'OCR_QUALITY_BLOCKED' WHERE version_id = $1", [versionId]); + }); + } + async activateVersion(sourceId: string, versionId: string, expectedActiveVersionId: string | null | undefined): Promise { const expected = normalizeExpectedActiveVersion(expectedActiveVersionId, true); return withTransaction(this.pool, async (client) => { @@ -1053,9 +1134,12 @@ export class CatalogRepository { }>( `SELECT version_id, source_id, expected_document_count, expected_point_count, embedding_dimensions, qdrant_collection FROM rag_source_versions - WHERE ( - state = 'pending' - AND created_at < now() - ($1::text || ' milliseconds')::interval + WHERE NOT EXISTS ( + SELECT 1 FROM rag_ocr_jobs j + WHERE j.version_id = rag_source_versions.version_id AND j.state IN ('queued', 'running') + ) AND ( + state = 'pending' + AND created_at < now() - ($1::text || ' milliseconds')::interval ) OR ( state = 'indexing' AND indexing_started_at IS NOT NULL @@ -1115,4 +1199,48 @@ export class CatalogRepository { return recovered; } } + +function assertQualityDiagnostic(page: { + qualityOutcome: OcrQualityOutcome; warnings: OcrQualityWarning[]; blockingReasons: OcrBlockingReason[]; + primaryBlockingReason: OcrBlockingReason | null; qualityReportSha256: string; +}): void { + const canonicalReasons: OcrBlockingReason[] = ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE", "EXCESS_LOW_CONFIDENCE_LINES"]; + const reasonIndexes = page.blockingReasons.map((reason) => canonicalReasons.indexOf(reason)); + const orderedReasons = reasonIndexes.every((index, position) => index >= 0 && (position === 0 || index > reasonIndexes[position - 1]!)); + const validHash = /^[0-9a-f]{64}$/u.test(page.qualityReportSha256); + const valid = validHash && orderedReasons + && (page.qualityOutcome === "blocked" + ? page.blockingReasons.length > 0 && page.primaryBlockingReason === page.blockingReasons[0] && page.warnings.length === 0 + : page.blockingReasons.length === 0 && page.primaryBlockingReason === null + && (page.qualityOutcome === "warning" ? page.warnings.length === 1 && page.warnings[0] === "LOW_P10_CONFIDENCE" : page.warnings.length === 0)); + if (!valid) throw new CatalogError("OCR quality diagnostics are invalid", 409, "OCR_ARTIFACT_INTEGRITY_FAILED"); +} + +function parseQualityBlockedDetail(detail: string | null): { + qualityReportSha256: string; + blockedPages: Array<{ documentId: string; page: number; reasons: OcrBlockingReason[]; primaryBlockingReason: OcrBlockingReason }>; +} | undefined { + if (!detail || detail.length > 262_144) return undefined; + let value: unknown; + try { value = JSON.parse(detail); } catch { return undefined; } + if (!value || typeof value !== "object" || Array.isArray(value)) return undefined; + const record = value as Record; + if (!/^[0-9a-f]{64}$/u.test(String(record.qualityReportSha256)) || !Array.isArray(record.blockedPages) + || record.blockedPages.length === 0 || record.blockedPages.length > 1000) return undefined; + const blockedPages = []; + for (const raw of record.blockedPages) { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return undefined; + const page = raw as Record; + if (typeof page.documentId !== "string" || page.documentId.length > 512 || !Number.isInteger(page.page) + || !Array.isArray(page.reasons) || typeof page.primaryBlockingReason !== "string") return undefined; + const reasons = page.reasons as OcrBlockingReason[]; + try { + assertQualityDiagnostic({ qualityOutcome: "blocked", warnings: [], blockingReasons: reasons, + primaryBlockingReason: page.primaryBlockingReason as OcrBlockingReason, qualityReportSha256: String(record.qualityReportSha256) }); + } catch { return undefined; } + blockedPages.push({ documentId: page.documentId, page: Number(page.page), reasons, + primaryBlockingReason: page.primaryBlockingReason as OcrBlockingReason }); + } + return blockedPages.length > 0 ? { qualityReportSha256: String(record.qualityReportSha256), blockedPages } : undefined; +} export { CatalogError }; diff --git a/src/modules/ingest/service.ts b/src/modules/ingest/service.ts index 4e23abf..756d61a 100644 --- a/src/modules/ingest/service.ts +++ b/src/modules/ingest/service.ts @@ -23,6 +23,7 @@ import { listFilesRecursively } from "../../shared/utils/files.js"; import { env } from "../../config/env.js"; import { CatalogError, type CatalogDocumentInput, type CatalogRepository } from "../catalog/repository.js"; import { DETECTION_POLICY_VERSION, selectPdfPagesForOcr } from "../ocr/detection.js"; +import { OCR_CONFIG, buildOcrIdempotencyKey } from "../ocr/client.js"; import { stageOcrArtifacts } from "../ocr/artifacts.js"; type PreparedDocument = CatalogDocumentInput & { @@ -387,7 +388,7 @@ export class IngestService { catalog: CatalogRepository ): Promise { const processingFingerprint = buildProcessingFingerprint({ - parserVersion: "native-pages-v1+ocr-v1", + parserVersion: "native-pages-v1+ocr-v2", detectionPolicyVersion: DETECTION_POLICY_VERSION, normalizationPolicy: "bom-crlf-trim-final-lf-v1", chunking: { code: codeChunkingPolicy, documental: documentalChunkingPolicy }, @@ -457,9 +458,9 @@ export class IngestService { }))), ocrJobs: plans.filter(({ requestedPages }) => requestedPages.length > 0).map(({ original, requestedPages }) => ({ documentId: original.documentId, - remoteIdempotencyKey: `${original.originalHash}:ocr-v1:${sha256Hex(JSON.stringify(requestedPages))}`, + remoteIdempotencyKey: buildOcrIdempotencyKey({ documentSha256: original.originalHash, pages: requestedPages }), requestedPages, - configVersion: "ocr-v1" + configVersion: OCR_CONFIG.configVersion })) }); } catch (error) { diff --git a/src/modules/ocr/artifacts.ts b/src/modules/ocr/artifacts.ts index bb23860..1e7d042 100644 --- a/src/modules/ocr/artifacts.ts +++ b/src/modules/ocr/artifacts.ts @@ -4,7 +4,8 @@ import path from "node:path"; import { canonicalJson, hashOrderedPairs, sha256Hex } from "../../shared/utils/ids.js"; import { isValidOcrResult, type OcrResult } from "./client.js"; import { composeCandidate, type CandidatePage } from "./composition.js"; -import { classifyOcrPage } from "./detection.js"; +import { classifyOcrPage, computeNativeMetrics, type OcrBlockingReason, type OcrExtractionMethod, + type OcrPageClassification, type OcrQualityOutcome, type OcrQualityWarning } from "./detection.js"; export class OcrArtifactError extends Error { constructor( @@ -118,7 +119,7 @@ export async function persistOcrResultArtifact(input: OcrResultArtifactInput): P }> { assertUuid(input.versionId); const pages = input.result.pages.map(({ page }) => page); - if (!isValidOcrResult(input.result, input.result.jobId, { documentSha256: input.result.documentSha256, pages })) { + if (!isValidOcrResult(input.result, input.result.jobId, input.result)) { throw resultIntegrityError(); } const resultSha256 = sha256Hex(canonicalJson(input.result)); @@ -156,8 +157,9 @@ export async function readOcrResultArtifact(input: OcrResultReadInput): Promise< if (!isRecord(envelope) || envelope.schemaVersion !== "1" || envelope.versionId !== input.versionId || envelope.documentId !== input.documentId) { throw new Error("OCR result artifact identity validation failed"); } - if (typeof envelope.resultSha256 !== "string" || sha256Hex(canonicalJson(envelope.result)) !== envelope.resultSha256 - || !isValidOcrResult(envelope.result, input.jobId, { documentSha256: input.documentSha256, pages: input.pages })) { + if (typeof envelope.resultSha256 !== "string" || !isRecord(envelope.result) + || sha256Hex(canonicalJson(envelope.result)) !== envelope.resultSha256 + || !isValidOcrResult(envelope.result, input.jobId, envelope.result as unknown as OcrResult)) { throw resultIntegrityError(); } return envelope.result; @@ -167,15 +169,33 @@ export interface CandidateArtifactPage extends CandidatePage { nativeTextSha256: string; ocrTextSha256: string | null; metrics: Record; + qualityReportSha256: string; } export interface ComposedCandidateArtifact { schemaVersion: "1"; versionId: string; + qualityReportSha256: string; candidateSha256: string; documents: Array<{ documentId: string; text: string; textSha256: string; pages: CandidateArtifactPage[] }>; } +export interface OcrQualityReportPage extends OcrPageClassification { + documentId: string; + page: number; + metrics: Record; +} + +export interface OcrQualityReport { + schemaVersion: "1"; + versionId: string; + configVersion: string; + sourceManifestSha256: string; + ocrResults: Array<{ documentId: string; artifactSha256: string; resultSha256: string }>; + reviewImageManifests: Array<{ documentId: string; manifestSha256: string }>; + pages: OcrQualityReportPage[]; +} + interface CandidateJobEvidence { documentId: string; remoteJobId: string | null; @@ -187,14 +207,20 @@ export async function persistComposedCandidateArtifact(input: { rootDirectory: string; versionId: string; jobs: CandidateJobEvidence[]; -}): Promise<{ artifactPath: string; artifactSha256: string; candidate: ComposedCandidateArtifact }> { +}): Promise<{ artifactPath: string | null; artifactSha256: string | null; qualityReportSha256: string; candidate: ComposedCandidateArtifact }> { assertUuid(input.versionId); const versionDirectory = path.join(path.resolve(input.rootDirectory), input.versionId); - const manifest = await readPrivateJson(path.join(versionDirectory, "manifest.json"), undefined, "OCR manifest integrity validation failed"); + const manifestPath = path.join(versionDirectory, "manifest.json"); + const manifest = await readPrivateJson(manifestPath, undefined, "OCR manifest integrity validation failed"); if (!isRecord(manifest) || manifest.schemaVersion !== "1" || manifest.versionId !== input.versionId || !Array.isArray(manifest.documents)) { throw new Error("OCR manifest identity validation failed"); } + const sourceManifestSha256 = sha256Hex(await readFile(manifestPath)); const documents = []; + const reportPages: OcrQualityReportPage[] = []; + const ocrResults: OcrQualityReport["ocrResults"] = []; + const reviewImageManifests: OcrQualityReport["reviewImageManifests"] = []; + const configVersions = new Set(); for (const entry of manifest.documents) { if (!isManifestDocument(entry)) throw new Error("OCR manifest identity validation failed"); const native = await readPrivateJson(resolveArtifactPath(versionDirectory, entry.nativePagesPath), entry.nativePagesSha256, "OCR native page artifact integrity validation failed"); @@ -206,14 +232,31 @@ export async function persistComposedCandidateArtifact(input: { if (!job || job.state !== "succeeded" || !job.remoteJobId || !sameNumbers(job.requestedPages, requestedPages)) throw new Error("OCR result artifact identity validation failed"); result = await readOcrResultArtifact({ rootDirectory: input.rootDirectory, versionId: input.versionId, documentId: entry.documentId, jobId: job.remoteJobId, documentSha256: entry.originalSha256, pages: requestedPages }); + configVersions.add(result.configVersion); + const resultPath = resultArtifactPath(input.rootDirectory, input.versionId, entry.documentId); + const resultBytes = await readFile(resultPath); + const resultEnvelope = await readPrivateJson(resultPath, undefined, "OCR result artifact integrity validation failed"); + if (!isRecord(resultEnvelope) || typeof resultEnvelope.resultSha256 !== "string") throw resultIntegrityError(); + ocrResults.push({ documentId: entry.documentId, artifactSha256: sha256Hex(resultBytes), resultSha256: resultEnvelope.resultSha256 }); + const imageManifestPath = resolveArtifactPath(versionDirectory, + path.posix.join("documents", entry.documentArtifactId, "review-images", "manifest.json")); + for (const page of requestedPages) await readReviewImageArtifact({ rootDirectory: input.rootDirectory, + versionId: input.versionId, documentId: entry.documentId, page }); + reviewImageManifests.push({ documentId: entry.documentId, manifestSha256: sha256Hex(await readFile(imageManifestPath)) }); } else if (job) throw new Error("OCR result artifact identity validation failed"); const resultPages = new Map(result?.pages.map((page) => [page.page, page])); const evidence = (native.pages as Array<{ page: number; text: string; textSha256: string }>).map((page) => { const ocr = resultPages.get(page.page); - if (!ocr) return { page: page.page, method: "native" as const, nativeText: page.text, rawOcrText: "", lines: [] }; - const classification = classifyOcrPage({ inkCoverage: ocr.metrics.inkCoverage, metrics: ocr.metrics }); - if (classification.method === "blocked") throw new Error(classification.errorCode); - return { page: page.page, method: classification.method, nativeText: page.text, rawOcrText: ocr.text, lines: ocr.lines }; + const classification: OcrPageClassification = ocr + ? classifyOcrPage({ inkCoverage: ocr.metrics.inkCoverage, metrics: ocr.metrics }) + : { extractionMethod: "native", qualityOutcome: "accepted", warnings: [], blockingReasons: [], primaryBlockingReason: null }; + const metrics: Record = ocr ? { nonWhitespaceCharacters: ocr.metrics.nonWhitespaceCharacters, + medianConfidence: ocr.metrics.medianConfidence, p10Confidence: ocr.metrics.p10Confidence, + lowConfidenceLineRatio: ocr.metrics.lowConfidenceLineRatio, inkCoverage: ocr.metrics.inkCoverage } + : { nonWhitespaceCharacters: computeNativeMetrics(page.text).nonWhitespaceCharacters }; + reportPages.push({ documentId: entry.documentId, page: page.page, ...classification, metrics }); + return { page: page.page, method: classification.extractionMethod, nativeText: page.text, + rawOcrText: ocr?.text ?? "", lines: ocr?.lines ?? [], ...classification }; }); const composed = composeCandidate(evidence); documents.push({ documentId: entry.documentId, text: composed.text, textSha256: composed.textSha256, @@ -221,13 +264,27 @@ export async function persistComposedCandidateArtifact(input: { nativeTextSha256: sha256Hex(page.nativeText), ocrTextSha256: ocr ? sha256Hex(ocr.text) : null, metrics: ocr?.metrics ?? {} }; }) }); } - const candidate: ComposedCandidateArtifact = { schemaVersion: "1", versionId: input.versionId, - candidateSha256: sha256Hex(canonicalJson(documents)), documents }; + if (configVersions.size !== 1) throw new Error("OCR quality report config identity validation failed"); + const qualityReport: OcrQualityReport = { schemaVersion: "1", versionId: input.versionId, + configVersion: [...configVersions][0]!, sourceManifestSha256, ocrResults, reviewImageManifests, pages: reportPages }; + const qualityBytes = Buffer.from(canonicalJson(qualityReport)); + const qualityReportPath = path.join(versionDirectory, "quality-report.json"); + await publishImmutable(qualityReportPath, qualityBytes); + const qualityReportSha256 = sha256Hex(qualityBytes); + await readOcrQualityReportArtifact({ rootDirectory: input.rootDirectory, versionId: input.versionId, artifactSha256: qualityReportSha256 }); + const linkedDocuments = documents.map((document) => ({ ...document, + pages: document.pages.map((page) => ({ ...page, qualityReportSha256 })) })); + const candidate: ComposedCandidateArtifact = { schemaVersion: "1", versionId: input.versionId, qualityReportSha256, + candidateSha256: sha256Hex(canonicalJson({ qualityReportSha256, documents: linkedDocuments })), documents: linkedDocuments }; + if (reportPages.some(({ qualityOutcome }) => qualityOutcome === "blocked")) { + return { artifactPath: null, artifactSha256: null, qualityReportSha256, candidate }; + } const serialized = canonicalJson(candidate); const artifactPath = path.join(versionDirectory, "candidate-pages.json"); await publishImmutable(artifactPath, Buffer.from(serialized)); const artifactSha256 = sha256Hex(serialized); - return { artifactPath, artifactSha256, candidate: await readComposedCandidateArtifact({ ...input, artifactSha256 }) }; + return { artifactPath, artifactSha256, qualityReportSha256, + candidate: await readComposedCandidateArtifact({ rootDirectory: input.rootDirectory, versionId: input.versionId, artifactSha256 }) }; } export async function readComposedCandidateArtifact(input: { rootDirectory: string; versionId: string; artifactSha256?: string }): Promise { @@ -242,10 +299,63 @@ export async function readComposedCandidateArtifact(input: { rootDirectory: stri throw new OcrArtifactError("OCR candidate artifact integrity validation failed", 422, "OCR_ARTIFACT_INTEGRITY_FAILED"); } if (!isRecord(value) || value.schemaVersion !== "1" || value.versionId !== input.versionId || !Array.isArray(value.documents) - || value.candidateSha256 !== sha256Hex(canonicalJson(value.documents))) throw new OcrArtifactError("OCR candidate artifact integrity validation failed", 422, "OCR_ARTIFACT_INTEGRITY_FAILED"); + || typeof value.qualityReportSha256 !== "string" + || value.candidateSha256 !== sha256Hex(canonicalJson({ qualityReportSha256: value.qualityReportSha256, documents: value.documents }))) { + throw new OcrArtifactError("OCR candidate artifact integrity validation failed", 422, "OCR_ARTIFACT_INTEGRITY_FAILED"); + } + await readOcrQualityReportArtifact({ rootDirectory: input.rootDirectory, versionId: input.versionId, + artifactSha256: value.qualityReportSha256 }); return value as unknown as ComposedCandidateArtifact; } +export async function readOcrQualityReportArtifact(input: { + rootDirectory: string; versionId: string; artifactSha256?: string; +}): Promise { + assertUuid(input.versionId); + const versionDirectory = path.join(path.resolve(input.rootDirectory), input.versionId); + const reportPath = path.join(versionDirectory, "quality-report.json"); + const value = await readPrivateJson(reportPath, input.artifactSha256, "OCR quality report integrity validation failed"); + validateQualityReportShape(value, input.versionId); + const manifestPath = path.join(versionDirectory, "manifest.json"); + if (sha256Hex(await readFile(manifestPath)) !== value.sourceManifestSha256) throw qualityReportIntegrityError(); + const manifest = await readPrivateJson(manifestPath, value.sourceManifestSha256, "OCR quality report integrity validation failed"); + if (!isRecord(manifest) || !Array.isArray(manifest.documents)) throw qualityReportIntegrityError(); + const expectedPages: Array<{ documentId: string; page: number }> = []; + const expectedOcrDocumentIds: string[] = []; + for (const rawEntry of manifest.documents) { + if (!isManifestDocument(rawEntry) || !isRecord(rawEntry) || typeof rawEntry.documentArtifactId !== "string") throw qualityReportIntegrityError(); + const entry = rawEntry as { documentId: string; documentArtifactId: string; originalSha256: string; nativePagesPath: string; nativePagesSha256: string }; + const native = await readPrivateJson(resolveArtifactPath(versionDirectory, entry.nativePagesPath), entry.nativePagesSha256, + "OCR quality report integrity validation failed"); + validateNativePages(native, input.versionId, entry.documentId, entry.originalSha256); + expectedPages.push(...(native.pages as Array<{ page: number }>).map(({ page }) => ({ documentId: entry.documentId, page }))); + const resultEvidence = value.ocrResults.find(({ documentId }) => documentId === entry.documentId); + const requestedPages = native.requestedPages as number[]; + if (requestedPages.length > 0) { + expectedOcrDocumentIds.push(entry.documentId); + if (!resultEvidence) throw qualityReportIntegrityError(); + const resultPath = resultArtifactPath(input.rootDirectory, input.versionId, entry.documentId); + const resultBytes = await readFile(resultPath); + const envelope = await readPrivateJson(resultPath, resultEvidence.artifactSha256, "OCR quality report integrity validation failed"); + if (sha256Hex(resultBytes) !== resultEvidence.artifactSha256 || !isRecord(envelope) + || envelope.resultSha256 !== resultEvidence.resultSha256 || !isRecord(envelope.result) + || sha256Hex(canonicalJson(envelope.result)) !== resultEvidence.resultSha256 + || envelope.result.documentSha256 !== entry.originalSha256 + || !Array.isArray(envelope.result.requestedPages) + || !sameNumbers(envelope.result.requestedPages as number[], requestedPages) + || envelope.result.configVersion !== value.configVersion) throw qualityReportIntegrityError(); + const imageEvidence = value.reviewImageManifests.find(({ documentId }) => documentId === entry.documentId); + const imageManifestPath = resolveArtifactPath(versionDirectory, + path.posix.join("documents", entry.documentArtifactId, "review-images", "manifest.json")); + if (!imageEvidence || sha256Hex(await readFile(imageManifestPath)) !== imageEvidence.manifestSha256) throw qualityReportIntegrityError(); + } else if (resultEvidence || value.reviewImageManifests.some(({ documentId }) => documentId === entry.documentId)) throw qualityReportIntegrityError(); + } + if (canonicalJson(value.ocrResults.map(({ documentId }) => documentId)) !== canonicalJson(expectedOcrDocumentIds) + || canonicalJson(value.reviewImageManifests.map(({ documentId }) => documentId)) !== canonicalJson(expectedOcrDocumentIds) + || canonicalJson(expectedPages) !== canonicalJson(value.pages.map(({ documentId, page }) => ({ documentId, page })))) throw qualityReportIntegrityError(); + return value; +} + export interface ReviewedPagesArtifact { schemaVersion: "1"; versionId: string; @@ -460,9 +570,73 @@ function isRecord(value: unknown): value is Record { return typeof value === "object" && value !== null && !Array.isArray(value); } -function isManifestDocument(value: unknown): value is { documentId: string; originalSha256: string; nativePagesPath: string; nativePagesSha256: string } { +function isManifestDocument(value: unknown): value is { documentId: string; documentArtifactId: string; originalSha256: string; nativePagesPath: string; nativePagesSha256: string } { return isRecord(value) && typeof value.documentId === "string" && typeof value.originalSha256 === "string" - && typeof value.nativePagesPath === "string" && typeof value.nativePagesSha256 === "string"; + && typeof value.documentArtifactId === "string" && typeof value.nativePagesPath === "string" && typeof value.nativePagesSha256 === "string"; +} + +function validateQualityReportShape(value: unknown, versionId: string): asserts value is OcrQualityReport { + if (!isRecord(value) || value.schemaVersion !== "1" || value.versionId !== versionId + || !hasExactKeys(value, ["schemaVersion", "versionId", "configVersion", "sourceManifestSha256", "ocrResults", "reviewImageManifests", "pages"]) + || typeof value.configVersion !== "string" || !/^[a-z0-9-]+$/u.test(value.configVersion) + || !isSha256(value.sourceManifestSha256) || !Array.isArray(value.ocrResults) + || !Array.isArray(value.reviewImageManifests) || !Array.isArray(value.pages)) throw qualityReportIntegrityError(); + const results = value.ocrResults as Array>; + const images = value.reviewImageManifests as Array>; + if (results.some((entry) => !isRecord(entry) || typeof entry.documentId !== "string" + || !hasExactKeys(entry, ["documentId", "artifactSha256", "resultSha256"]) + || !isSha256(entry.artifactSha256) || !isSha256(entry.resultSha256)) + || images.some((entry) => !isRecord(entry) || !hasExactKeys(entry, ["documentId", "manifestSha256"]) + || typeof entry.documentId !== "string" || !isSha256(entry.manifestSha256)) + || new Set(results.map(({ documentId }) => documentId)).size !== results.length + || new Set(images.map(({ documentId }) => documentId)).size !== images.length) throw qualityReportIntegrityError(); + const pages = value.pages as Array>; + const pageKeys = new Set(); + for (const page of pages) { + if (!isRecord(page) || typeof page.documentId !== "string" || !Number.isInteger(page.page) || Number(page.page) < 1 + || !hasExactKeys(page, ["documentId", "page", "extractionMethod", "qualityOutcome", "warnings", "blockingReasons", "primaryBlockingReason", "metrics"]) + || !["native", "ocr", "blank"].includes(String(page.extractionMethod)) + || !["accepted", "warning", "blocked"].includes(String(page.qualityOutcome)) + || !Array.isArray(page.warnings) || !Array.isArray(page.blockingReasons) || !isRecord(page.metrics) + || (page.primaryBlockingReason !== null && typeof page.primaryBlockingReason !== "string")) throw qualityReportIntegrityError(); + const key = `${page.documentId}:${page.page}`; + if (pageKeys.has(key)) throw qualityReportIntegrityError(); + pageKeys.add(key); + if (Object.values(page.metrics).some((metric) => typeof metric !== "number" || !Number.isFinite(metric))) throw qualityReportIntegrityError(); + const warnings = page.warnings as unknown[]; + const reasons = page.blockingReasons as unknown[]; + if (warnings.some((warning) => warning !== "LOW_P10_CONFIDENCE") + || reasons.some((reason) => !["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE", "EXCESS_LOW_CONFIDENCE_LINES"].includes(String(reason)))) { + throw qualityReportIntegrityError(); + } + if (page.extractionMethod === "native") { + if (page.qualityOutcome !== "accepted" || warnings.length || reasons.length || page.primaryBlockingReason !== null) throw qualityReportIntegrityError(); + continue; + } + const metrics = page.metrics as Record; + if (!["nonWhitespaceCharacters", "medianConfidence", "p10Confidence", "lowConfidenceLineRatio", "inkCoverage"] + .every((name) => typeof metrics[name] === "number")) throw qualityReportIntegrityError(); + const classification = classifyOcrPage({ inkCoverage: metrics.inkCoverage!, metrics: { + nonWhitespaceCharacters: metrics.nonWhitespaceCharacters!, medianConfidence: metrics.medianConfidence!, + p10Confidence: metrics.p10Confidence!, lowConfidenceLineRatio: metrics.lowConfidenceLineRatio! + } }); + if (canonicalJson({ extractionMethod: page.extractionMethod, qualityOutcome: page.qualityOutcome, warnings, + blockingReasons: reasons, primaryBlockingReason: page.primaryBlockingReason }) !== canonicalJson(classification)) throw qualityReportIntegrityError(); + } +} + +function isSha256(value: unknown): value is string { + return typeof value === "string" && /^[0-9a-f]{64}$/u.test(value); +} + +function hasExactKeys(value: Record, keys: string[]): boolean { + const actual = Object.keys(value).sort(); + const expected = [...keys].sort(); + return actual.length === expected.length && actual.every((key, index) => key === expected[index]); +} + +function qualityReportIntegrityError(): Error { + return new Error("OCR quality report integrity validation failed"); } async function readNativeDocument(input: { rootDirectory: string; versionId: string; documentId: string }) { diff --git a/src/modules/ocr/client.ts b/src/modules/ocr/client.ts index d6d49bd..2766433 100644 --- a/src/modules/ocr/client.ts +++ b/src/modules/ocr/client.ts @@ -1,39 +1,45 @@ import { sha256Hex } from "../../shared/utils/ids.js"; -const OCR_CONFIG = { +export const OCR_CONFIG = { languages: ["es", "en"], dpi: 200, engine: "paddleocr", engineVersion: "3.4.0", runtimeVersion: "3.2.2", - configVersion: "ocr-v1", + configVersion: "ocr-v2", returnLayout: true } as const; const TRANSIENT_STATUSES = new Set([502, 503]); const JOB_STATUSES = new Set(["queued", "running", "succeeded", "failed"]); -export interface OcrAck { - jobId: string; - status: "queued"; +export type OcrJobState = "queued" | "running" | "succeeded" | "failed"; + +export interface OcrIdentity { + idempotencyKey: string; documentSha256: string; requestedPages: number[]; - configVersion: "ocr-v1"; + configVersion: typeof OCR_CONFIG.configVersion; + requestIdentitySha256: string; +} + +export interface OcrAck extends OcrIdentity { + jobId: string; + status: OcrJobState; createdAt: string; } -export interface OcrJobStatus { +export interface OcrJobStatus extends OcrIdentity { jobId: string; - status: "queued" | "running" | "succeeded" | "failed"; + status: OcrJobState; completedPages: number; totalPages: number; error: { code: string; message: string } | null; } -export interface OcrResult { +export interface OcrResult extends OcrIdentity { schemaVersion: "1"; jobId: string; - documentSha256: string; - engine: { name: "paddleocr"; version: "3.4.0"; runtime: "paddlepaddle-3.2.2"; device: "cpu"; configVersion: "ocr-v1"; dpi: 200 }; + engine: { name: "paddleocr"; version: "3.4.0"; runtime: "paddlepaddle-3.2.2"; device: "cpu"; configVersion: typeof OCR_CONFIG.configVersion; dpi: 200 }; pages: Array<{ page: number; width: number; @@ -46,8 +52,8 @@ export interface OcrResult { } export class OcrClientError extends Error { - constructor(public readonly code: string, public readonly status: number | undefined, public readonly retryable: boolean) { - super(`OCR request failed: ${code}`); + constructor(public readonly code: string, public readonly status: number | undefined, public readonly retryable: boolean, message?: string) { + super(message ?? `OCR request failed: ${code}`); } } @@ -73,8 +79,9 @@ export class OcrClient { assertExpected(expected); if (sha256Hex(file) !== expected.documentSha256) throw integrityError(); const payload = { documentSha256: expected.documentSha256, pages: expected.pages, ...OCR_CONFIG }; - const idempotencyKey = persistedIdempotencyKey ?? `${expected.documentSha256}:ocr-v1:${sha256Hex(JSON.stringify(expected.pages))}`; + const idempotencyKey = persistedIdempotencyKey ?? buildOcrIdempotencyKey(expected); if (!idempotencyKey.trim() || /[\r\n]/u.test(idempotencyKey)) throw new TypeError("OCR idempotency key is invalid"); + const identity = buildOcrIdentity({ ...expected, idempotencyKey }); const value = await this.requestJson("/v1/jobs", () => { const form = new FormData(); form.set("file", new Blob([new Uint8Array(file)], { type: "application/pdf" }), "original.pdf"); @@ -83,18 +90,15 @@ export class OcrClient { }); if (!isObject(value) || value.jobId === undefined - || value.status !== "queued" - || value.documentSha256 !== expected.documentSha256 - || value.configVersion !== "ocr-v1" - || !sameNumbers(value.requestedPages, expected.pages) + || typeof value.status !== "string" || !JOB_STATUSES.has(value.status) || !isOcrIdentity(value, identity) || typeof value.createdAt !== "string" || Number.isNaN(Date.parse(value.createdAt))) throw integrityError(); return value as unknown as OcrAck; } - async getStatus(jobId: string): Promise { + async getStatus(jobId: string, identity: OcrIdentity): Promise { const value = await this.requestJson(`/v1/jobs/${encodeURIComponent(jobId)}`, () => ({ headers: this.headers() })); - if (!isObject(value) || value.jobId !== jobId || typeof value.status !== "string" || !JOB_STATUSES.has(value.status) + if (!isObject(value) || value.jobId !== jobId || typeof value.status !== "string" || !JOB_STATUSES.has(value.status) || !isOcrIdentity(value, identity) || !isCount(value.completedPages) || !isCount(value.totalPages) || value.completedPages > value.totalPages || !(value.error === null || (isObject(value.error) && typeof value.error.code === "string" && typeof value.error.message === "string"))) { throw integrityError(); @@ -102,25 +106,25 @@ export class OcrClient { return value as unknown as OcrJobStatus; } - async pollUntilTerminal(jobId: string): Promise { + async pollUntilTerminal(jobId: string, identity: OcrIdentity): Promise { let delay = 2_000; while (true) { - const status = await this.getStatus(jobId); + const status = await this.getStatus(jobId, identity); if (status.status === "succeeded" || status.status === "failed") return status; await this.sleep(delay); delay = Math.min(delay * 2, 15_000); } } - async getResult(jobId: string, expected: { documentSha256: string; pages: number[] }): Promise { - assertExpected(expected); + async getResult(jobId: string, identity: OcrIdentity): Promise { + assertExpected({ documentSha256: identity.documentSha256, pages: identity.requestedPages }); const value = await this.requestJson(`/v1/jobs/${encodeURIComponent(jobId)}/result`, () => ({ headers: this.headers() })); - if (!isValidOcrResult(value, jobId, expected)) throw integrityError(); + if (!isValidOcrResult(value, jobId, identity)) throw integrityError(); return value; } - async getReviewImage(jobId: string, page: number, documentSha256: string): Promise<{ bytes: Buffer; sha256: string }> { - if (!Number.isInteger(page) || page < 1 || !/^[a-f0-9]{64}$/u.test(documentSha256)) throw new TypeError("OCR review image identity is invalid"); + async getReviewImage(jobId: string, page: number, identity: OcrIdentity): Promise<{ bytes: Buffer; sha256: string }> { + if (!Number.isInteger(page) || page < 1 || !identity.requestedPages.includes(page)) throw new TypeError("OCR review image identity is invalid"); for (let attempt = 0; attempt < 3; attempt += 1) { let response: Response; try { @@ -137,7 +141,10 @@ export class OcrClient { const sha256 = sha256Hex(bytes); if (response.headers.get("content-type")?.split(";")[0] !== "image/png" || response.headers.get("content-length") !== String(bytes.length) - || response.headers.get("x-document-sha256") !== documentSha256 + || response.headers.get("x-ocr-job-id") !== jobId + || response.headers.get("x-ocr-identity-sha256") !== identity.requestIdentitySha256 + || response.headers.get("x-ocr-config-version") !== identity.configVersion + || response.headers.get("x-document-sha256") !== identity.documentSha256 || response.headers.get("x-content-sha256") !== sha256 || response.headers.get("x-page-number") !== String(page) || !bytes.subarray(0, 8).equals(Buffer.from("89504e470d0a1a0a", "hex"))) throw integrityError(); @@ -182,10 +189,22 @@ function assertExpected(expected: { documentSha256: string; pages: number[] }): } } -export function isValidOcrResult(value: unknown, jobId: string, expected: { documentSha256: string; pages: number[] }): value is OcrResult { - if (!isObject(value) || value.schemaVersion !== "1" || value.jobId !== jobId || value.documentSha256 !== expected.documentSha256 - || !isObject(value.engine) || canonicalEngine(value.engine) !== "paddleocr|3.4.0|paddlepaddle-3.2.2|cpu|ocr-v1|200" - || !Array.isArray(value.pages) || !sameNumbers(value.pages.map((page) => isObject(page) ? page.page : undefined), expected.pages)) return false; +export function buildOcrIdempotencyKey(expected: { documentSha256: string; pages: number[] }): string { + assertExpected(expected); + return `${expected.documentSha256}:${OCR_CONFIG.configVersion}:${sha256Hex(JSON.stringify(expected.pages))}`; +} + +export function buildOcrIdentity(input: { documentSha256: string; pages: number[]; idempotencyKey?: string }): OcrIdentity { + assertExpected(input); + const idempotencyKey = input.idempotencyKey ?? buildOcrIdempotencyKey(input); + const identity = { configVersion: OCR_CONFIG.configVersion, documentSha256: input.documentSha256, idempotencyKey, requestedPages: [...input.pages] }; + return { ...identity, requestIdentitySha256: sha256Hex(canonicalIdentity(identity)) }; +} + +export function isValidOcrResult(value: unknown, jobId: string, identity: Pick): value is OcrResult { + if (!isObject(value) || value.schemaVersion !== "1" || value.jobId !== jobId || !isOcrResultIdentity(value, identity) + || !isObject(value.engine) || canonicalEngine(value.engine) !== "paddleocr|3.4.0|paddlepaddle-3.2.2|cpu|ocr-v2|200" + || !Array.isArray(value.pages) || !sameNumbers(value.pages.map((page) => isObject(page) ? page.page : undefined), identity.requestedPages)) return false; return value.pages.every((page) => isObject(page) && isCount(page.page) && positiveCount(page.width) && positiveCount(page.height) && isCount(page.processingMs) && typeof page.text === "string" && isObject(page.metrics) && Array.isArray(page.lines) && page.lines.every((line) => isObject(line) && typeof line.lineId === "string" && line.lineId.length > 0 && typeof line.text === "string" @@ -198,9 +217,23 @@ export function isValidOcrResult(value: unknown, jobId: string, expected: { docu function canonicalEngine(engine: Record): string { return [engine.name, engine.version, engine.runtime, engine.device, engine.configVersion, engine.dpi].join("|"); } +function canonicalIdentity(identity: Pick): string { + return JSON.stringify({ configVersion: identity.configVersion, documentSha256: identity.documentSha256, idempotencyKey: identity.idempotencyKey, requestedPages: identity.requestedPages }); +} +function isOcrIdentity(value: Record, identity: OcrIdentity): boolean { + return value.idempotencyKey === identity.idempotencyKey && isOcrResultIdentity(value, identity); +} +function isOcrResultIdentity(value: Record, identity: Pick): boolean { + return value.documentSha256 === identity.documentSha256 && value.configVersion === identity.configVersion + && value.requestIdentitySha256 === identity.requestIdentitySha256 && sameNumbers(value.requestedPages, identity.requestedPages) + && typeof value.idempotencyKey === "string" + && value.requestIdentitySha256 === buildOcrIdentity({ documentSha256: value.documentSha256, pages: value.requestedPages as number[], idempotencyKey: value.idempotencyKey }).requestIdentitySha256; +} function isObject(value: unknown): value is Record { return typeof value === "object" && value !== null && !Array.isArray(value); } function isCount(value: unknown): value is number { return Number.isInteger(value) && Number(value) >= 0; } function positiveCount(value: unknown): value is number { return isCount(value) && value > 0; } function validRatio(value: unknown): value is number { return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1; } function sameNumbers(value: unknown, expected: number[]): boolean { return Array.isArray(value) && value.length === expected.length && value.every((entry, index) => entry === expected[index]); } -function integrityError(): Error { return new Error("OCR response integrity validation failed"); } +function integrityError(): OcrClientError { + return new OcrClientError("OCR_RESPONSE_INTEGRITY_FAILED", undefined, false, "OCR response integrity validation failed"); +} diff --git a/src/modules/ocr/composition.ts b/src/modules/ocr/composition.ts index 2470e3a..7eab8cb 100644 --- a/src/modules/ocr/composition.ts +++ b/src/modules/ocr/composition.ts @@ -1,4 +1,5 @@ import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js"; +import type { OcrBlockingReason, OcrQualityOutcome, OcrQualityWarning } from "./detection.js"; export interface CandidateLineInput { lineId: string; @@ -13,6 +14,10 @@ export interface CandidatePageInput { nativeText: string; rawOcrText: string; lines: CandidateLineInput[]; + qualityOutcome?: OcrQualityOutcome; + warnings?: OcrQualityWarning[]; + blockingReasons?: OcrBlockingReason[]; + primaryBlockingReason?: OcrBlockingReason | null; } export interface CandidateLine extends CandidateLineInput { @@ -28,6 +33,10 @@ export interface CandidatePage { candidateText: string; candidateTextSha256: string; risks: string[]; + qualityOutcome: OcrQualityOutcome; + warnings: OcrQualityWarning[]; + blockingReasons: OcrBlockingReason[]; + primaryBlockingReason: OcrBlockingReason | null; } const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g; @@ -97,7 +106,11 @@ export function composeCandidate(inputPages: CandidatePageInput[]): { lines, candidateText, candidateTextSha256: sha256Hex(candidateText), - risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText) + risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText), + qualityOutcome: input.qualityOutcome ?? "accepted", + warnings: [...(input.warnings ?? [])], + blockingReasons: [...(input.blockingReasons ?? [])], + primaryBlockingReason: input.primaryBlockingReason ?? null }; }); diff --git a/src/modules/ocr/detection.ts b/src/modules/ocr/detection.ts index 0a31793..3e57d8d 100644 --- a/src/modules/ocr/detection.ts +++ b/src/modules/ocr/detection.ts @@ -17,6 +17,19 @@ export interface OcrQualityMetrics { lowConfidenceLineRatio: number; } +export type OcrExtractionMethod = "native" | "ocr" | "blank"; +export type OcrQualityOutcome = "accepted" | "warning" | "blocked"; +export type OcrQualityWarning = "LOW_P10_CONFIDENCE"; +export type OcrBlockingReason = "INSUFFICIENT_TEXT" | "LOW_MEDIAN_CONFIDENCE" | "EXCESS_LOW_CONFIDENCE_LINES"; + +export interface OcrPageClassification { + extractionMethod: OcrExtractionMethod; + qualityOutcome: OcrQualityOutcome; + warnings: OcrQualityWarning[]; + blockingReasons: OcrBlockingReason[]; + primaryBlockingReason: OcrBlockingReason | null; +} + const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u; export function computeNativeMetrics(text: string): NativeTextMetrics { @@ -56,16 +69,33 @@ export function selectPdfPagesForOcr( export function classifyOcrPage(input: { inkCoverage: number; metrics: OcrQualityMetrics; -}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } { +}): OcrPageClassification { const { inkCoverage, metrics } = input; - if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" }; - if ( - metrics.nonWhitespaceCharacters >= 40 - && metrics.medianConfidence >= 0.8 - && metrics.p10Confidence >= 0.5 - && metrics.lowConfidenceLineRatio <= 0.2 - ) { - return { method: "ocr" }; + if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) { + return { extractionMethod: "blank", qualityOutcome: "accepted", warnings: [], blockingReasons: [], primaryBlockingReason: null }; } - return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" }; + + const blockingReasons: OcrBlockingReason[] = []; + if (metrics.nonWhitespaceCharacters < 40) blockingReasons.push("INSUFFICIENT_TEXT"); + if (metrics.medianConfidence < 0.8) blockingReasons.push("LOW_MEDIAN_CONFIDENCE"); + if (metrics.lowConfidenceLineRatio > 0.2) blockingReasons.push("EXCESS_LOW_CONFIDENCE_LINES"); + if (blockingReasons.length > 0) { + return { + extractionMethod: "ocr", + qualityOutcome: "blocked", + warnings: [], + blockingReasons, + primaryBlockingReason: blockingReasons[0]! + }; + } + if (metrics.p10Confidence < 0.5) { + return { + extractionMethod: "ocr", + qualityOutcome: "warning", + warnings: ["LOW_P10_CONFIDENCE"], + blockingReasons: [], + primaryBlockingReason: null + }; + } + return { extractionMethod: "ocr", qualityOutcome: "accepted", warnings: [], blockingReasons: [], primaryBlockingReason: null }; } diff --git a/src/modules/ocr/dispatcher.ts b/src/modules/ocr/dispatcher.ts index cf1d95b..a241ab7 100644 --- a/src/modules/ocr/dispatcher.ts +++ b/src/modules/ocr/dispatcher.ts @@ -1,5 +1,5 @@ import type { OcrJobRow } from "../catalog/repository.js"; -import { OcrClientError, type OcrClient, type OcrResult } from "./client.js"; +import { buildOcrIdentity, OcrClientError, type OcrClient, type OcrResult } from "./client.js"; export interface OcrDispatchStore { claimNextOcrJob(leaseMs: number): Promise; @@ -11,6 +11,8 @@ export interface OcrDispatchStore { failOcrJob(jobId: string, code: string, detail: string): Promise; markReviewRequired(versionId: string): Promise; markFailed(versionId: string, code: string, detail: string): Promise; + listOcrVersionsAwaitingCandidate?(): Promise; + listOcrJobs?(versionId: string): Promise; } export type OcrDispatchInput = { bytes: Buffer; documentSha256: string }; @@ -38,9 +40,10 @@ export class OcrDispatcher { } private async dispatch(job: OcrJobRow): Promise { + let remoteJobId = job.remoteJobId; try { const input = await this.loadInput(job); - let remoteJobId = job.remoteJobId; + const identity = buildOcrIdentity({ documentSha256: input.documentSha256, pages: job.requestedPages, idempotencyKey: job.remoteIdempotencyKey }); if (!remoteJobId) { const acknowledgement = await this.client.submit(input.bytes, { documentSha256: input.documentSha256, @@ -48,9 +51,15 @@ export class OcrDispatcher { }, job.remoteIdempotencyKey); remoteJobId = acknowledgement.jobId; await this.store.setOcrRemoteJob(job.jobId, remoteJobId, this.leaseMs); + if (acknowledgement.status === "failed") { + const status = await this.client.getStatus(remoteJobId, identity); + const code = status.error?.code ?? "OCR_REMOTE_FAILED"; + await this.fail(job, code, status.error?.message ?? "OCR processing failed"); + return "failed"; + } } - const status = await this.client.getStatus(remoteJobId); + const status = await this.client.getStatus(remoteJobId, identity); if (status.status === "queued" || status.status === "running") { const delayMs = Math.min(2_000 * 2 ** Math.max(0, job.attemptCount - 1), 15_000); await this.store.requeueOcrJob(job.jobId, "OCR_PENDING", status.status, delayMs); @@ -62,20 +71,37 @@ export class OcrDispatcher { return "failed"; } - const result = await this.client.getResult(remoteJobId, { - documentSha256: input.documentSha256, - pages: job.requestedPages - }); + const result = await this.client.getResult(remoteJobId, identity); const durableResult = await this.persistResult(job, result); const versionComplete = await this.store.completeOcrJob(job.jobId, durableResult); if (versionComplete) { - await this.finalizeCandidate(job.versionId); - await this.store.markReviewRequired(job.versionId); + if (this.store.listOcrVersionsAwaitingCandidate && this.store.listOcrJobs) { + await this.recoverCompletedCandidates(); + } else { + try { + await this.finalizeCandidate(job.versionId); + await this.store.markReviewRequired(job.versionId); + await this.client.delete(remoteJobId).catch(() => undefined); + } catch (error) { + if (error instanceof Error && error.message === "OCR_QUALITY_BLOCKED") { + await this.store.markFailed(job.versionId, error.message, "OCR quality report contains blocking pages"); + return "failed"; + } + return "pending"; + } + } } - await this.client.delete(remoteJobId).catch(() => undefined); return "succeeded"; } catch (error) { const detail = error instanceof Error ? error.message : "Unknown OCR dispatch failure"; + if (error instanceof OcrClientError && !error.retryable) { + await this.fail(job, error.code, detail); + return "failed"; + } + if (remoteJobId) { + await this.store.requeueOcrJob(job.jobId, "OCR_LOCAL_FINALIZATION_PENDING", detail, 15_000).catch(() => undefined); + return "pending"; + } if (error instanceof OcrClientError && error.retryable) { await this.store.requeueOcrJob(job.jobId, error.code, detail, 15_000); return "pending"; @@ -95,6 +121,28 @@ export class OcrDispatcher { return recovered.length; } + async recoverCompletedCandidates(): Promise { + if (!this.store.listOcrVersionsAwaitingCandidate || !this.store.listOcrJobs) return 0; + let recovered = 0; + for (const versionId of await this.store.listOcrVersionsAwaitingCandidate()) { + try { + await this.finalizeCandidate(versionId); + await this.store.markReviewRequired(versionId); + for (const job of await this.store.listOcrJobs(versionId)) { + if (job.remoteJobId) await this.client.delete(job.remoteJobId).catch(() => undefined); + } + recovered += 1; + } catch (error) { + if (error instanceof Error && error.message === "OCR_QUALITY_BLOCKED") { + await this.store.markFailed(versionId, error.message, "OCR quality report contains blocking pages"); + recovered += 1; + } + // Durable evidence remains intact; a future reconciliation retries local finalization. + } + } + return recovered; + } + dispatchAvailable(): Promise { if (this.activeDrain) return this.activeDrain; const drain = this.drainAvailable(); diff --git a/src/modules/ocr/review.ts b/src/modules/ocr/review.ts index 0c09687..dec2419 100644 --- a/src/modules/ocr/review.ts +++ b/src/modules/ocr/review.ts @@ -2,6 +2,7 @@ import { CatalogError } from "../catalog/errors.js"; import { withTransaction, type PgPool, type PgPoolClient } from "../catalog/client.js"; import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js"; import { OcrArtifactError, persistReviewedPagesArtifact, readComposedCandidateArtifact, readReviewImageArtifact, removeReviewedPagesArtifact } from "./artifacts.js"; +import type { OcrBlockingReason, OcrQualityOutcome, OcrQualityWarning } from "./detection.js"; export interface OcrReviewLine { lineId: string; @@ -16,6 +17,7 @@ export interface OcrReviewCandidate { sourceId: string; state: "review_required" | "indexing" | "rejected"; candidateSha256: string; + qualityReportSha256: string; baseActiveVersionId: string | null; currentActiveVersionId: string | null; activateRequested: boolean; @@ -29,6 +31,10 @@ export interface OcrReviewCandidate { candidateText: string; differences: string[]; risks: string[]; + qualityOutcome: OcrQualityOutcome; + warnings: OcrQualityWarning[]; + blockingReasons: OcrBlockingReason[]; + primaryBlockingReason: OcrBlockingReason | null; }> }>; } @@ -67,7 +73,9 @@ export interface OcrReviewContext { versionId: string; sourceId: string; state: string; baseActiveVersionId: string | null; currentActiveVersionId: string | null; activateRequested: boolean; processingFingerprint: string; metadataHash: string; pages: Array<{ documentId: string; page: number; nativeTextSha256: string; ocrTextSha256: string | null; - candidateTextSha256: string; metrics: Record; risks: string[] }>; + candidateTextSha256: string; metrics: Record; risks: string[]; qualityOutcome: OcrQualityOutcome; + warnings: OcrQualityWarning[]; blockingReasons: OcrBlockingReason[]; primaryBlockingReason: OcrBlockingReason | null; + qualityReportSha256: string }>; } type ReviewReader = Pick; @@ -169,8 +177,11 @@ export class PostgresOcrReviewStore implements OcrReviewStore { if (!row) throw new CatalogError("OCR review candidate not found", 404, "REVIEW_NOT_FOUND"); if (row.state !== "review_required") throw conflict("Version is not awaiting OCR review", "INVALID_VERSION_STATE"); const pages = await client.query<{ document_id: string; page_number: number | string; native_text_hash: string; - ocr_text_hash: string | null; candidate_text_hash: string; risk_tokens: string[] }>( - `SELECT document_id, page_number, native_text_hash, ocr_text_hash, candidate_text_hash, risk_tokens + ocr_text_hash: string | null; candidate_text_hash: string; risk_tokens: string[]; quality_outcome: OcrQualityOutcome; + quality_warnings: OcrQualityWarning[]; blocking_reasons: OcrBlockingReason[]; + primary_blocking_reason: OcrBlockingReason | null; quality_report_sha256: string }>( + `SELECT document_id, page_number, native_text_hash, ocr_text_hash, candidate_text_hash, risk_tokens, + quality_outcome, quality_warnings, blocking_reasons, primary_blocking_reason, quality_report_sha256 FROM rag_document_pages WHERE version_id = $1 ORDER BY document_id, page_number FOR UPDATE`, [versionId]); const candidate = await this.reader.view(versionId); if (candidate.versionId !== row.version_id || candidate.sourceId !== row.source_id || candidate.state !== row.state @@ -182,7 +193,10 @@ export class PostgresOcrReviewStore implements OcrReviewStore { const locked = pages.rows.find((value) => value.document_id === documentId && Number(value.page_number) === page.page); return !locked || locked.native_text_hash !== sha256Hex(page.nativeText) || locked.ocr_text_hash !== (page.ocr.text ? sha256Hex(page.ocr.text) : null) - || locked.candidate_text_hash !== sha256Hex(page.candidateText) || canonicalJson(locked.risk_tokens) !== canonicalJson(page.risks); + || locked.candidate_text_hash !== sha256Hex(page.candidateText) || canonicalJson(locked.risk_tokens) !== canonicalJson(page.risks) + || locked.quality_outcome !== page.qualityOutcome || canonicalJson(locked.quality_warnings) !== canonicalJson(page.warnings) + || canonicalJson(locked.blocking_reasons) !== canonicalJson(page.blockingReasons) + || locked.primary_blocking_reason !== page.primaryBlockingReason || locked.quality_report_sha256 !== candidate.qualityReportSha256; })) throw conflict("OCR review page evidence changed"); return candidate; } @@ -203,16 +217,25 @@ export class DurableOcrReviewReader { const persisted = context.pages.find(({ documentId, page }) => documentId === item.documentId && page === item.page.page); if (!persisted || persisted.nativeTextSha256 !== item.page.nativeTextSha256 || persisted.ocrTextSha256 !== item.page.ocrTextSha256 || persisted.candidateTextSha256 !== item.page.candidateTextSha256 || canonicalJson(persisted.metrics) !== canonicalJson(item.page.metrics) - || canonicalJson(persisted.risks) !== canonicalJson(item.page.risks)) throw new Error("OCR review candidate lifecycle validation failed"); + || canonicalJson(persisted.risks) !== canonicalJson(item.page.risks) || persisted.qualityOutcome !== item.page.qualityOutcome + || canonicalJson(persisted.warnings) !== canonicalJson(item.page.warnings) + || canonicalJson(persisted.blockingReasons) !== canonicalJson(item.page.blockingReasons) + || persisted.primaryBlockingReason !== item.page.primaryBlockingReason + || persisted.qualityReportSha256 !== item.page.qualityReportSha256 + || persisted.qualityReportSha256 !== candidate.qualityReportSha256) throw new Error("OCR review candidate lifecycle validation failed"); await readReviewImageArtifact({ rootDirectory: this.rootDirectory, versionId, documentId: item.documentId, page: item.page.page }); } const { pages: _lifecyclePages, ...identity } = context; - return { ...identity, state: "review_required", candidateSha256: candidate.candidateSha256, - documents: candidate.documents.map((document) => ({ documentId: document.documentId, pages: document.pages.map((page) => ({ + const documents = candidate.documents.map((document) => ({ documentId: document.documentId, + pages: prioritizeReviewPages(document.pages.map((page) => ({ page: page.page, imageUrl: `/ingestions/${versionId}/documents/${encodeURIComponent(document.documentId)}/pages/${page.page}/image`, nativeText: page.nativeText, ocr: { text: page.rawOcrText, lines: page.lines }, candidateText: page.candidateText, - differences: page.nativeText === page.rawOcrText ? [] : ["Native and OCR text differ"], risks: page.risks - })) })) }; + differences: page.nativeText === page.rawOcrText ? [] : ["Native and OCR text differ"], risks: page.risks, + qualityOutcome: page.qualityOutcome, warnings: page.warnings, blockingReasons: page.blockingReasons, + primaryBlockingReason: page.primaryBlockingReason + }))) })).sort((left, right) => compareReviewPages(left.pages[0]!, right.pages[0]!)); + return { ...identity, state: "review_required", candidateSha256: candidate.candidateSha256, + qualityReportSha256: candidate.qualityReportSha256, documents }; } async image(versionId: string, documentId: string, page: number) { @@ -346,3 +369,14 @@ function applyCorrections(candidate: OcrReviewCandidate, corrections: OcrCorrect function reviewedText(candidate: OcrReviewCandidate): string { return candidate.documents.flatMap(({ pages }) => pages.filter(({ candidateText }) => candidateText).map(({ candidateText }) => candidateText)).join("\n\n"); } + +export function prioritizeReviewPages(pages: ReadonlyArray): T[] { + return [...pages].sort(compareReviewPages); +} + +function compareReviewPages(left: { page: number; qualityOutcome: OcrQualityOutcome; risks: string[] }, + right: { page: number; qualityOutcome: OcrQualityOutcome; risks: string[] }): number { + return Number(right.qualityOutcome === "warning") - Number(left.qualityOutcome === "warning") + || Number(right.risks.length > 0) - Number(left.risks.length > 0) + || left.page - right.page; +} diff --git a/tests/catalog/migration-004.test.ts b/tests/catalog/migration-004.test.ts new file mode 100644 index 0000000..8f52063 --- /dev/null +++ b/tests/catalog/migration-004.test.ts @@ -0,0 +1,15 @@ +import assert from "node:assert/strict"; +import { readFile } from "node:fs/promises"; +import test from "node:test"; + +test("migration 004 adds nullable quality identity without rewriting historical rows", async () => { + const sql = await readFile(new URL("../../migrations/004_ocr_quality_diagnostics.sql", import.meta.url), "utf8"); + assert.match(sql, /quality_outcome text NULL/i); + assert.match(sql, /quality_warnings jsonb NOT NULL DEFAULT '\[\]'::jsonb/i); + assert.match(sql, /blocking_reasons jsonb NOT NULL DEFAULT '\[\]'::jsonb/i); + assert.match(sql, /primary_blocking_reason text NULL/i); + assert.match(sql, /quality_report_sha256 char\(64\) NULL/i); + assert.match(sql, /CHECK \(quality_outcome IN \('accepted', 'warning', 'blocked'\)\)/i); + assert.doesNotMatch(sql, /UPDATE\s+rag_document_pages/i); + assert.doesNotMatch(sql, /UPDATE\s+rag_source_versions/i); +}); diff --git a/tests/catalog/repository-ocr.test.ts b/tests/catalog/repository-ocr.test.ts index 8f1e1c9..54529bf 100644 --- a/tests/catalog/repository-ocr.test.ts +++ b/tests/catalog/repository-ocr.test.ts @@ -83,6 +83,17 @@ test("pending OCR lookup returns undefined when only terminal candidates exist", assert.equal(candidate, undefined); }); +test("orphan sweep excludes versions with active durable OCR jobs", async () => { + const repository = new CatalogRepository(poolFor(async (sql, params) => { + assert.match(sql, /NOT EXISTS \(\s*SELECT 1 FROM rag_ocr_jobs j/s); + assert.match(sql, /j\.state IN \('queued', 'running'\)/); + assert.deepEqual(params, ["1800000"]); + return { rowCount: 0, rows: [] }; + }) as never); + + assert.deepEqual(await repository.listOrphanedIndexingCandidates(), []); +}); + test("lease claiming uses a locked queue row and keeps its remote identity", async () => { const calls: string[] = []; const pool = poolFor(async (sql, params) => { @@ -140,12 +151,14 @@ test("candidate lifecycle persistence requires exact native, OCR, and metrics ev assert.deepEqual(await repository.listOcrVersionsAwaitingCandidate(), ["version-1"]); await repository.persistOcrCandidate("version-1", [{ documentId: "document-1", page: 1, method: "ocr", nativeTextSha256: "a".repeat(64), - ocrTextSha256: "b".repeat(64), candidateTextSha256: "b".repeat(64), metrics: { inkCoverage: 0.4 }, risks: ["CBGO4a"] + ocrTextSha256: "b".repeat(64), candidateTextSha256: "b".repeat(64), metrics: { inkCoverage: 0.4 }, risks: ["CBGO4a"], + qualityOutcome: "warning", warnings: ["LOW_P10_CONFIDENCE"], blockingReasons: [], primaryBlockingReason: null, + qualityReportSha256: "d".repeat(64) }]); assert.deepEqual(jobs.map(({ jobId, state }) => [jobId, state]), [["job-1", "succeeded"]]); - assert.match(calls[6]!.sql, /native_text_hash = \$8.*ocr_text_hash IS NOT DISTINCT FROM \$9.*metrics = \$10::jsonb/s); - assert.deepEqual(calls[6]!.params?.slice(-3), ["a".repeat(64), "b".repeat(64), JSON.stringify({ inkCoverage: 0.4 })]); + assert.match(calls[6]!.sql, /quality_outcome = \$11.*quality_warnings = \$12::jsonb.*blocking_reasons = \$13::jsonb/s); + assert.deepEqual(calls[6]!.params?.slice(-5), ["warning", JSON.stringify(["LOW_P10_CONFIDENCE"]), "[]", null, "d".repeat(64)]); }); test("review transitions enforce indexing, approval, and rejection state guards", async () => { @@ -170,9 +183,30 @@ test("review transitions enforce indexing, approval, and rejection state guards" ); }); +test("blocked candidate diagnostics commit before signaling the terminal quality outcome", async () => { + const calls: Array<{ sql: string; params?: unknown[] }> = []; + const client = { async query(sql: string, params?: unknown[]) { + calls.push({ sql, params }); + return { rowCount: /^(BEGIN|COMMIT|ROLLBACK|SET CONSTRAINTS)/u.test(sql.trim()) ? 0 : 1, rows: [] }; + }, release() {} }; + const repository = new CatalogRepository({ connect: async () => client } as never); + + await assert.rejects(repository.persistOcrCandidate("version-1", [{ + documentId: "document-1", page: 4, method: "ocr", nativeTextSha256: "a".repeat(64), + ocrTextSha256: "b".repeat(64), candidateTextSha256: "c".repeat(64), metrics: { inkCoverage: 0.5 }, risks: [], + qualityOutcome: "blocked", warnings: [], blockingReasons: ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE"], + primaryBlockingReason: "INSUFFICIENT_TEXT", qualityReportSha256: "d".repeat(64) + }]), /OCR_QUALITY_BLOCKED/u); + assert.equal(calls.at(-1)?.sql.trim(), "COMMIT"); + assert.deepEqual(calls.find(({ sql }) => sql.includes("quality_outcome = $11"))?.params?.slice(-5), + ["blocked", "[]", JSON.stringify(["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE"]), "INSUFFICIENT_TEXT", "d".repeat(64)]); +}); + test("review context reconstructs exact lifecycle identity and rejects incomplete page evidence", async () => { const rows = [{ document_id: "document-1", page_number: 1, native_text_hash: "a".repeat(64), ocr_text_hash: "b".repeat(64), - candidate_text_hash: "c".repeat(64), metrics: { inkCoverage: 0.4 }, risk_tokens: ["CBGO4a"] }]; + candidate_text_hash: "c".repeat(64), metrics: { inkCoverage: 0.4 }, risk_tokens: ["CBGO4a"], quality_outcome: "warning", + quality_warnings: ["LOW_P10_CONFIDENCE"], blocking_reasons: [], primary_blocking_reason: null, + quality_report_sha256: "d".repeat(64) }]; const repository = new CatalogRepository(poolFor(async (sql) => sql.includes("JOIN rag_sources") ? { rowCount: 1, rows: [{ version_id: "version-1", source_id: "source-1", state: "review_required", base_active_version_id: null, current_active_version_id: null, activate_requested: false, processing_fingerprint: "fingerprint", metadata_hash: "metadata" }] } @@ -180,7 +214,44 @@ test("review context reconstructs exact lifecycle identity and rejects incomplet const context = await repository.loadOcrReviewContext("version-1"); assert.deepEqual(context?.pages[0], { documentId: "document-1", page: 1, nativeTextSha256: "a".repeat(64), - ocrTextSha256: "b".repeat(64), candidateTextSha256: "c".repeat(64), metrics: { inkCoverage: 0.4 }, risks: ["CBGO4a"] }); + ocrTextSha256: "b".repeat(64), candidateTextSha256: "c".repeat(64), metrics: { inkCoverage: 0.4 }, risks: ["CBGO4a"], + qualityOutcome: "warning", warnings: ["LOW_P10_CONFIDENCE"], blockingReasons: [], primaryBlockingReason: null, + qualityReportSha256: "d".repeat(64) }); rows[0]!.candidate_text_hash = null as never; await assert.rejects(repository.loadOcrReviewContext("version-1"), (error) => error instanceof CatalogError && error.code === "OCR_REVIEW_EVIDENCE_INCOMPLETE"); }); + +test("quality blocking persists deterministic safe status details without OCR text or secrets", async () => { + const calls: Array<{ sql: string; params?: unknown[] }> = []; + let statusRead = false; + const repository = new CatalogRepository(poolFor(async (sql, params) => { + calls.push({ sql, params }); + if (sql.includes("quality_outcome = 'blocked'")) return { rowCount: 1, rows: [{ + document_id: "document-1", page_number: 3, + blocking_reasons: ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE"], primary_blocking_reason: "INSUFFICIENT_TEXT", + quality_report_sha256: "e".repeat(64) + }] }; + if (sql.includes("UPDATE rag_source_versions SET state = 'failed'")) return { rowCount: 1, rows: [] }; + if (sql.includes("SELECT source_id, state, error_code")) { + statusRead = true; + const errorDetail = calls.find(({ sql: statement }) => statement.includes("UPDATE rag_source_versions SET state = 'failed'"))?.params?.[1]; + return { rowCount: 1, rows: [{ source_id: "source-1", state: "failed", error_code: "OCR_QUALITY_BLOCKED", error_detail: errorDetail }] }; + } + if (statusRead && sql.includes("rag_version_documents")) return { rowCount: 1, rows: [{ + document_id: "document-1", index_state: "failed", job_state: "succeeded", completed_pages: 1, requested_pages: [3] + }] }; + if (statusRead && sql.includes("rag_document_pages")) return { rowCount: 1, rows: [{ + document_id: "document-1", page_number: 3, extraction_method: "ocr", blocked_reason: null, quality_outcome: "blocked", + quality_warnings: [], blocking_reasons: ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE"], + primary_blocking_reason: "INSUFFICIENT_TEXT", quality_report_sha256: "e".repeat(64) + }] }; + return { rowCount: 1, rows: [] }; + }) as never); + + await repository.markFailed("version-1", "OCR_QUALITY_BLOCKED", "unsafe OCR text token=secret /private/path"); + const status = await repository.getIngestionStatus("version-1"); + const serialized = JSON.stringify(status); + assert.match(serialized, /OCR quality validation blocked one or more pages/u); + assert.match(serialized, /INSUFFICIENT_TEXT/u); + assert.doesNotMatch(serialized, /unsafe OCR text|token=secret|private\/path/u); +}); diff --git a/tests/ocr/client.test.ts b/tests/ocr/client.test.ts index 02553a1..11209cd 100644 --- a/tests/ocr/client.test.ts +++ b/tests/ocr/client.test.ts @@ -3,10 +3,11 @@ import { access, lstat, mkdir, mkdtemp, readFile, stat, symlink, utimes, writeFi import os from "node:os"; import path from "node:path"; import test from "node:test"; -import { OcrClient, OcrClientError, type OcrResult } from "../../src/modules/ocr/client.js"; +import { buildOcrIdentity, OcrClient, OcrClientError, type OcrResult } from "../../src/modules/ocr/client.js"; import { persistComposedCandidateArtifact, persistOcrResultArtifact, + persistReviewImageArtifacts, readComposedCandidateArtifact, readOcrResultArtifact, resolveArtifactPath, @@ -19,6 +20,7 @@ import { canonicalJson, hashOrderedPairs, sha256Hex } from "../../src/shared/uti const document = Buffer.from("%PDF-1.4\nunit-7\n%%EOF\n"); const documentSha256 = sha256Hex(document); const jobId = "ocr_job-7"; +const identity = buildOcrIdentity({ documentSha256, pages: [1, 3] }); function jsonResponse(status: number, body: unknown): Response { return new Response(JSON.stringify(body), { status, headers: { "content-type": "application/json" } }); @@ -28,25 +30,25 @@ function ack(overrides: Record = {}): Record { return { jobId, status: "queued", - documentSha256, - requestedPages: [1, 3], - configVersion: "ocr-v1", + ...identity, createdAt: "2026-09-14T10:00:00.000Z", ...overrides }; } function result(overrides: Record = {}): Record { + const pages = Array.isArray(overrides.pages) ? overrides.pages.map((page) => (page as { page: number }).page) : [1, 3]; + const resultIdentity = buildOcrIdentity({ documentSha256, pages, idempotencyKey: typeof overrides.idempotencyKey === "string" ? overrides.idempotencyKey : undefined }); return { schemaVersion: "1", jobId, - documentSha256, + ...resultIdentity, engine: { name: "paddleocr", version: "3.4.0", runtime: "paddlepaddle-3.2.2", device: "cpu", - configVersion: "ocr-v1", + configVersion: "ocr-v2", dpi: 200 }, pages: [1, 3].map((page) => ({ @@ -91,13 +93,13 @@ test("runtime retry stub preserves identity, exact attempts, backoff, and result }); const submitted = await client.submit(document, { documentSha256, pages: [1, 3] }); - const completed = await client.getResult(jobId, { documentSha256, pages: [1, 3] }); + const completed = await client.getResult(jobId, identity); assert.equal(submitted.jobId, jobId); assert.deepEqual(delays, [2_000, 4_000]); assert.equal(calls.filter(({ url }) => url.endsWith("/v1/jobs")).length, 3); assert.equal(new Set(calls.slice(0, 3).map(({ key }) => key)).size, 1); - assert.equal(calls[0]?.key, `${documentSha256}:ocr-v1:${sha256Hex("[1,3]")}`); + assert.equal(calls[0]?.key, `${documentSha256}:ocr-v2:${sha256Hex("[1,3]")}`); assert.deepEqual(completed.pages.map(({ page }) => page), [1, 3]); }); @@ -123,10 +125,13 @@ test("queue pressure and deterministic failures are surfaced without retries", a }); test("strict validation rejects mismatched acknowledgements and result contracts", async () => { + const malformedResult = result(); + malformedResult.pages = [result().pages]; for (const response of [ jsonResponse(202, ack({ requestedPages: [3, 1] })), + jsonResponse(202, ack({ status: "unknown" })), jsonResponse(200, result({ schemaVersion: "2" })), - jsonResponse(200, result({ pages: [result().pages as unknown] })) + jsonResponse(200, malformedResult) ]) { const client = new OcrClient({ baseUrl: "http://ocr.internal:8000", @@ -136,7 +141,7 @@ test("strict validation rejects mismatched acknowledgements and result contracts }); const operation = response.status === 202 ? client.submit(document, { documentSha256, pages: [1, 3] }) - : client.getResult(jobId, { documentSha256, pages: [1, 3] }); + : client.getResult(jobId, identity); await assert.rejects(operation, /OCR response integrity validation failed/); } }); @@ -150,6 +155,7 @@ test("polling uses bounded exponential backoff until a terminal status", async ( fetch: (async () => jsonResponse(200, { jobId, status: states.shift(), + ...identity, completedPages: states.length === 0 ? 2 : 0, totalPages: 2, error: null @@ -157,7 +163,7 @@ test("polling uses bounded exponential backoff until a terminal status", async ( sleep: async (milliseconds) => { delays.push(milliseconds); } }); - assert.equal((await client.pollUntilTerminal(jobId)).status, "succeeded"); + assert.equal((await client.pollUntilTerminal(jobId, identity)).status, "succeeded"); assert.deepEqual(delays, [2_000, 4_000]); }); @@ -183,13 +189,14 @@ test("review image transfer validates authentication, identity, content type, an token: "image-token", fetch: (async (_input, init) => new Response(png, { status: 200, headers: { "content-type": "image/png", "content-length": String(png.length), "x-document-sha256": documentSha256, + "x-ocr-job-id": jobId, "x-ocr-identity-sha256": identity.requestIdentitySha256, "x-ocr-config-version": "ocr-v2", "x-content-sha256": sha256Hex(png), "x-page-number": "1", "x-seen-authorization": new Headers(init?.headers).get("authorization") ?? "" } })) as typeof fetch }); - assert.deepEqual(await client.getReviewImage(jobId, 1, documentSha256), { bytes: png, sha256: sha256Hex(png) }); - await assert.rejects(client.getReviewImage(jobId, 2, documentSha256), /integrity validation failed/); + assert.deepEqual(await client.getReviewImage(jobId, 1, identity), { bytes: png, sha256: sha256Hex(png) }); + await assert.rejects(client.getReviewImage(jobId, 2, identity), /identity is invalid/); }); test("review image transfer treats size and HTTP 500 as terminal while retrying transient failures", async () => { @@ -199,14 +206,15 @@ test("review image transfer treats size and HTTP 500 as terminal while retrying const client = new OcrClient({ baseUrl: "http://ocr.internal:8000", token: "token", fetch: (async () => { attempts += 1; return new Response(null, { status }); }) as typeof fetch, sleep: async () => undefined }); - await assert.rejects(client.getReviewImage(jobId, 1, documentSha256), + await assert.rejects(client.getReviewImage(jobId, 1, identity), (error: unknown) => error instanceof OcrClientError && error.status === status && error.retryable === retryable); assert.equal(attempts, expectedAttempts); } const invalidSize = new OcrClient({ baseUrl: "http://ocr.internal:8000", token: "token", fetch: (async () => new Response(png, { headers: { "content-type": "image/png", "content-length": String(png.length + 1), + "x-ocr-job-id": jobId, "x-ocr-identity-sha256": identity.requestIdentitySha256, "x-ocr-config-version": "ocr-v2", "x-document-sha256": documentSha256, "x-content-sha256": sha256Hex(png), "x-page-number": "1" } })) as typeof fetch }); - await assert.rejects(invalidSize.getReviewImage(jobId, 1, documentSha256), /integrity validation failed/); + await assert.rejects(invalidSize.getReviewImage(jobId, 1, identity), /integrity validation failed/); }); test("review image transfer is sequential and resumes from durable per-page checkpoints", async (context) => { @@ -343,6 +351,10 @@ test("restart-safe candidate composition persists exact native, OCR, and blank p { page: 3, width: 1700, height: 2200, processingMs: 10, text: "", metrics: { lineCount: 0, nonWhitespaceCharacters: 0, inkCoverage: 0.001, medianConfidence: 0, p10Confidence: 0, lowConfidenceLineRatio: 0 }, lines: [] } ] }) as unknown as OcrResult }); + await persistReviewImageArtifacts({ rootDirectory, versionId, documentId, images: [2, 3].map((page) => { + const bytes = Buffer.from(`89504e470d0a1a0a${String(page).padStart(4, "0")}`, "hex"); + return { page, bytes, sha256: sha256Hex(bytes) }; + }) }); const persisted = await persistComposedCandidateArtifact({ rootDirectory, versionId, jobs: [{ documentId, remoteJobId: jobId, requestedPages: [2, 3], state: "succeeded" }] }); const restarted = await readComposedCandidateArtifact({ rootDirectory, versionId, artifactSha256: persisted.artifactSha256 }); @@ -375,9 +387,11 @@ test("candidate composition fails closed on quality failure and corrupt native e page: 1, width: 1700, height: 2200, processingMs: 10, text: "", lines: [], metrics: { lineCount: 0, nonWhitespaceCharacters: 0, inkCoverage: 0.5, medianConfidence: 0, p10Confidence: 0, lowConfidenceLineRatio: 0 } }] }) as unknown as OcrResult }); + const blockedPng = Buffer.from("89504e470d0a1a0a0001", "hex"); + await persistReviewImageArtifacts({ rootDirectory, versionId, documentId, images: [{ page: 1, bytes: blockedPng, sha256: sha256Hex(blockedPng) }] }); const jobs = [{ documentId, remoteJobId: jobId, requestedPages: [1], state: "succeeded" as const }]; - await assert.rejects(persistComposedCandidateArtifact({ rootDirectory, versionId, jobs }), /OCR_QUALITY_BLOCKED/); + assert.equal((await persistComposedCandidateArtifact({ rootDirectory, versionId, jobs })).artifactPath, null); const manifest = JSON.parse(await readFile(staged.manifestPath, "utf8")) as { documents: Array<{ nativePagesPath: string }> }; await writeFile(resolveArtifactPath(staged.versionDirectory, manifest.documents[0]!.nativePagesPath), "{corrupt", { mode: 0o600 }); await assert.rejects(persistComposedCandidateArtifact({ rootDirectory, versionId, jobs }), /native page artifact integrity/); diff --git a/tests/ocr/contracts-deploy.test.ts b/tests/ocr/contracts-deploy.test.ts index 86664fb..132aa41 100644 --- a/tests/ocr/contracts-deploy.test.ts +++ b/tests/ocr/contracts-deploy.test.ts @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; import { readFile } from "node:fs/promises"; import { buildApiHelp, openApiDocument } from "../../src/api/openapi.js"; import { env } from "../../src/config/env.js"; +import { releaseVersion, resolveReleaseVersion } from "../../src/config/version.js"; const api = openApiDocument as unknown as { paths: Record>>; @@ -64,8 +65,13 @@ test("OpenAPI OCR schemas preserve progress, review evidence, corrections, and d }); test("OCR deployment defaults and container wiring match the private durable contract", async () => { - assert.equal(env.ragVersion, "0.2.0"); + assert.equal(env.ragVersion, releaseVersion); assert.equal(env.buildRevision, "unknown"); + assert.throws(() => resolveReleaseVersion("RAG_VERSION", "incorrect"), /must match VERSION/u); + + const packageJson = JSON.parse(await readFile(new URL("../../package.json", import.meta.url), "utf8")) as { version: string }; + assert.equal(packageJson.version, releaseVersion); + assert.equal((openApiDocument as { info: { version: string } }).info.version, releaseVersion); assert.deepEqual({ root: env.ocrArtifactRoot, uploadBytes: (env as unknown as Record).ocrMaxUploadBytes, @@ -75,8 +81,12 @@ test("OCR deployment defaults and container wiring match the private durable con }, { root: "/data/ingestions", uploadBytes: 50 * 1024 * 1024, pages: 100, pageTimeoutMs: 60_000, totalTimeoutMs: 15 * 60_000 }); const dockerfile = await readFile(new URL("../../Dockerfile", import.meta.url), "utf8"); + const ocrDockerfile = await readFile(new URL("../../ocr-service/Dockerfile", import.meta.url), "utf8"); + const ocrDockerignore = await readFile(new URL("../../ocr-service/Dockerfile.dockerignore", import.meta.url), "utf8"); assert.match(dockerfile, /VOLUME \["\/data\/ingestions"\]/u); assert.match(dockerfile, /ENV NODE_ENV=production OCR_ARTIFACT_ROOT=\/data\/ingestions/u); + assert.match(dockerfile, /COPY VERSION package\.json package-lock\.json tsconfig\.json/u); + assert.match(dockerfile, /test "\$RAG_VERSION" = "\$\(tr -d/u); assert.match(dockerfile, /RAG_VERSION=\$\{RAG_VERSION\}/u); assert.match(dockerfile, /BUILD_REVISION=\$\{BUILD_REVISION\}/u); assert.match(dockerfile, /org\.opencontainers\.image\.version/u); @@ -84,4 +94,10 @@ test("OCR deployment defaults and container wiring match the private durable con assert.match(dockerfile, /USER node/u); assert.match(dockerfile, /dist\/modules\/catalog\/migrations\.js.*dist\/server\.js/u); assert.doesNotMatch(dockerfile, /(?:COPY|ADD)\s+\.env|ENV\s+(?:OCR_INTERNAL_TOKEN|LIFECYCLE_ADMIN_TOKEN)=/u); + assert.match(ocrDockerfile, /ARG OCR_VERSION\n/u); + assert.match(ocrDockerfile, /COPY VERSION \/srv\/VERSION/u); + assert.match(ocrDockerfile, /test "\$OCR_VERSION" = "\$\(tr -d/u); + assert.match(ocrDockerfile, /OCR_VERSION=\$\{OCR_VERSION\}/u); + assert.match(ocrDockerfile, /org\.opencontainers\.image\.version/u); + assert.match(ocrDockerignore, /^!VERSION$/mu); }); diff --git a/tests/ocr/detection.test.ts b/tests/ocr/detection.test.ts index fd2b185..5443bdb 100644 --- a/tests/ocr/detection.test.ts +++ b/tests/ocr/detection.test.ts @@ -87,19 +87,36 @@ test("a parsed mixed PDF selects only unique ordered insufficient pages and non- } }); -test("OCR blank and quality gates fail closed at exact thresholds", () => { +test("OCR quality classification preserves every blocker and treats isolated low p10 as a warning", () => { const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 }; - assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" }); - assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" }); - assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" }); - for (const metrics of [ - { ...passing, nonWhitespaceCharacters: 39 }, - { ...passing, medianConfidence: 0.799 }, - { ...passing, p10Confidence: 0.499 }, - { ...passing, lowConfidenceLineRatio: 0.201 } - ]) { - assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked"); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { + extractionMethod: "blank", qualityOutcome: "accepted", warnings: [], blockingReasons: [], primaryBlockingReason: null + }); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { + extractionMethod: "ocr", qualityOutcome: "accepted", warnings: [], blockingReasons: [], primaryBlockingReason: null + }); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: { ...passing, p10Confidence: 0.339 } }), { + extractionMethod: "ocr", qualityOutcome: "warning", warnings: ["LOW_P10_CONFIDENCE"], blockingReasons: [], primaryBlockingReason: null + }); + assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: { + nonWhitespaceCharacters: 39, medianConfidence: 0.799, p10Confidence: 0.2, lowConfidenceLineRatio: 0.201 + } }), { + extractionMethod: "ocr", qualityOutcome: "blocked", warnings: [], + blockingReasons: ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE", "EXCESS_LOW_CONFIDENCE_LINES"], + primaryBlockingReason: "INSUFFICIENT_TEXT" + }); +}); + +test("FacturaTech v7 low-p10 regressions are warnings without client text", () => { + const p10ByPage = new Map([[2, 0.339], [20, 0.350], [21, 0.285]]); + for (const [page, p10Confidence] of p10ByPage) { + const classified = classifyOcrPage({ inkCoverage: 0.321, metrics: { + nonWhitespaceCharacters: 399, medianConfidence: page === 2 ? 0.988 : page === 21 ? 0.983 : 0.999, + p10Confidence, lowConfidenceLineRatio: page === 2 ? 0.158 : 0.122 + } }); + assert.equal(classified.qualityOutcome, "warning", `page ${page}`); + assert.deepEqual(classified.warnings, ["LOW_P10_CONFIDENCE"]); } }); diff --git a/tests/ocr/dispatcher.test.ts b/tests/ocr/dispatcher.test.ts index 14b03fb..6cd0010 100644 --- a/tests/ocr/dispatcher.test.ts +++ b/tests/ocr/dispatcher.test.ts @@ -14,6 +14,7 @@ import { IngestService } from "../../src/modules/ingest/service.js"; import type { EmbeddingProvider } from "../../src/modules/embeddings/provider.js"; import type { VectorStoreClient } from "../../src/modules/vectorstore/client.js"; import { sha256Hex } from "../../src/shared/utils/ids.js"; +import { buildOcrIdentity, OcrClientError } from "../../src/modules/ocr/client.js"; function setEnvFlag(name: keyof typeof env, value: unknown): void { (env as unknown as Record)[name] = value; @@ -212,9 +213,10 @@ test("application wiring dispatches a newly accepted OCR job from durable artifa const png = Buffer.from("89504e470d0a1a0a0102", "hex"); const client = { async submit(bytes: Buffer, expected: { documentSha256: string }, key: string) { assert.equal(key, persistedKey); documentSha256 = expected.documentSha256; dispatchCalls.push(`${bytes.length}:${key}`); return { jobId: "runtime-remote" }; }, - async getStatus() { return { jobId: "runtime-remote", status: "succeeded", completedPages: 1, totalPages: 1, error: null }; }, + async getStatus() { return { jobId: "runtime-remote", status: "succeeded", ...buildOcrIdentity({ documentSha256, pages: [1], idempotencyKey: String(persistedKey) }), completedPages: 1, totalPages: 1, error: null }; }, async getResult() { const text = "Factura FAT07 has enough OCR characters for review"; return { schemaVersion: "1", jobId: "runtime-remote", documentSha256, - engine: { name: "paddleocr", version: "3.4.0", runtime: "paddlepaddle-3.2.2", device: "cpu", configVersion: "ocr-v1", dpi: 200 }, + ...buildOcrIdentity({ documentSha256, pages: [1], idempotencyKey: String(persistedKey) }), + engine: { name: "paddleocr", version: "3.4.0", runtime: "paddlepaddle-3.2.2", device: "cpu", configVersion: "ocr-v2", dpi: 200 }, pages: [{ page: 1, width: 100, height: 100, processingMs: 1, text, metrics: { lineCount: 1, nonWhitespaceCharacters: 40, inkCoverage: 0.5, medianConfidence: 0.95, p10Confidence: 0.95, lowConfidenceLineRatio: 0 }, lines: [{ lineId: "p1-l1", text, confidence: 0.95, bbox: [1, 2, 3, 4] }] }] }; }, async getReviewImage() { return { bytes: png, sha256: sha256Hex(png) }; }, async delete() {} @@ -232,8 +234,8 @@ test("application wiring dispatches a newly accepted OCR job from durable artifa const response = await fetch(`http://127.0.0.1:${address.port}/ingest/upload`, { method: "POST", body: form }); assert.equal(response.status, 202); assert.equal((await response.json() as { uploadedResource: string }).uploadedResource, "runtime.pdf"); - for (let attempt = 0; attempt < 40 && !dispatchCalls.includes("review-required"); attempt += 1) await new Promise((resolve) => setTimeout(resolve, 5)); - assert.match(dispatchCalls[0]!, /^\d+:.+:ocr-v1:.+$/u); + for (let attempt = 0; attempt < 200 && !dispatchCalls.includes("review-required"); attempt += 1) await new Promise((resolve) => setTimeout(resolve, 10)); + assert.match(dispatchCalls[0]!, /^\d+:.+:ocr-v2:.+$/u); assert.deepEqual(dispatchCalls.slice(1), ["completed", "candidate", "review-required"]); assert.deepEqual((await readReviewImageArtifact({ rootDirectory: path.join(directory, "artifacts"), versionId: String(completedJob?.versionId), documentId: String(completedJob?.documentId), page: 1 })).bytes, png); }); @@ -249,7 +251,7 @@ test("dispatcher recovers only expired work, reuses its remote job, and complete state: "queued" as const, requestedPages: [1], completedPages: 0, - configVersion: "ocr-v1", + configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, @@ -289,11 +291,11 @@ test("dispatcher recovers only expired work, reuses its remote job, and complete test("dispatcher retains remote OCR state when durable result transfer fails", async () => { const calls: string[] = []; - const job = { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + const job = { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; const store = { async claimNextOcrJob() { return job; }, async completeOcrJob() { calls.push("complete"); return true; }, - async failOcrJob() { calls.push("job-failed"); }, async markFailed() { calls.push("version-failed"); } + async requeueOcrJob() { calls.push("requeued"); }, async failOcrJob() { calls.push("job-failed"); }, async markFailed() { calls.push("version-failed"); } }; const client = { async getStatus() { return { status: "succeeded" }; }, async getResult() { return { pages: [] }; }, @@ -307,13 +309,42 @@ test("dispatcher retains remote OCR state when durable result transfer fails", a async () => undefined ); + assert.equal(await dispatcher.runOnce(), "pending"); + assert.deepEqual(calls, ["artifact-write", "requeued"]); +}); + +test("dispatcher fails closed on an unrecoverable remote OCR response", async () => { + const calls: string[] = []; + const job = { jobId: "job-integrity", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + const dispatcher = new OcrDispatcher({ + async claimNextOcrJob() { return job; }, async requeueOcrJob() { calls.push("requeued"); }, + async failOcrJob(_jobId: string, code: string) { calls.push(`job:${code}`); }, + async markFailed(_versionId: string, code: string) { calls.push(`version:${code}`); } + } as never, { + async getStatus() { throw new OcrClientError("OCR_RESPONSE_INTEGRITY_FAILED", undefined, false); } + } as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }), async (_job, result) => result, async () => undefined); + assert.equal(await dispatcher.runOnce(), "failed"); - assert.deepEqual(calls, ["artifact-write", "job-failed", "version-failed"]); + assert.deepEqual(calls, ["job:OCR_RESPONSE_INTEGRITY_FAILED", "version:OCR_RESPONSE_INTEGRITY_FAILED"]); +}); + +test("dispatcher propagates the remote terminal failure code", async () => { + const calls: string[] = []; + const job = { jobId: "job-failed", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + const dispatcher = new OcrDispatcher({ + async claimNextOcrJob() { return job; }, async failOcrJob(_jobId: string, code: string) { calls.push(`job:${code}`); }, + async markFailed(_versionId: string, code: string) { calls.push(`version:${code}`); } + } as never, { + async getStatus() { return { status: "failed", error: { code: "OCR_ENGINE_FAILED", message: "engine failed" } }; } + } as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }), async (_job, result) => result, async () => undefined); + + assert.equal(await dispatcher.runOnce(), "failed"); + assert.deepEqual(calls, ["job:OCR_ENGINE_FAILED", "version:OCR_ENGINE_FAILED"]); }); test("dispatcher retains remote OCR state when durable candidate finalization fails", async () => { const calls: string[] = []; - const job = { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + const job = { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: "remote-1", remoteIdempotencyKey: "key", state: "running" as const, requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; const dispatcher = new OcrDispatcher({ async claimNextOcrJob() { return job; }, async completeOcrJob() { calls.push("complete"); return true; }, async failOcrJob() { calls.push("job-failed"); }, async markFailed(_versionId: string, code: string) { calls.push(`version-failed:${code}`); }, @@ -324,14 +355,112 @@ test("dispatcher retains remote OCR state when durable candidate finalization fa async (_job, result) => result, async () => { calls.push("candidate-write-readback"); throw new Error("OCR_QUALITY_BLOCKED"); }); assert.equal(await dispatcher.runOnce(), "failed"); - assert.deepEqual(calls, ["complete", "candidate-write-readback", "job-failed", "version-failed:OCR_QUALITY_BLOCKED"]); + assert.deepEqual(calls, ["complete", "candidate-write-readback", "version-failed:OCR_QUALITY_BLOCKED"]); +}); + +test("local finalization resumes after every durable boundary exactly once without resubmitting OCR", async () => { + for (const failedBoundary of ["result", "images", "report", "diagnostics"] as const) { + const completedBoundaries = new Set(); + let boundaryFailed = false; + let jobState: "queued" | "running" | "succeeded" = "queued"; + let versionState: "indexing" | "review_required" = "indexing"; + let reviewTransitions = 0; + let submitCalls = 0; + const job = { jobId: `job-${failedBoundary}`, versionId: `version-${failedBoundary}`, documentId: "document-1", + remoteJobId: "remote-1", remoteIdempotencyKey: "persisted-key", state: "running" as const, + requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, + heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + const persistBoundary = (boundary: typeof failedBoundary) => { + completedBoundaries.add(boundary); + if (!boundaryFailed && boundary === failedBoundary) { + boundaryFailed = true; + throw new Error(`simulated crash after ${boundary}`); + } + }; + const dispatcher = new OcrDispatcher({ + async claimNextOcrJob() { + if (jobState !== "queued") return undefined; + jobState = "running"; + return job; + }, + async claimOcrJob() { return undefined; }, + async recoverExpiredOcrLeases() { return []; }, + async setOcrRemoteJob() { throw new Error("a known remote job must never be resubmitted"); }, + async requeueOcrJob() { jobState = "queued"; }, + async completeOcrJob() { assert.equal(jobState, "running"); jobState = "succeeded"; return true; }, + async failOcrJob() { throw new Error("a recoverable local crash must not fail OCR"); }, + async markReviewRequired() { assert.equal(versionState, "indexing"); versionState = "review_required"; reviewTransitions += 1; }, + async markFailed() { throw new Error("a recoverable local crash must not fail the version"); }, + async listOcrVersionsAwaitingCandidate() { return jobState === "succeeded" && versionState === "indexing" ? [job.versionId] : []; }, + async listOcrJobs() { return [{ ...job, state: "succeeded" as const }]; } + } as never, { + async submit() { submitCalls += 1; throw new Error("a known remote job must never be submitted"); }, + async getStatus() { return { status: "succeeded" }; }, + async getResult() { return { pages: [{ page: 1 }] }; }, + async delete() {} + } as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }), + async (_job, result) => { + persistBoundary("result"); + persistBoundary("images"); + return result; + }, async () => { + persistBoundary("report"); + persistBoundary("diagnostics"); + }); + + const firstAttempt = await dispatcher.runOnce(); + if (failedBoundary === "result" || failedBoundary === "images") { + assert.equal(firstAttempt, "pending", failedBoundary); + assert.equal(await dispatcher.runOnce(), "succeeded", failedBoundary); + } else { + assert.equal(firstAttempt, "succeeded", failedBoundary); + } + await dispatcher.recoverCompletedCandidates(); + await dispatcher.recoverCompletedCandidates(); + + assert.deepEqual([...completedBoundaries].sort(), ["diagnostics", "images", "report", "result"]); + assert.equal(reviewTransitions, 1, failedBoundary); + assert.equal(submitCalls, 0, failedBoundary); + assert.equal(versionState, "review_required", failedBoundary); + } +}); + +test("administrative routes expose only classified errors", async (context) => { + const previousToken = env.lifecycleAdminToken; + setEnvFlag("lifecycleAdminToken", "admin-token"); + context.after(() => setEnvFlag("lifecycleAdminToken", previousToken)); + let failure: Error = new Error("unexpected failure at /private/path with token=secret"); + const server = createApp({ catalog: { + async activateVersion() { throw failure; } + } as never, startReconciler: false }).listen(0); + context.after(() => server.close()); + const address = server.address(); + assert.ok(address && typeof address === "object"); + const activate = () => fetch(`http://127.0.0.1:${address.port}/sources/source-1/versions/version-1/activate`, { + method: "POST", headers: { authorization: "Bearer admin-token", "content-type": "application/json" }, + body: JSON.stringify({ expectedActiveVersionId: "active-1" }) + }); + + const unexpected = await activate(); + assert.equal(unexpected.status, 500); + const unexpectedBody = await unexpected.json() as { error: string; code?: string }; + assert.equal(unexpectedBody.error, "Unknown activation error"); + assert.equal(unexpectedBody.code, undefined); + assert.doesNotMatch(JSON.stringify(unexpectedBody), /private\/path|token=secret/u); + + failure = new CatalogError("Active version changed during activation", 409, "ACTIVE_VERSION_CHANGED"); + const classified = await activate(); + assert.equal(classified.status, 409); + assert.deepEqual(await classified.json(), { + ok: false, error: "Active version changed during activation", code: "ACTIVE_VERSION_CHANGED" + }); }); test("OCR exhaustion fails the leased candidate without activation or engine substitution", async () => { const calls: string[] = []; const repository = { async claimNextOcrJob() { - return { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: null, remoteIdempotencyKey: "same-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; + return { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: null, remoteIdempotencyKey: "same-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; }, async setOcrRemoteJob() { calls.push("remote"); }, async requeueOcrJob() { calls.push("requeue"); }, @@ -373,7 +502,7 @@ test("reconciler makes expired OCR leases dispatchable while leaving live leases assert.deepEqual(calls, ["ocr-recovery", "ocr-dispatch"]); }); -test("reconciler restart leaves completed OCR work without a durable candidate untouched", async () => { +test("reconciler restart completes durable OCR work without resubmitting OCR", async () => { const candidate = { state: "indexing", durableArtifact: false }; const calls: string[] = []; const catalog = { @@ -400,8 +529,8 @@ test("reconciler restart leaves completed OCR work without a durable candidate u assert.equal((await reconciler.runOnce()).ok, true); } - assert.deepEqual(candidate, { state: "indexing", durableArtifact: false }); - assert.deepEqual(calls, ["ocr-recovery", "ocr-dispatch", "ocr-recovery", "ocr-dispatch"]); + assert.deepEqual(candidate, { state: "review_required", durableArtifact: true }); + assert.deepEqual(calls, ["ocr-recovery", "ocr-dispatch", "candidate-recovery", "ocr-recovery", "ocr-dispatch", "candidate-recovery"]); }); test("runtime HTTP routing returns native 201, OCR 202/status, and catalog-down 503", async () => { diff --git a/tests/ocr/e2e.test.ts b/tests/ocr/e2e.test.ts index 1b148f7..b9382df 100644 --- a/tests/ocr/e2e.test.ts +++ b/tests/ocr/e2e.test.ts @@ -166,7 +166,7 @@ test("resends reuse identity, OCR exhaustion fails closed, and catalog absence r const calls: string[] = []; let candidateState = "indexing"; const dispatcher = new OcrDispatcher({ - async claimNextOcrJob() { return { jobId: "job", versionId: scanVersion, documentId: "doc", remoteJobId: null, remoteIdempotencyKey: "stable-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; }, + async claimNextOcrJob() { return { jobId: "job", versionId: scanVersion, documentId: "doc", remoteJobId: null, remoteIdempotencyKey: "stable-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v2", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; }, async claimOcrJob() { return undefined; }, async recoverExpiredOcrLeases() { return []; }, async setOcrRemoteJob() { calls.push("remote"); }, async requeueOcrJob() { calls.push("requeue"); }, async completeOcrJob() { calls.push("complete"); return false; }, async failOcrJob() { calls.push("job-failed"); }, async markReviewRequired() { calls.push("review"); }, diff --git a/tests/ocr/indexing-store.test.ts b/tests/ocr/indexing-store.test.ts index c080e5c..8611ca7 100644 --- a/tests/ocr/indexing-store.test.ts +++ b/tests/ocr/indexing-store.test.ts @@ -1,36 +1,55 @@ import assert from "node:assert/strict"; -import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import test from "node:test"; import { CatalogError } from "../../src/modules/catalog/errors.js"; import type { EmbeddingProvider } from "../../src/modules/embeddings/provider.js"; import { PostgresOcrIndexingStore, OcrReadyIndexingService, type ApprovedOcrCandidate } from "../../src/modules/ocr/indexing.js"; -import { persistReviewedPagesArtifact } from "../../src/modules/ocr/artifacts.js"; +import { persistComposedCandidateArtifact, persistOcrResultArtifact, persistReviewedPagesArtifact, persistReviewImageArtifacts, + stageOcrArtifacts } from "../../src/modules/ocr/artifacts.js"; +import { buildOcrIdentity } from "../../src/modules/ocr/client.js"; +import { computeNativeMetrics } from "../../src/modules/ocr/detection.js"; import { chunkDocument, documentalChunkingPolicy } from "../../src/modules/process/chunking.js"; import type { VectorStoreClient } from "../../src/modules/vectorstore/client.js"; -import { buildChunkId, buildVersionedQdrantPointId, canonicalJson, normalizeContentForHash, sha256Hex } from "../../src/shared/utils/ids.js"; +import { buildChunkId, buildVersionedQdrantPointId, normalizeContentForHash, sha256Hex } from "../../src/shared/utils/ids.js"; const versionId = "eeeeeeee-eeee-4eee-8eee-eeeeeeeeeeee"; const documentId = "doc:indexed-review"; const sourceId = "src:reviewed"; +const jobId = "indexed-review"; +const png = Buffer.from("89504e470d0a1a0a01020304", "hex"); + +function ocrPage(page: number, text: string) { + return { page, width: 100, height: 100, processingMs: 1, text, + metrics: { lineCount: 1, nonWhitespaceCharacters: computeNativeMetrics(text).nonWhitespaceCharacters, inkCoverage: 0.4, + medianConfidence: 0.97, p10Confidence: 0.97, lowConfidenceLineRatio: 0 }, + lines: [{ lineId: `p${page}-l1`, text, confidence: 0.97, bbox: [1, 2, 30, 10] as [number, number, number, number] }] }; +} async function fixture(rootDirectory: string) { const first = "Reviewed alpha content. ".repeat(110); - const second = "Reviewed beta content."; - const lines = [first, second].map((text, index) => ({ - lineId: `line-${index + 1}`, text, confidence: 0.97, bbox: [0, index * 10, 50, index * 10 + 8] as [number, number, number, number], lineSha256: sha256Hex(text) - })); - const pages = lines.map((line, index) => ({ page: index + 1, method: "ocr" as const, nativeText: "", rawOcrText: line.text, - candidateText: line.text, candidateTextSha256: sha256Hex(line.text), risks: [], lines: [line], - nativeTextSha256: sha256Hex(""), ocrTextSha256: sha256Hex(line.text), metrics: { medianConfidence: 0.97 } })); - const documents = [{ documentId, text: `${first}\n\n${second}`, textSha256: sha256Hex(`${first}\n\n${second}`), pages }]; - const candidateSha256 = sha256Hex(canonicalJson(documents)); - const reviewedText = `${first}\n\n${second}`; - await mkdir(path.join(rootDirectory, versionId)); - await writeFile(path.join(rootDirectory, versionId, "candidate-pages.json"), canonicalJson({ schemaVersion: "1", versionId, candidateSha256, documents }), { mode: 0o600 }); - await persistReviewedPagesArtifact({ rootDirectory, versionId, sourceId, candidateSha256, reviewedTextSha256: sha256Hex(reviewedText), - reviewedBy: "reviewer", documents: [{ documentId, pages: pages.map(({ page, candidateText, lines: ocrLines }) => ({ page, candidateText, ocr: { lines: ocrLines } })) }] }); + const second = "Reviewed beta content. ".repeat(3).trim(); + const original = Buffer.from("%PDF-reviewed-index"); + const identity = buildOcrIdentity({ documentSha256: sha256Hex(original), pages: [1, 2], idempotencyKey: "indexed-review-key" }); + await stageOcrArtifacts({ rootDirectory, versionId, createdAt: "2026-09-22T12:00:00.000Z", documents: [{ + documentId, documentKey: "review.pdf", bytes: original, requestedPages: [1, 2], + pages: [1, 2].map((page) => ({ page, text: "", rasterCoverage: 1, textSha256: sha256Hex("") })) + }] }); + await persistOcrResultArtifact({ rootDirectory, versionId, documentId, result: { + schemaVersion: "1", jobId, ...identity, + engine: { name: "paddleocr", version: "3.4.0", runtime: "paddlepaddle-3.2.2", device: "cpu", configVersion: "ocr-v2", dpi: 200 }, + pages: [ocrPage(1, first), ocrPage(2, second)] + } }); + await persistReviewImageArtifacts({ rootDirectory, versionId, documentId, + images: [1, 2].map((page) => ({ page, bytes: png, sha256: sha256Hex(png) })) }); + const { candidate } = await persistComposedCandidateArtifact({ rootDirectory, versionId, + jobs: [{ documentId, remoteJobId: jobId, requestedPages: [1, 2], state: "succeeded" }] }); + const pages = candidate.documents[0]!.pages; + const reviewedText = pages.map(({ candidateText }) => candidateText).join("\n\n"); + await persistReviewedPagesArtifact({ rootDirectory, versionId, sourceId, candidateSha256: candidate.candidateSha256, + reviewedTextSha256: sha256Hex(reviewedText), reviewedBy: "reviewer", + documents: [{ documentId, pages: pages.map(({ page, candidateText, lines }) => ({ page, candidateText, ocr: { lines } })) }] }); return { reviewedText, pages }; } diff --git a/tests/ocr/quality-diagnostics.test.ts b/tests/ocr/quality-diagnostics.test.ts new file mode 100644 index 0000000..eb0e067 --- /dev/null +++ b/tests/ocr/quality-diagnostics.test.ts @@ -0,0 +1,139 @@ +import assert from "node:assert/strict"; +import { access, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { buildOcrIdentity, type OcrResult } from "../../src/modules/ocr/client.js"; +import { persistComposedCandidateArtifact, persistOcrResultArtifact, persistReviewImageArtifacts, + readOcrQualityReportArtifact, stageOcrArtifacts } from "../../src/modules/ocr/artifacts.js"; +import { sha256Hex } from "../../src/shared/utils/ids.js"; +import { prioritizeReviewPages } from "../../src/modules/ocr/review.js"; + +const png = Buffer.from("89504e470d0a1a0a01020304", "hex"); + +function fixturePng(page: number): Buffer { + return Buffer.concat([png, Buffer.from([page])]); +} + +function result(documentSha256: string, pages: OcrResult["pages"]): OcrResult { + const identity = buildOcrIdentity({ documentSha256, pages: pages.map(({ page }) => page), idempotencyKey: "quality-job-key" }); + return { + schemaVersion: "1", jobId: "quality-job", ...identity, + engine: { name: "paddleocr", version: "3.4.0", runtime: "paddlepaddle-3.2.2", device: "cpu", configVersion: "ocr-v2", dpi: 200 }, + pages + }; +} + +async function stageQualityVersion(rootDirectory: string, versionId: string, quality: "warning" | "blocked") { + const documentId = `doc:${quality}`; + const original = Buffer.from(`%PDF-${quality}`); + const text = quality === "warning" ? "safe diagnostic fixture with enough non whitespace characters 123456789" : "short"; + await stageOcrArtifacts({ rootDirectory, versionId, createdAt: "2026-09-22T12:00:00.000Z", documents: [{ + documentId, documentKey: `${quality}.pdf`, bytes: original, requestedPages: [1], + pages: [{ page: 1, text: "weak native", rasterCoverage: 1, textSha256: sha256Hex("weak native") }] + }] }); + await persistOcrResultArtifact({ rootDirectory, versionId, documentId, result: result(sha256Hex(original), [{ + page: 1, width: 100, height: 100, processingMs: 1, text, + metrics: { lineCount: 1, nonWhitespaceCharacters: quality === "warning" ? 60 : 5, inkCoverage: 0.4, + medianConfidence: quality === "warning" ? 0.99 : 0.7, p10Confidence: 0.339, + lowConfidenceLineRatio: quality === "warning" ? 0.1 : 0.3 }, + lines: [{ lineId: "p1-l1", text, confidence: 0.9, bbox: [1, 2, 30, 10] }] + }]) }); + await persistReviewImageArtifacts({ rootDirectory, versionId, documentId, + images: [{ page: 1, bytes: png, sha256: sha256Hex(png) }] }); + return { documentId, jobs: [{ documentId, remoteJobId: "quality-job", requestedPages: [1], state: "succeeded" }] }; +} + +test("quality report is private, immutable, text-free, and chained into a warning candidate", async (context) => { + const rootDirectory = await mkdtemp(path.join(os.tmpdir(), "rag-quality-report-")); + const versionId = "12121212-1212-4121-8121-121212121212"; + context.after(() => rm(rootDirectory, { recursive: true, force: true })); + const input = await stageQualityVersion(rootDirectory, versionId, "warning"); + + const persisted = await persistComposedCandidateArtifact({ rootDirectory, versionId, jobs: input.jobs }); + assert.ok(persisted.artifactPath); + const reportPath = path.join(rootDirectory, versionId, "quality-report.json"); + const report = await readOcrQualityReportArtifact({ rootDirectory, versionId, artifactSha256: persisted.qualityReportSha256 }); + assert.equal((await stat(reportPath)).mode & 0o777, 0o600); + assert.equal((await stat(persisted.artifactPath)).mode & 0o777, 0o600); + assert.equal(report.pages[0]?.qualityOutcome, "warning"); + assert.deepEqual(report.pages[0]?.warnings, ["LOW_P10_CONFIDENCE"]); + assert.equal(persisted.candidate.qualityReportSha256, persisted.qualityReportSha256); + assert.equal(persisted.candidate.documents[0]?.pages[0]?.qualityReportSha256, persisted.qualityReportSha256); + const serialized = await readFile(reportPath, "utf8"); + assert.doesNotMatch(serialized, /safe diagnostic fixture|weak native|quality-job-key/u); + const repeated = await persistComposedCandidateArtifact({ rootDirectory, versionId, jobs: input.jobs }); + assert.equal(repeated.qualityReportSha256, persisted.qualityReportSha256); + await writeFile(path.join(rootDirectory, versionId, "manifest.json"), "{}", { mode: 0o600 }); + await assert.rejects(readOcrQualityReportArtifact({ rootDirectory, versionId, + artifactSha256: persisted.qualityReportSha256 }), /quality report integrity validation failed/u); +}); + +test("blocked pages retain every ordered reason and publish no candidate artifact", async (context) => { + const rootDirectory = await mkdtemp(path.join(os.tmpdir(), "rag-quality-blocked-")); + const versionId = "34343434-3434-4343-8343-343434343434"; + context.after(() => rm(rootDirectory, { recursive: true, force: true })); + const input = await stageQualityVersion(rootDirectory, versionId, "blocked"); + + const persisted = await persistComposedCandidateArtifact({ rootDirectory, versionId, jobs: input.jobs }); + assert.equal(persisted.artifactPath, null); + assert.deepEqual(persisted.candidate.documents[0]?.pages[0]?.blockingReasons, + ["INSUFFICIENT_TEXT", "LOW_MEDIAN_CONFIDENCE", "EXCESS_LOW_CONFIDENCE_LINES"]); + assert.equal(persisted.candidate.documents[0]?.pages[0]?.primaryBlockingReason, "INSUFFICIENT_TEXT"); + await access(path.join(rootDirectory, versionId, "quality-report.json")); + await assert.rejects(access(path.join(rootDirectory, versionId, "candidate-pages.json"))); +}); + +test("review ordering places warned pages first and then pages with ambiguous tokens", () => { + const ordered = prioritizeReviewPages([ + { page: 1, qualityOutcome: "accepted" as const, risks: [] }, + { page: 2, qualityOutcome: "accepted" as const, risks: ["FATo7"] }, + { page: 3, qualityOutcome: "warning" as const, risks: [] } + ]); + assert.deepEqual(ordered.map(({ page }) => page), [3, 2, 1]); +}); + +test("self-contained 25-page fixture retains p10 warnings, every review PNG, and a reviewable candidate", async (context) => { + const rootDirectory = await mkdtemp(path.join(os.tmpdir(), "rag-quality-25-pages-")); + const versionId = "56565656-5656-4565-8565-565656565656"; + const documentId = "doc:twenty-five-pages"; + const pages = Array.from({ length: 25 }, (_, index) => index + 1); + const warningPages = new Set([2, 20, 21]); + const original = Buffer.from("%PDF-self-contained-twenty-five-page-fixture"); + context.after(() => rm(rootDirectory, { recursive: true, force: true })); + + await stageOcrArtifacts({ rootDirectory, versionId, createdAt: "2026-09-22T12:00:00.000Z", documents: [{ + documentId, documentKey: "twenty-five-pages.pdf", bytes: original, requestedPages: pages, + pages: pages.map((page) => { + const text = `native fixture page ${page}`; + return { page, text, rasterCoverage: 1, textSha256: sha256Hex(text) }; + }) + }] }); + await persistOcrResultArtifact({ rootDirectory, versionId, documentId, result: result(sha256Hex(original), pages.map((page) => { + const text = `self-contained OCR fixture page ${page} has enough text for human review`; + const p10Confidence = warningPages.has(page) ? 0.339 : 0.95; + return { + page, width: 100, height: 100, processingMs: 1, text, + metrics: { lineCount: 1, nonWhitespaceCharacters: text.replace(/\s/gu, "").length, inkCoverage: 0.4, + medianConfidence: 0.99, p10Confidence, lowConfidenceLineRatio: 0.1 }, + lines: [{ lineId: `p${page}-l1`, text, confidence: 0.99, bbox: [1, 2, 30, 10] }] + }; + })) }); + const images = await persistReviewImageArtifacts({ rootDirectory, versionId, documentId, + images: pages.map((page) => { + const bytes = fixturePng(page); + return { page, bytes, sha256: sha256Hex(bytes) }; + }) }); + + const persisted = await persistComposedCandidateArtifact({ rootDirectory, versionId, + jobs: [{ documentId, remoteJobId: "quality-job", requestedPages: pages, state: "succeeded" }] }); + const report = await readOcrQualityReportArtifact({ rootDirectory, versionId, artifactSha256: persisted.qualityReportSha256 }); + + assert.ok(persisted.artifactPath); + assert.equal(images.images.length, 25); + assert.equal(report.pages.length, 25); + assert.deepEqual(report.pages.filter(({ qualityOutcome }) => qualityOutcome === "warning").map(({ page }) => page), [2, 20, 21]); + assert.ok(report.pages.every(({ qualityOutcome }) => qualityOutcome === "accepted" || qualityOutcome === "warning")); + assert.equal(persisted.candidate.documents[0]?.pages.length, 25); + assert.deepEqual(persisted.candidate.documents[0]?.pages.filter(({ warnings }) => warnings.includes("LOW_P10_CONFIDENCE")).map(({ page }) => page), [2, 20, 21]); +}); diff --git a/tests/ocr/review.test.ts b/tests/ocr/review.test.ts index eb139d1..dbd6c38 100644 --- a/tests/ocr/review.test.ts +++ b/tests/ocr/review.test.ts @@ -11,6 +11,7 @@ import { classifyOcrReviewError, OcrReviewService, PostgresOcrReviewStore, type import { DurableOcrReviewReader } from "../../src/modules/ocr/review.js"; import { OcrArtifactError, persistComposedCandidateArtifact, persistOcrResultArtifact, persistReviewImageArtifacts, readReviewedPagesArtifact, stageOcrArtifacts } from "../../src/modules/ocr/artifacts.js"; import { sha256Hex } from "../../src/shared/utils/ids.js"; +import { buildOcrIdentity } from "../../src/modules/ocr/client.js"; function candidate(state: OcrReviewCandidate["state"] = "review_required"): OcrReviewCandidate { const lines = ["CBGO4a", "FATo7"].map((text, index) => ({ @@ -240,8 +241,9 @@ test("production review reader survives restart and fails closed on unauthorized documentId, documentKey: "review.pdf", bytes: original, requestedPages: [1], pages: [{ page: 1, text: "weak native", rasterCoverage: 1, textSha256: sha256Hex("weak native") }] }] }); - const result = { schemaVersion: "1" as const, jobId: "ocr-review", documentSha256: sha256Hex(original), - engine: { name: "paddleocr" as const, version: "3.4.0" as const, runtime: "paddlepaddle-3.2.2" as const, device: "cpu" as const, configVersion: "ocr-v1" as const, dpi: 200 as const }, + const identity = buildOcrIdentity({ documentSha256: sha256Hex(original), pages: [1], idempotencyKey: "review-key" }); + const result = { schemaVersion: "1" as const, jobId: "ocr-review", ...identity, + engine: { name: "paddleocr" as const, version: "3.4.0" as const, runtime: "paddlepaddle-3.2.2" as const, device: "cpu" as const, configVersion: "ocr-v2" as const, dpi: 200 as const }, pages: [{ page: 1, width: 100, height: 100, processingMs: 1, text, metrics: { lineCount: 1, nonWhitespaceCharacters: 50, inkCoverage: 0.5, medianConfidence: 0.95, p10Confidence: 0.95, lowConfidenceLineRatio: 0 }, lines: [{ lineId: "p1-l1", text, confidence: 0.95, bbox: [1, 2, 30, 10] as [number, number, number, number] }] }] }; @@ -253,7 +255,9 @@ test("production review reader survives restart and fails closed on unauthorized let reads = 0; const contextValue = { versionId, sourceId: "source-1", state: "review_required" as const, baseActiveVersionId: null, currentActiveVersionId: null, activateRequested: false, processingFingerprint: "fingerprint", metadataHash: "metadata", pages: [{ documentId, page: 1, - nativeTextSha256: page.nativeTextSha256, ocrTextSha256: page.ocrTextSha256, candidateTextSha256: page.candidateTextSha256, metrics: page.metrics, risks: page.risks }] }; + nativeTextSha256: page.nativeTextSha256, ocrTextSha256: page.ocrTextSha256, candidateTextSha256: page.candidateTextSha256, metrics: page.metrics, risks: page.risks, + qualityOutcome: page.qualityOutcome, warnings: page.warnings, blockingReasons: page.blockingReasons, + primaryBlockingReason: page.primaryBlockingReason, qualityReportSha256: page.qualityReportSha256 }] }; const catalog = { async loadOcrReviewContext() { reads += 1; return contextValue; } }; const restarted = new DurableOcrReviewReader(catalog, rootDirectory); assert.equal((await restarted.view(versionId)).documents[0]!.pages[0]!.ocr.lines[0]!.lineSha256, sha256Hex(text));