Compare commits
12 commits
270f693587
...
53c4e0a62b
| Author | SHA1 | Date | |
|---|---|---|---|
| 53c4e0a62b | |||
| ac046e37c1 | |||
| b0fdef001f | |||
| 6e6970a6fa | |||
| badd982904 | |||
| 905f8854db | |||
| d2ecf19102 | |||
| 936881b38f | |||
| 1898eeb002 | |||
| ba025fc93d | |||
| 8931bf8078 | |||
| 11139bd14e |
57 changed files with 5973 additions and 62 deletions
3
.gitignore
vendored
3
.gitignore
vendored
|
|
@ -9,6 +9,9 @@ dist/
|
|||
llaves
|
||||
backups/
|
||||
npm-debug.log*
|
||||
ocr-service/.venv/
|
||||
ocr-service/*.db
|
||||
ocr-service/**/__pycache__/
|
||||
|
||||
# Local agent caches and tooling artifacts
|
||||
.atl/
|
||||
|
|
|
|||
|
|
@ -9,11 +9,14 @@ RUN npm run build
|
|||
|
||||
FROM node:22-bookworm-slim AS runtime
|
||||
WORKDIR /app
|
||||
ENV NODE_ENV=production
|
||||
ENV NODE_ENV=production OCR_ARTIFACT_ROOT=/data/ingestions
|
||||
COPY package.json package-lock.json ./
|
||||
RUN npm ci --omit=dev
|
||||
COPY --from=build /app/dist ./dist
|
||||
COPY --from=build /app/migrations ./migrations
|
||||
COPY public ./public
|
||||
RUN install -d -o node -g node -m 700 /data/ingestions
|
||||
VOLUME ["/data/ingestions"]
|
||||
USER node
|
||||
EXPOSE 3000
|
||||
CMD ["node", "dist/server.js"]
|
||||
CMD ["sh", "-c", "node dist/modules/catalog/migrations.js && exec node dist/server.js"]
|
||||
|
|
|
|||
|
|
@ -2,14 +2,131 @@
|
|||
|
||||
**Proyecto:** Workspace de tools IA para empresas
|
||||
**Modulo:** RAG
|
||||
**Ultima actualizacion:** 2026-09-08
|
||||
**Ultima modificacion por:** Agente RAG 2
|
||||
**Ultima actualizacion:** 2026-09-15
|
||||
**Ultima modificacion por:** Subagente Implement Unit 13 Local E2E
|
||||
**Estado:** Activo
|
||||
|
||||
---
|
||||
|
||||
## Registro de sesion
|
||||
|
||||
### 2026-09-15 - Subagente Implement Unit 13 Local E2E
|
||||
**Agent:** Subagente Implement Unit 13 Local E2E · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5af65977ffehYI4eK2l177YAL`
|
||||
**Responsibility:** Implement only OCR Unit 13 tasks 7.1–7.3 under strict TDD, preserving Units 8–12 and excluding production acceptance task 7.4 and Git delivery work.
|
||||
**Work:** Added deterministic localhost E2E coverage for native, scanned review/approval/activation, mixed documents, resend identity, OCR-down fail-closed behavior, and catalog-down `503` responses. No production code changed.
|
||||
**Validation:** The new requirement tests passed immediately as 2/2 baseline/characterization evidence; canonical Node passed 78/78, offline Python passed 13/13, and check/build/whitespace/cleanup/process gates passed.
|
||||
**Accounting:** 183 functional plus 52 metadata changed lines, 235 total, within the 400-line Unit 13 budget. Tasks 7.1–7.3 are complete; production task 7.4 remains unchecked and untouched.
|
||||
**Files:** `tests/ocr/e2e.test.ts`, OpenSpec tasks/progress, and this history. No external or production service was contacted; `ocr-service/.venv` was preserved.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-15 - Subagente Implement Unit 12 Contracts Deploy
|
||||
**Agent:** Subagente Implement Unit 12 Contracts Deploy · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5b09a3a3ffetN0oeikOd4f33P`
|
||||
**Responsibility:** Implement only OCR Unit 12 tasks 6.3–6.4 under strict TDD, preserving Units 8–11 and excluding all 7.x E2E, production, and Git delivery work.
|
||||
**Work:** Added accurate authenticated OCR ingestion/status/review/correction/decision OpenAPI contracts, durable private OCR deployment defaults, and a non-root migration-aware RAG container with `/data/ingestions` volume wiring.
|
||||
**Validation:** Genuine initial RED was 0/3 and triangulation produced a second RED; focused GREEN passed 3/3, relevant regressions 45/45, canonical Node 76/76, check/build/whitespace gates, sanitized Docker build/inspection, cleanup, and process checks passed.
|
||||
**Accounting:** 219 functional plus 50 metadata changed lines, 269 total, within the 400-line Unit 12 budget. Tasks 6.3–6.4 are complete; all 7.x tasks remain pending.
|
||||
**Files:** `src/api/openapi.ts`, `src/config/env.ts`, `Dockerfile`, `tests/ocr/contracts-deploy.test.ts`, OpenSpec tasks/progress, and this history. The authoritative lifecycle/OCR contract remained read-only.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-15 - Subagente Implement Unit 11 Retention
|
||||
**Agent:** Subagente Implement Unit 11 Retention · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5b1c9293ffeDc3y3lJLgJ8UHF`
|
||||
**Responsibility:** Implement only OCR Unit 11 tasks 6.1–6.2 under strict TDD, preserving Units 8–10 and excluding OpenAPI, deployment, E2E, production, and Git delivery work.
|
||||
**Work:** Added state-aware OCR retention with exact TTL selection, review expiry, active-safe CAS deletion, restart resumption, reconciler execution, and flag-off candidate API isolation while preserving native synchronous ingestion.
|
||||
**Validation:** Genuine missing-module and activation-race RED evidence reached focused GREEN 6/6; relevant regressions passed 28/28; canonical Node passed 73/73; check, build, tracked/untracked whitespace, temporary-directory cleanup, and process checks passed.
|
||||
**Accounting:** 277 functional plus 50 metadata changed lines, 327 total, within the 400-line Unit 11 budget. Tasks 6.1–6.2 are complete; tasks 6.3–6.4 and 7.x remain pending.
|
||||
**Files:** `src/app.ts`, `src/modules/catalog/{repository,reconciler}.ts`, `src/modules/ocr/retention.ts`, `tests/ocr/{dispatcher,review,retention}.test.ts`, OpenSpec tasks/progress, and this history.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-15 - Subagente Implement Unit 10 Review UI
|
||||
**Agent:** Subagente Implement Unit 10 Review UI · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5b3af43bffewaqPFGRUOsi6Qt`
|
||||
**Responsibility:** Implement only OCR Unit 10 task 5.4 under strict TDD, preserving Units 8–9 and excluding retention, deploy, and E2E work.
|
||||
**Work:** Added authenticated, state-safe rejection with durable audit completion and zero indexing/activation side effects. Extended the existing static playground with authenticated candidate inspection, protected images, corrections, approval, and rejection controls.
|
||||
**Validation:** Genuine RED was 5/7; focused GREEN/refactor passed 7/7, the localhost rejection harness passed 1/1, canonical Node passed 67/67, and check/build/whitespace gates passed. No live external dependency or persistent process was used.
|
||||
**Accounting:** 271 functional plus 42 metadata changed lines, 313 total, within the 400-line Unit 10 budget. Task 5.4 is complete; tasks 6.x/7.x remain pending.
|
||||
**Files:** `src/app.ts`, `src/modules/ocr/review.ts`, `tests/ocr/review.test.ts`, `public/playground/{index.html,app.js,styles.css}`, OpenSpec tasks/progress, and this history.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-15 - Subagente Implement Unit 9 Core
|
||||
**Agent:** Subagente Implement Unit 9 Core · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5b4d20b9ffewoOAG5oiIu25fk`
|
||||
**Responsibility:** Implement only OCR Unit 9 tasks 5.1–5.3 under strict TDD, preserving Unit 8 and excluding rejection/UI work.
|
||||
**Work:** Added authenticated review/approval routes, auditable review and atomic correction boundaries, stale-write conflicts, and new/reusable indexing activation CAS behavior. Rejected candidates are refused by indexing without embeddings.
|
||||
**Validation:** RED failed on the missing indexing module; focused GREEN passed 5/5, canonical Node passed 65/65, and check/build/tracked plus untracked whitespace gates passed. No external service or persistent process was used.
|
||||
**Accounting:** 341 functional plus 38 metadata changed lines, 379 total. Unit 10 task 5.4 remains pending.
|
||||
**Files:** `src/app.ts`, `src/modules/ocr/{review,indexing}.ts`, `tests/ocr/review.test.ts`, OpenSpec tasks/progress, and this history.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Agente RAG 2 - Diagnostico de resultados vacios de subagentes
|
||||
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_29bdbd003ffeLrLjUlFgnp08Y7`
|
||||
**Rol/responsabilidad:** Cerrar directamente en Build el diagnostico de `<task_result>` vacio antes de retomar `sdd-apply` desde orchestrator.
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Reconstruidas en modo read-only las sesiones afectadas desde las tablas `session`, `message` y `part` de OpenCode.
|
||||
- Confirmado que las ejecuciones vacias con actividad terminaron en `finish=length`, 32.000 tokens de salida y sin texto final del asistente.
|
||||
- Acotado el disparador observado a `sdd-tasks` y a reintentos `general` que seguian reescribiendo el mismo `tasks.md` bajo un limite estricto; las fases GLM `sdd-init`, `sdd-explore` y `sdd-propose` del cambio terminaron normalmente.
|
||||
- Separado el `APIError` inicial por limite de uso del agotamiento posterior de longitud de salida.
|
||||
- Verificado con una prueba breve que GLM devuelve `task_result` cuando termina con `finish=stop` y que no modifico los archivos observados.
|
||||
- Confirmado que el siguiente subagente `sdd-apply` usa `openai/gpt-5.6-sol`, no GLM.
|
||||
- Persistida en Engram y en el seguimiento central de Gentle-AI la regla de acotar tareas para GLM; el presupuesto de 400 lineas ayuda, pero no garantiza el limite de tokens.
|
||||
|
||||
**Estado final:**
|
||||
- Incidencia explicada y cerrada como bloqueo para el siguiente paso OCR.
|
||||
- No se modifico la configuracion de OpenCode.
|
||||
- Listo para volver a orchestrator y retomar `sdd-apply`.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
### 2026-09-15 - Subagente Recover OCR Unit 8
|
||||
**Agent:** Subagente Recover OCR Unit 8 · **Model:** openai/gpt-5.6-sol · **Session:** `ses_f5b7b4242ffekQxMZTYXOvPWGA`
|
||||
**Responsibility:** Recover only Unit 8 tasks 4.1, 4.4, and 4.5 under strict TDD; no Unit 9 or Git delivery work.
|
||||
**Work:** Completed OCR runtime wiring, exact lease recovery, deterministic pending reuse, mixed-document progress, upload/status HTTP behavior, and fail-closed retrieval evidence.
|
||||
**Validation:** Persisted evidence records focused dispatcher 9/9, focused repository OCR 6/6, canonical Node 60/60, and clean check/build/whitespace gates; curl returned 202 then authenticated 200. The initially leaked harness child was detected and terminated; temporary files and port were rechecked clean.
|
||||
**Evidence chain:** Native failed/interrupted revision `sha256:182cc73954a8518d03bd6c3ef5f4141b352aa9b69fca4177447d76deaddad223` preserved an 815-line candidate. The maintainer-selected `auto-chain` / `feature-branch-chain` preflight reset at revision `sha256:c0db82cf3db8ae790d34e4c310b998f97bb36864c97c3aa84e4c87ab77d07c2a` preserved that candidate and bounded only the additional recovery work. Fresh passed evidence revision: `sha256:e424afe7b2f8efd547dcfd0baf64ec1b46ac344f39afd87259560a094d59163f`.
|
||||
**Accounting:** Relative to committed HEAD `ac046e3`, the complete Unit 8 review slice is 1,045 functional changed lines plus 41 required metadata lines, for 1,086 total; it is not within 400 lines and has no `size:exception`. Only the post-reset recovery delta—350 functional plus 41 metadata lines, 391 total—fits the separate 400-line recovery objective.
|
||||
**Files:** `src/app.ts`, `src/config/env.ts`, `src/modules/{catalog,ingest,ocr}/`, `tests/ocr/dispatcher.test.ts`, OpenSpec tasks/progress, and this history.
|
||||
- `/home/pancho/Documentos/Empresa/IA/herramientas/docs/gentle-ai/SEGUIMIENTO_DESCUBRIMIENTOS_MEJORAS_GENTLE_AI.md`
|
||||
|
||||
---
|
||||
|
||||
## Registro de sesion
|
||||
|
||||
### 2026-09-13 - Subagente Inicialización SDD del RAG - sdd-init
|
||||
|
||||
**Rol/responsabilidad:** Bootstrap del contexto SDD hibrido (OpenSpec + Engram), capacidades de testing y skill registry para el trabajo de OCR del RAG. Invocado por el agente orquestador.
|
||||
|
||||
**Modelo:** ollama/glm-5.3:cloud
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Deteccion del stack real del repo: Node 22 ESM, TypeScript 5.8 strict, Express 4, node:test via tsx, tsc --noEmit; sin linter, formatter ni coverage.
|
||||
- Verificacion en vivo: `npm test` 25/25 OK, `npm run check` OK.
|
||||
- Resolucion Strict TDD: `true` (un unico proyecto en scope, `npm test` de raiz lo cubre).
|
||||
- Inicializacion OpenSpec: `openspec/config.yaml`, `openspec/specs/`, `openspec/changes/archive/`.
|
||||
- Refresco de `.atl/skill-registry.md` (ya existia con fecha 2026-09-11; el preflight decia que no existia): 20 skills indexados, sdd-*/_shared/skill-registry excluidos, dedup canonico.
|
||||
- Guardado en Engram (proyecto `rag-service`): contexto de proyecto, capacidades de testing y skill registry (obs 3372, 3373, 3374).
|
||||
|
||||
**Estado final:**
|
||||
- SDD inicializado en modo hibrido para `rag-service`; siguiente fase: `sdd-explore` para el cambio de OCR en la ingesta.
|
||||
- Sin cambios de codigo ni comportamiento de aplicacion.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `openspec/config.yaml` (nuevo)
|
||||
- `openspec/specs/.gitkeep` (nuevo)
|
||||
- `openspec/changes/archive/.gitkeep` (nuevo)
|
||||
- `.atl/skill-registry.md` (regenerado)
|
||||
- `docs/HISTORIAL_SESIONES.md` (esta entrada)
|
||||
|
||||
---
|
||||
|
||||
## Registro de sesion
|
||||
|
||||
### 2026-09-13 - Agente RAG 2 - Propuesta de backups y persistencia
|
||||
|
||||
**Modelo:** gpt-5.6-luna
|
||||
|
|
@ -352,3 +469,258 @@ Continuidad operativa y evolutiva del modulo RAG.
|
|||
- `docs/DESPLIEGUE_EASYPANEL.md`
|
||||
- `docs/CONTRATO_CICLO_VIDA_Y_OCR.md`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Agente RAG 2 - Reparacion directa de tasks OCR
|
||||
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_29bdbd003ffeLrLjUlFgnp08Y7`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Reparado directamente `openspec/changes/ocr-ingest-integration/tasks.md` tras varios resultados vacios de subagentes.
|
||||
- Reducido el artefacto de 646 a 524 palabras, por debajo del limite estricto de 530.
|
||||
- Conservados 30 tareas, 13 work units, trazabilidad de 17 requisitos/34 escenarios, Strict TDD, amenazas HTTP, contratos de activacion y estrategia feature-branch-chain.
|
||||
- Sincronizado el artefacto hibrido en Engram mediante el topic `sdd/ocr-ingest-integration/tasks`.
|
||||
|
||||
**Validacion:**
|
||||
- Cuatro guard lines requeridas presentes con valores correctos.
|
||||
- `gentle-ai sdd-status ocr-ingest-integration --json`: `applyState=ready`, `blockedReasons=[]`.
|
||||
- Revision semantica contra las tres specs completada.
|
||||
|
||||
**Estado final:**
|
||||
- Fase de tasks recuperada y lista para iniciar `sdd-apply` desde Unit 1.
|
||||
- No se implemento codigo OCR, ni se hizo commit, push o deploy.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `openspec/changes/ocr-ingest-integration/tasks.md`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 1 Migration - Migracion OCR revisable
|
||||
|
||||
**Agente:** **Subagente OCR Unit 1 Migration**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 1 (Migration) del cambio SDD `ocr-ingest-integration` mediante TDD estricto, sin ejecutar Unit 2 ni acciones de entrega Git.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_f607397e7ffeF3y4cM8FvprlCB`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Creada primero la prueba focalizada de contrato de esquema para la migracion OCR.
|
||||
- Capturado RED valido con 4/4 pruebas fallidas por ausencia de `migrations/002_ocr_review.sql`; una ejecucion previa con error sintactico en la prueba fue corregida y no se conto como RED.
|
||||
- Creada `migrations/002_ocr_review.sql` con jobs OCR durables, campos de idempotencia y leases, registros auditables por pagina, correcciones atomicas y un indice unico parcial para identidades OCR no terminales.
|
||||
- Capturado GREEN con 4/4 pruebas aprobadas mediante `npx --no-install tsx --test tests/catalog/migration-002.test.ts`.
|
||||
- Marcada exclusivamente la tarea 1.2 como completada; la tarea 1.1 permanece pendiente porque requiere las pruebas de comportamiento de repositorio de Unit 2.
|
||||
- Persistida la evidencia acumulativa de apply en OpenSpec y sincronizada con Engram mediante `sdd/ocr-ingest-integration/apply-progress`.
|
||||
|
||||
**Estado final:**
|
||||
- Unit 1 completada dentro de su limite autonomo y con rollback acotado a los dos archivos nuevos y sus metadatos de progreso.
|
||||
- Runtime PostgreSQL indicado como no aplicable en esta unidad: el plan la define como schema-only y el proyecto no dispone de harness PostgreSQL de pruebas.
|
||||
- No se modificaron la migracion 001 ni la logica de repositorio; no se hizo commit, push, PR, deploy ni acceso a secretos.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `migrations/002_ocr_review.sql`
|
||||
- `tests/catalog/migration-002.test.ts`
|
||||
- `openspec/changes/ocr-ingest-integration/tasks.md`
|
||||
- `openspec/changes/ocr-ingest-integration/apply-progress.md`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 2 Repository - Repositorio OCR
|
||||
|
||||
**Agente:** **Subagente OCR Unit 2 Repository**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 2 del cambio SDD `ocr-ingest-integration` mediante TDD estricto.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_f6068e516ffemqjXllduA068YL`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Creada primero la prueba focalizada `tests/catalog/repository-ocr.test.ts`; RED valido: 0/6 por metodos ausentes.
|
||||
- Añadidas al repositorio la recuperacion de candidato OCR pendiente, la toma atomica de jobs, la recuperacion de leases vencidos sin perder identidad remota y las transiciones protegidas de revision.
|
||||
- Refactorizada tras GREEN la proyeccion comun de filas OCR y sincronizadas las tareas 1.1, 1.3 y 1.4 con el progreso acumulado hibrido.
|
||||
|
||||
**Validacion:**
|
||||
- Reejecucion segura: 6/6; `npm test`: 10/10; suite raiz explicita: 25/25; `npm run check` y `git diff --check`: correctos.
|
||||
- Slice de codigo y pruebas: 313 lineas autoradas; 389 en total con metadatos e historial, dentro del presupuesto de 400.
|
||||
|
||||
**Estado final:**
|
||||
- Unit 2 completada sin modificar dispatcher/reconciler, crear commits, publicar ni desplegar. Incidencias: una ejecucion inicial pudo activar dotenv sin exponer contenido y `npm test` omite ahora las pruebas raiz por expansion del glob; la suite raiz explicita paso.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `src/modules/catalog/repository.ts`
|
||||
- `tests/catalog/repository-ocr.test.ts`
|
||||
- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 2 Corrective Rerun - Correccion de la verificacion canonica
|
||||
|
||||
**Agente:** **Subagente OCR Unit 2 Corrective Rerun**
|
||||
**Rol/responsabilidad:** Ejecutar la unica repeticion correctiva de Unit 2 para restaurar la cobertura canonica de pruebas y renovar su evidencia, sin implementar Unit 3.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_f60564170ffec1x6E520GuSMHd`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Corregido de forma minima el script `npm test`: ahora pasa explicitamente los patrones de suites raiz y anidadas a `tsx --test`.
|
||||
- Repetidos todos los comandos directos de Unit 2 con `NODE_ENV=test` para evitar la omision insegura anterior.
|
||||
- Fusionada la evidencia correctiva con el progreso acumulado de Units 1 y 2; las tareas 1.1–1.4 permanecen completadas y Unit 3 no se inicio.
|
||||
- Registrada la excepcion `size:exception` autorizada por el maintainer exclusivamente para las 511 lineas nativas de Unit 2; el presupuesto correctivo autorizado fue de 600 lineas y las unidades posteriores conservan el limite normal de 400.
|
||||
|
||||
**Validacion:**
|
||||
- Revision focalizada de repositorio: 6/6 pruebas correctas.
|
||||
- Suites de catalogo: 10/10 pruebas correctas.
|
||||
- Suites raiz explicitas: 25/25 pruebas correctas.
|
||||
- `npm test` canonico: 35/35 pruebas correctas, demostrando la ejecucion conjunta de 25 pruebas raiz y 10 anidadas.
|
||||
- `npm run check` y `git diff --check`: salida correcta.
|
||||
- Revision fallida corregida: `sha256:9433c0cd6628f36f4c96a27d41c6d429f4303d038ffe928f297367c274f21c2f`.
|
||||
- Nueva revision de evidencia: `sha256:a19119f84381447f54a6223e2a203acaf9d3ae08b6ae8f585e3bd52cd510909b`.
|
||||
|
||||
**Estado final:**
|
||||
- Repeticion correctiva de Unit 2 completada con todos los gates solicitados en verde.
|
||||
- No se hizo commit, push, PR ni deploy; no se accedio a `.env*`, `llaves` ni `backups/`.
|
||||
- Unit 3 permanece pendiente.
|
||||
|
||||
**Archivos modificados en esta correccion:**
|
||||
- `package.json`
|
||||
- `openspec/changes/ocr-ingest-integration/apply-progress.md`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 3 Extraction - Extraccion PDF por pagina
|
||||
|
||||
**Agente:** **Subagente OCR Unit 3 Extraction**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 3 del cambio SDD `ocr-ingest-integration` mediante TDD estricto.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `SIN_SESION_EN_ESTE_WORKSPACE`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Creada primero la prueba focalizada de extraccion PDF y routing; RED valido por ausencia de `parsePdfPages`.
|
||||
- Añadidos un fixture PDF de tres paginas, el spike Node 22 y extraccion por pagina con texto y SHA-256 estable.
|
||||
- Probado que `requirements.txt` y Markdown se leen como datos sin ejecutarse, mientras `CMakeLists.txt`, MDX y shell siguen sin soporte.
|
||||
- Marcadas 2.1 y 2.3; 2.2 permanece pendiente porque sus aserciones de deteccion, composicion y riesgos pertenecen a Unit 4.
|
||||
|
||||
**Validacion:**
|
||||
- Suite focalizada: 5/5; spike: tres paginas ordenadas con pagina 2 vacia; `npm test`: 40/40.
|
||||
- `npm run check` y `git diff --check`: correctos.
|
||||
- Slice nativo antes de metadatos: 225 lineas, dentro del limite de 400.
|
||||
|
||||
**Estado final:**
|
||||
- Unit 3 completada sin iniciar 2.4/2.5, sin commit, push, PR o deploy y sin acceder a rutas restringidas.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `src/modules/parsers/parser-registry.ts`
|
||||
- `scripts/spike-pdfjs.ts`
|
||||
- `tests/parsers/pdf-pages.test.ts`
|
||||
- `tests/fixtures/ocr/native-three-pages.pdf`
|
||||
- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 4 Detection and Composition - Deteccion y composicion OCR
|
||||
|
||||
**Agente:** **Subagente OCR Unit 4 Detection and Composition**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 4 del cambio SDD `ocr-ingest-integration` mediante TDD estricto.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `SIN_SESION_EN_ESTE_WORKSPACE`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Creada primero la prueba focalizada; RED valido por ausencia de los modulos OCR.
|
||||
- Implementadas las tareas 2.2, 2.4 y 2.5: deteccion y quality gates exactos, composicion determinista, hashes canonicos y riesgos sin autocorreccion.
|
||||
|
||||
**Validacion:**
|
||||
- Suite focalizada 5/5, harness PDF mixto 1/1 y canonica 45/45; check y comprobaciones de espacios correctos; slice nativo: 322 lineas.
|
||||
|
||||
**Estado final:**
|
||||
- Unit 4 completada; Unit 5 no iniciada y sin commit, push, PR, deploy o acceso a rutas restringidas.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `src/modules/ocr/{detection,composition}.ts`
|
||||
- `tests/ocr/detection.test.ts`
|
||||
- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}`, `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 5 Reintento y Correccion - Servicio OCR HTTP validado
|
||||
|
||||
**Agente:** **Subagente OCR Unit 5 Reintento y Correccion**
|
||||
**Rol/responsabilidad:** Implementar y corregir exclusivamente Unit 5, tareas 3.1 y 3.2, mediante TDD estricto y entorno virtual local autorizado.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session IDs OpenCode:** retry `ses_f5fa0eea0ffegi8n6KEhZDWYzU`; correction `ses_f5edabfa9ffeDm6JTyPnX7Bh5M`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Creado el manifiesto Python fijado, el entorno ignorado y retenido `ocr-service/.venv`, las pruebas HTTP RED-first y la API FastAPI con cola SQLite, autenticacion, idempotencia, allowlist, limites y health.
|
||||
- Diagnosticado el harness fallido: `&` envio la lista shell previa a un subshell y dejo vacios `KEY`, `REQUEST` y `CONFLICT` en los curls; se reprodujo `400 INVALID_REQUEST` y se corrigio solo el limite de backgrounding, sin cambiar la implementacion.
|
||||
|
||||
**Validacion y estado final:**
|
||||
- RED valido por ausencia de `app`; GREEN focalizado: 8/8 pruebas aprobadas.
|
||||
- Se conserva la evidencia fallida `sha256:4a5c8ed8d8b302f7ff654d4f40fe0f13049bfdd29db4c76576f3847c0a9be15a`; el harness corregido devolvio exactamente `401/202/409` con cuerpos contractuales.
|
||||
- `npm test` 45/45, `npm run check`, compileall y espacios correctos; servidor y temporales eliminados, `.venv` retenido; tareas 3.1/3.2 completadas. Revision: `sha256:0d8b2c57bbab967473eaf99a7ca900168554e64254fa80197b3626b4570b58d0`.
|
||||
|
||||
**Archivos modificados:** `.gitignore`, `ocr-service/{README.md,requirements.txt,app/,tests/}`, `openspec/changes/ocr-ingest-integration/apply-progress.md`, `docs/HISTORIAL_SESIONES.md`.
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 6 Render - Renderizado y runtime PaddleOCR
|
||||
|
||||
**Agente:** **Subagente OCR Unit 6 Render**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 6, tareas 3.3 y 3.4, mediante TDD estricto, sin iniciar Unit 7 ni realizar acciones de entrega Git.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_f5e8c95a3ffeugBz7od2TtJr4n`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Implementados renderizado PDF determinista a 200 DPI, limite previo de 25 megapixeles, adaptador PaddleOCR, IDs de linea, metricas y esquema de resultado contractual.
|
||||
- Añadidas pruebas RED-first con motor falso determinista y cobertura de paginas seleccionadas, limites, payload Paddle y contrato de imagen.
|
||||
- Creada imagen CPU no privilegiada con PaddleOCR 3.4.0/PaddlePaddle 3.2.2, modelos baked, un worker, volumen privado y limites operativos documentados.
|
||||
- Restringido el contexto Docker con una lista de inclusion minima para no enviar rutas sensibles o ajenas al servicio.
|
||||
|
||||
**Validacion y estado final:**
|
||||
- RED valido por ausencia de `app.engine`; GREEN focalizado 5/5 y suite OCR completa 13/13.
|
||||
- Build Docker corregido tras detectar `libGL.so.1` ausente; harness offline limitado a 3 CPU/5 GiB cargo modelos baked y genero PNG 1700x2200 con esquema `1`.
|
||||
- Un rebuild opcional posterior agoto el almacenamiento Docker; se revirtio esa unica optimizacion no validada, se restauro exactamente el Dockerfile ya probado y se elimino la imagen local.
|
||||
- `npm test` 45/45, `npm run check`, `compileall` y espacios correctos; 397 lineas nativas, dentro del limite de 400. Tareas 3.3/3.4 completadas; Unit 7 no iniciada.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `ocr-service/{Dockerfile,Dockerfile.dockerignore,README.md,requirements.txt}`
|
||||
- `ocr-service/app/{engine.py,main.py,models.py,render.py}`
|
||||
- `ocr-service/tests/test_render.py`
|
||||
- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
||||
---
|
||||
|
||||
### 2026-09-14 - Subagente OCR Unit 7 Client - Cliente y artefactos OCR
|
||||
|
||||
**Agente:** **Subagente OCR Unit 7 Client**
|
||||
**Rol/responsabilidad:** Implementar exclusivamente Unit 7, tareas 4.2 y 4.3 y la parte de reintentos/integridad de 4.1, mediante TDD estricto, sin iniciar dispatcher ni routing de Unit 8.
|
||||
**Modelo:** openai/gpt-5.6-sol
|
||||
**Session ID OpenCode:** `ses_f5e4fa3e6ffeabETlcJxsURUKb`
|
||||
**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG`
|
||||
|
||||
**Trabajo realizado:**
|
||||
- Añadido el cliente OCR privado con clave idempotente estable, dos reenvios transitorios como maximo, clasificacion terminal de fallos deterministas, presion `429` reintentable, polling acotado y validacion estricta de identidad/esquema/resultados.
|
||||
- Añadido el gestor de artefactos con originales y manifiesto canonico durables, permisos `0600`, IDs UUIDv5, hashes verificables, contencion de rutas y barrido de huerfanos antiguo y conservador.
|
||||
- Mantenida 4.1 pendiente: Unit 7 solo cubre reintentos e integridad; los casos HTTP de ingesta, estado y fallo cerrado pertenecen a Unit 8.
|
||||
|
||||
**Validacion y estado final:**
|
||||
- RED valido por ausencia de los dos modulos; GREEN focalizado 6/6 y harness determinista 1/1 con tres intentos, misma clave y backoff exacto de 2/4 segundos.
|
||||
- `npm test` 51/51, `npm run check`, `npm run build` y comprobaciones de espacios correctos; temporales eliminados y ningun proceso o llamada OCR externa iniciados.
|
||||
- El slice funcional minimo es de 506 lineas autoradas y la contabilidad nativa total con metadatos obligatorios es de 585 lineas. El maintainer aprobo explicitamente `size:exception` porque cliente, artefactos y pruebas forman una unidad cohesiva; no se comprimio codigo ni se inicio un segundo particionado.
|
||||
- El intento nativo se reinicio para validar la excepcion sin modificar produccion ni pruebas. El nuevo work unit es `unit-7-size-exception-validation` y remedia la evidencia fallida `sha256:dd224c3864adc8380dc7d9fd147cb448f5da8138277f3f346a35d7eeb3aca158`.
|
||||
- No se hizo commit, push, PR, review nativa ni trabajo de Unit 8; no se accedio a rutas restringidas y se preservo `ocr-service/.venv`.
|
||||
|
||||
**Archivos modificados:**
|
||||
- `src/modules/ocr/{client,artifacts}.ts`
|
||||
- `tests/ocr/client.test.ts`
|
||||
- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}`
|
||||
- `docs/HISTORIAL_SESIONES.md`
|
||||
|
|
|
|||
61
migrations/002_ocr_review.sql
Normal file
61
migrations/002_ocr_review.sql
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
CREATE TABLE IF NOT EXISTS rag_ocr_jobs (
|
||||
job_id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
version_id uuid NOT NULL,
|
||||
document_id text NOT NULL,
|
||||
remote_job_id text NULL,
|
||||
remote_idempotency_key text NOT NULL,
|
||||
state text NOT NULL CHECK (state IN ('queued', 'running', 'succeeded', 'failed')),
|
||||
requested_pages integer[] NOT NULL,
|
||||
completed_pages integer NOT NULL DEFAULT 0 CHECK (completed_pages >= 0),
|
||||
config_version text NOT NULL,
|
||||
attempt_count integer NOT NULL DEFAULT 0 CHECK (attempt_count >= 0),
|
||||
heartbeat_at timestamptz NULL,
|
||||
lease_expires_at timestamptz NULL,
|
||||
next_attempt_at timestamptz NULL,
|
||||
error_code text NULL,
|
||||
error_detail text NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
started_at timestamptz NULL,
|
||||
completed_at timestamptz NULL,
|
||||
UNIQUE (version_id, document_id),
|
||||
FOREIGN KEY (version_id, document_id)
|
||||
REFERENCES rag_version_documents(version_id, document_id) ON DELETE RESTRICT
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS rag_document_pages (
|
||||
version_id uuid NOT NULL,
|
||||
document_id text NOT NULL,
|
||||
page_number integer NOT NULL CHECK (page_number > 0),
|
||||
extraction_method text NOT NULL CHECK (extraction_method IN ('native', 'ocr', 'blank')),
|
||||
native_text_hash char(64) NULL,
|
||||
ocr_text_hash char(64) NULL,
|
||||
candidate_text_hash char(64) NULL,
|
||||
reviewed_text_hash char(64) NULL,
|
||||
metrics jsonb NOT NULL DEFAULT '{}',
|
||||
risk_tokens jsonb NOT NULL DEFAULT '[]',
|
||||
blocked_reason text NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (version_id, document_id, page_number),
|
||||
FOREIGN KEY (version_id, document_id)
|
||||
REFERENCES rag_version_documents(version_id, document_id) ON DELETE RESTRICT
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS rag_review_corrections (
|
||||
correction_id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
version_id uuid NOT NULL,
|
||||
document_id text NOT NULL,
|
||||
page_number integer NOT NULL,
|
||||
line_id text NOT NULL,
|
||||
expected_line_hash char(64) NOT NULL,
|
||||
replacement_text text NOT NULL,
|
||||
reviewed_by text NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
FOREIGN KEY (version_id, document_id, page_number)
|
||||
REFERENCES rag_document_pages(version_id, document_id, page_number) ON DELETE RESTRICT,
|
||||
UNIQUE (version_id, document_id, page_number, line_id)
|
||||
);
|
||||
|
||||
CREATE UNIQUE INDEX IF NOT EXISTS rag_one_pending_ocr_identity
|
||||
ON rag_source_versions(source_id, original_manifest_hash, processing_fingerprint, metadata_hash)
|
||||
WHERE source_content_hash IS NULL
|
||||
AND state IN ('pending', 'indexing', 'review_required');
|
||||
32
ocr-service/Dockerfile
Normal file
32
ocr-service/Dockerfile
Normal file
|
|
@ -0,0 +1,32 @@
|
|||
FROM python:3.11.13-slim-bookworm
|
||||
|
||||
LABEL resource.cpu.max="3" \
|
||||
resource.memory.max="5GiB"
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
OCR_JOBS_DB="/data/jobs/jobs.db" \
|
||||
OCR_LOAD_ENGINE="1" \
|
||||
HOME="/opt/ocr-home"
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install --yes --no-install-recommends libgl1 libglib2.0-0 libgomp1 \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& groupadd --system ocr && useradd --system --gid ocr --home-dir /opt/ocr-home ocr \
|
||||
&& mkdir -p /opt/ocr-home /data/jobs \
|
||||
&& chown -R ocr:ocr /opt/ocr-home /data/jobs
|
||||
|
||||
WORKDIR /srv/ocr
|
||||
COPY ocr-service/requirements.txt ./requirements.txt
|
||||
RUN python -m pip install --no-cache-dir --requirement requirements.txt
|
||||
|
||||
COPY ocr-service/app ./app
|
||||
USER ocr
|
||||
RUN python -m app.models
|
||||
|
||||
VOLUME ["/data/jobs"]
|
||||
EXPOSE 8000
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=90s --retries=3 \
|
||||
CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health/ready', timeout=4)"
|
||||
|
||||
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]
|
||||
5
ocr-service/Dockerfile.dockerignore
Normal file
5
ocr-service/Dockerfile.dockerignore
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
**
|
||||
!ocr-service/
|
||||
!ocr-service/app/
|
||||
!ocr-service/app/**
|
||||
!ocr-service/requirements.txt
|
||||
27
ocr-service/README.md
Normal file
27
ocr-service/README.md
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
# OCR Service Development Environment
|
||||
|
||||
Unit 5 uses the repository-local virtual environment `ocr-service/.venv`. It is intentionally retained between runs and ignored by Git. Do not install these dependencies globally.
|
||||
|
||||
## Install
|
||||
|
||||
From the repository root:
|
||||
|
||||
```bash
|
||||
ocr-service/.venv/bin/python -m pip install -r ocr-service/requirements.txt
|
||||
```
|
||||
|
||||
Activate it with `source ocr-service/.venv/bin/activate`.
|
||||
|
||||
## Test
|
||||
|
||||
```bash
|
||||
ocr-service/.venv/bin/python -m pytest ocr-service/tests
|
||||
```
|
||||
|
||||
The container build downloads the pinned PaddleOCR models. Runtime startup loads those baked models before readiness can report healthy. Deploy one replica with a maximum of 3 CPU and 5 GiB RAM; Docker image labels document these platform-enforced limits.
|
||||
|
||||
Build from the repository root so the Dockerfile can use the service-scoped paths:
|
||||
|
||||
```bash
|
||||
docker build -f ocr-service/Dockerfile -t rag-ocr-service .
|
||||
```
|
||||
0
ocr-service/app/__init__.py
Normal file
0
ocr-service/app/__init__.py
Normal file
48
ocr-service/app/engine.py
Normal file
48
ocr-service/app/engine.py
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
from dataclasses import dataclass
|
||||
from typing import Any, Protocol
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EngineLine:
|
||||
text: str
|
||||
confidence: float
|
||||
bbox: tuple[int, int, int, int]
|
||||
|
||||
|
||||
class OcrEngine(Protocol):
|
||||
def recognize(self, image: object) -> list[EngineLine]: ...
|
||||
|
||||
|
||||
class PaddleOcrEngine:
|
||||
def __init__(self, pipeline: Any | None = None) -> None:
|
||||
self._injected = pipeline is not None
|
||||
if pipeline is None:
|
||||
from paddleocr import PaddleOCR
|
||||
|
||||
pipeline = PaddleOCR(
|
||||
text_detection_model_name="PP-OCRv5_mobile_det",
|
||||
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
|
||||
use_doc_orientation_classify=False,
|
||||
use_doc_unwarping=False,
|
||||
use_textline_orientation=False,
|
||||
device="cpu",
|
||||
)
|
||||
self.pipeline = pipeline
|
||||
|
||||
def recognize(self, image: object) -> list[EngineLine]:
|
||||
if not self._injected:
|
||||
import numpy
|
||||
|
||||
image = numpy.asarray(image)
|
||||
lines: list[EngineLine] = []
|
||||
for prediction in self.pipeline.predict(image):
|
||||
payload = prediction.json() if callable(prediction.json) else prediction.json
|
||||
result = payload.get("res", payload)
|
||||
texts = result.get("rec_texts", [])
|
||||
scores = result.get("rec_scores", [])
|
||||
boxes = result.get("rec_boxes", [])
|
||||
if hasattr(boxes, "tolist"):
|
||||
boxes = boxes.tolist()
|
||||
for text, score, box in zip(texts, scores, boxes, strict=True):
|
||||
lines.append(EngineLine(str(text), float(score), tuple(int(value) for value in box)))
|
||||
return lines
|
||||
179
ocr-service/app/main.py
Normal file
179
ocr-service/app/main.py
Normal file
|
|
@ -0,0 +1,179 @@
|
|||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import threading
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any
|
||||
|
||||
from fastapi import Depends, FastAPI, File, Form, Header, HTTPException, Response, UploadFile
|
||||
|
||||
from .models import load_runtime_engine
|
||||
|
||||
|
||||
MAX_UPLOAD_BYTES = 50 * 1024 * 1024
|
||||
MAX_PAGES = 100
|
||||
QUEUE_CAPACITY = 3
|
||||
ALLOWED_CONFIG = {
|
||||
"languages": ["es", "en"],
|
||||
"dpi": 200,
|
||||
"engine": "paddleocr",
|
||||
"engineVersion": "3.4.0",
|
||||
"runtimeVersion": "3.2.2",
|
||||
"configVersion": "ocr-v1",
|
||||
"returnLayout": True,
|
||||
}
|
||||
REQUEST_FIELDS = {"documentSha256", "pages", *ALLOWED_CONFIG}
|
||||
|
||||
|
||||
def fail(status: int, code: str, message: str, retryable: bool = False, headers: dict[str, str] | None = None) -> None:
|
||||
raise HTTPException(status, {"code": code, "message": message, "retryable": retryable}, headers)
|
||||
|
||||
|
||||
def page_hash(pages: list[int]) -> str:
|
||||
value = json.dumps(pages, separators=(",", ":")).encode()
|
||||
return hashlib.sha256(value).hexdigest()
|
||||
|
||||
|
||||
class JobQueue:
|
||||
def __init__(self, path: str | Path):
|
||||
self.connection = sqlite3.connect(str(path), check_same_thread=False)
|
||||
self.connection.row_factory = sqlite3.Row
|
||||
self.lock = threading.Lock()
|
||||
self.connection.execute(
|
||||
"CREATE TABLE IF NOT EXISTS jobs (job_id TEXT PRIMARY KEY, idempotency_key TEXT UNIQUE, "
|
||||
"payload_hash TEXT, document_sha256 TEXT, pages TEXT, status TEXT, created_at TEXT)"
|
||||
)
|
||||
self.connection.commit()
|
||||
|
||||
def depth(self) -> int:
|
||||
row = self.connection.execute("SELECT count(*) AS count FROM jobs WHERE status IN ('queued','running')").fetchone()
|
||||
return int(row["count"])
|
||||
|
||||
def contains(self, key: str) -> bool:
|
||||
return self.connection.execute("SELECT 1 FROM jobs WHERE idempotency_key=?", (key,)).fetchone() is not None
|
||||
|
||||
@staticmethod
|
||||
def ack(row: sqlite3.Row) -> dict[str, Any]:
|
||||
return {
|
||||
"jobId": row["job_id"],
|
||||
"status": "queued",
|
||||
"documentSha256": row["document_sha256"],
|
||||
"requestedPages": json.loads(row["pages"]),
|
||||
"configVersion": "ocr-v1",
|
||||
"createdAt": row["created_at"],
|
||||
}
|
||||
|
||||
def submit(self, key: str, request: dict[str, Any]) -> dict[str, Any]:
|
||||
payload_hash = hashlib.sha256(json.dumps(request, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
|
||||
with self.lock:
|
||||
row = self.connection.execute("SELECT * FROM jobs WHERE idempotency_key=?", (key,)).fetchone()
|
||||
if row:
|
||||
if row["payload_hash"] != payload_hash:
|
||||
fail(409, "IDEMPOTENCY_CONFLICT", "The idempotency key is already bound to another request")
|
||||
return self.ack(row)
|
||||
if self.depth() >= QUEUE_CAPACITY:
|
||||
fail(429, "QUEUE_FULL", "The OCR queue is full", True, {"Retry-After": "2"})
|
||||
values = (
|
||||
f"ocr_{uuid.uuid4()}", key, payload_hash, request["documentSha256"],
|
||||
json.dumps(request["pages"]), "queued", datetime.now(timezone.utc).isoformat(),
|
||||
)
|
||||
self.connection.execute("INSERT INTO jobs VALUES (?,?,?,?,?,?,?)", values)
|
||||
self.connection.commit()
|
||||
return self.ack(self.connection.execute("SELECT * FROM jobs WHERE job_id=?", (values[0],)).fetchone())
|
||||
|
||||
def status(self, job_id: str) -> dict[str, Any] | None:
|
||||
row = self.connection.execute("SELECT * FROM jobs WHERE job_id=?", (job_id,)).fetchone()
|
||||
if not row:
|
||||
return None
|
||||
return {"jobId": job_id, "status": row["status"], "completedPages": 0,
|
||||
"totalPages": len(json.loads(row["pages"])), "error": None}
|
||||
|
||||
def delete(self, job_id: str) -> None:
|
||||
with self.lock:
|
||||
self.connection.execute("DELETE FROM jobs WHERE job_id=?", (job_id,))
|
||||
self.connection.commit()
|
||||
|
||||
|
||||
def create_app(
|
||||
token: str,
|
||||
db_path: str | Path = ":memory:",
|
||||
max_upload_bytes: int = MAX_UPLOAD_BYTES,
|
||||
engine_ready: bool = False,
|
||||
) -> FastAPI:
|
||||
application = FastAPI(title="Private OCR Service", docs_url=None, redoc_url=None)
|
||||
queue = JobQueue(db_path)
|
||||
|
||||
def authorize(authorization: Annotated[str | None, Header()] = None) -> None:
|
||||
scheme, _, supplied = (authorization or "").partition(" ")
|
||||
if not token or scheme != "Bearer" or not hmac.compare_digest(supplied, token):
|
||||
fail(401, "UNAUTHORIZED", "Valid bearer authorization is required", headers={"WWW-Authenticate": "Bearer"})
|
||||
|
||||
@application.get("/health/live")
|
||||
def live() -> dict[str, str]:
|
||||
return {"status": "ok"}
|
||||
|
||||
@application.get("/health/ready")
|
||||
def ready(response: Response) -> dict[str, int | bool]:
|
||||
if not engine_ready:
|
||||
response.status_code = 503
|
||||
return {"ready": engine_ready, "queueDepth": queue.depth(), "queueCapacity": QUEUE_CAPACITY, "concurrency": 1}
|
||||
|
||||
@application.post("/v1/jobs", status_code=202, dependencies=[Depends(authorize)])
|
||||
async def create_job(
|
||||
file: Annotated[UploadFile, File()], request: Annotated[str, Form()],
|
||||
idempotency_key: Annotated[str | None, Header(alias="Idempotency-Key")] = None,
|
||||
) -> dict[str, Any]:
|
||||
content = await file.read(max_upload_bytes + 1)
|
||||
if len(content) > max_upload_bytes:
|
||||
fail(413, "UPLOAD_LIMIT_EXCEEDED", "PDF exceeds the upload limit")
|
||||
try:
|
||||
payload = json.loads(request)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
fail(400, "INVALID_REQUEST", "Request must be valid JSON")
|
||||
if not isinstance(payload, dict) or set(payload) != REQUEST_FIELDS:
|
||||
fail(400, "INVALID_REQUEST", "Request fields do not match the contract")
|
||||
pages = payload["pages"]
|
||||
if not isinstance(pages, list) or not pages or any(type(page) is not int or page < 1 for page in pages):
|
||||
fail(422, "INVALID_PAGES", "Pages must be positive one-based integers")
|
||||
if len(pages) > MAX_PAGES:
|
||||
fail(413, "PAGE_LIMIT_EXCEEDED", "OCR jobs accept at most 100 pages")
|
||||
if pages != sorted(set(pages)):
|
||||
fail(422, "INVALID_PAGES", "Pages must be unique and ordered")
|
||||
if any(payload[name] != value for name, value in ALLOWED_CONFIG.items()):
|
||||
fail(400, "CONFIG_NOT_ALLOWED", "OCR configuration is not allowlisted")
|
||||
digest = hashlib.sha256(content).hexdigest()
|
||||
if payload["documentSha256"] != digest:
|
||||
fail(422, "INTEGRITY_MISMATCH", "PDF bytes do not match documentSha256")
|
||||
if file.content_type != "application/pdf" or not content.startswith(b"%PDF-") or b"/Encrypt" in content:
|
||||
fail(422, "UNSUPPORTED_PDF", "PDF is corrupt, encrypted, or unsupported")
|
||||
expected_key = f'{digest}:ocr-v1:{page_hash(pages)}'
|
||||
if idempotency_key != expected_key and not queue.contains(idempotency_key or ""):
|
||||
fail(400, "INVALID_IDEMPOTENCY_KEY", "Idempotency-Key does not match request identity")
|
||||
return queue.submit(idempotency_key or "", payload)
|
||||
|
||||
@application.get("/v1/jobs/{job_id}", dependencies=[Depends(authorize)])
|
||||
def get_job(job_id: str) -> dict[str, Any]:
|
||||
status = queue.status(job_id)
|
||||
if status is None:
|
||||
fail(404, "JOB_NOT_FOUND", "OCR job does not exist")
|
||||
return status
|
||||
|
||||
@application.delete("/v1/jobs/{job_id}", status_code=204, dependencies=[Depends(authorize)])
|
||||
def delete_job(job_id: str) -> Response:
|
||||
queue.delete(job_id)
|
||||
return Response(status_code=204)
|
||||
|
||||
return application
|
||||
|
||||
|
||||
runtime_engine = load_runtime_engine()
|
||||
|
||||
app = create_app(
|
||||
os.getenv("OCR_INTERNAL_TOKEN", ""),
|
||||
os.getenv("OCR_JOBS_DB", ":memory:"),
|
||||
engine_ready=runtime_engine is not None,
|
||||
)
|
||||
17
ocr-service/app/models.py
Normal file
17
ocr-service/app/models.py
Normal file
|
|
@ -0,0 +1,17 @@
|
|||
import os
|
||||
|
||||
from .engine import PaddleOcrEngine
|
||||
|
||||
|
||||
def load_runtime_engine() -> PaddleOcrEngine | None:
|
||||
if os.getenv("OCR_LOAD_ENGINE") != "1":
|
||||
return None
|
||||
try:
|
||||
return PaddleOcrEngine()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
PaddleOcrEngine()
|
||||
print("PaddleOCR models loaded")
|
||||
123
ocr-service/app/render.py
Normal file
123
ocr-service/app/render.py
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
import io
|
||||
import math
|
||||
import statistics
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable
|
||||
|
||||
from .engine import EngineLine, OcrEngine
|
||||
|
||||
|
||||
RENDER_DPI = 200
|
||||
MAX_RENDER_PIXELS = 25_000_000
|
||||
|
||||
|
||||
class PdfRenderError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RenderedPage:
|
||||
page: int
|
||||
width: int
|
||||
height: int
|
||||
png: bytes
|
||||
image: object
|
||||
|
||||
|
||||
def render_pdf_pages(pdf: bytes, pages: list[int], max_pixels: int = MAX_RENDER_PIXELS) -> list[RenderedPage]:
|
||||
import pypdfium2
|
||||
|
||||
if not pages or pages != sorted(set(pages)) or any(type(page) is not int or page < 1 for page in pages):
|
||||
raise PdfRenderError("PDF pages must be unique, ordered, one-based integers")
|
||||
try:
|
||||
document = pypdfium2.PdfDocument(pdf)
|
||||
except Exception as error:
|
||||
raise PdfRenderError("PDF cannot be opened for deterministic rendering") from error
|
||||
|
||||
rendered: list[RenderedPage] = []
|
||||
try:
|
||||
for page_number in pages:
|
||||
if page_number > len(document):
|
||||
raise PdfRenderError(f"PDF page {page_number} does not exist")
|
||||
page = document[page_number - 1]
|
||||
try:
|
||||
page_width, page_height = page.get_size()
|
||||
width = math.ceil(page_width * RENDER_DPI / 72)
|
||||
height = math.ceil(page_height * RENDER_DPI / 72)
|
||||
if width * height > max_pixels:
|
||||
raise PdfRenderError("Rendered page exceeds the 25 megapixels limit")
|
||||
bitmap = page.render(scale=RENDER_DPI / 72)
|
||||
try:
|
||||
image = bitmap.to_pil()
|
||||
output = io.BytesIO()
|
||||
image.save(output, format="PNG")
|
||||
rendered.append(RenderedPage(page_number, image.width, image.height, output.getvalue(), image))
|
||||
finally:
|
||||
bitmap.close()
|
||||
finally:
|
||||
page.close()
|
||||
finally:
|
||||
document.close()
|
||||
return rendered
|
||||
|
||||
|
||||
def _metrics(lines: list[EngineLine], text: str) -> dict[str, int | float]:
|
||||
confidences = sorted(line.confidence for line in lines)
|
||||
return {
|
||||
"lineCount": len(lines),
|
||||
"nonWhitespaceCharacters": sum(not character.isspace() for character in text),
|
||||
"medianConfidence": statistics.median(confidences) if confidences else 0.0,
|
||||
"p10Confidence": confidences[math.floor((len(confidences) - 1) * 0.1)] if confidences else 0.0,
|
||||
"lowConfidenceLineRatio": sum(value < 0.5 for value in confidences) / len(confidences) if confidences else 0.0,
|
||||
}
|
||||
|
||||
|
||||
def process_pdf(
|
||||
job_id: str,
|
||||
document_sha256: str,
|
||||
pdf: bytes,
|
||||
requested_pages: list[int],
|
||||
engine: OcrEngine,
|
||||
processing_ms: Callable[[int], int] | None = None,
|
||||
) -> dict:
|
||||
results = []
|
||||
for rendered in render_pdf_pages(pdf, requested_pages):
|
||||
started = time.perf_counter_ns()
|
||||
indexed = list(enumerate(engine.recognize(rendered.image), start=1))
|
||||
indexed.sort(key=lambda item: (item[1].bbox[1], item[1].bbox[0], item[0]))
|
||||
lines = [line for _, line in indexed]
|
||||
text = "\n".join(line.text for line in lines)
|
||||
serialized_lines = [
|
||||
{
|
||||
"lineId": f"p{rendered.page}-l{index:02d}-{'-'.join(map(str, line.bbox))}",
|
||||
"text": line.text,
|
||||
"confidence": line.confidence,
|
||||
"bbox": list(line.bbox),
|
||||
}
|
||||
for index, line in enumerate(lines, start=1)
|
||||
]
|
||||
elapsed = math.ceil((time.perf_counter_ns() - started) / 1_000_000)
|
||||
results.append({
|
||||
"page": rendered.page,
|
||||
"width": rendered.width,
|
||||
"height": rendered.height,
|
||||
"processingMs": processing_ms(rendered.page) if processing_ms else elapsed,
|
||||
"text": text,
|
||||
"metrics": _metrics(lines, text),
|
||||
"lines": serialized_lines,
|
||||
})
|
||||
return {
|
||||
"schemaVersion": "1",
|
||||
"jobId": job_id,
|
||||
"documentSha256": document_sha256,
|
||||
"engine": {
|
||||
"name": "paddleocr",
|
||||
"version": "3.4.0",
|
||||
"runtime": "paddlepaddle-3.2.2",
|
||||
"device": "cpu",
|
||||
"configVersion": "ocr-v1",
|
||||
"dpi": RENDER_DPI,
|
||||
},
|
||||
"pages": results,
|
||||
}
|
||||
10
ocr-service/requirements.txt
Normal file
10
ocr-service/requirements.txt
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
fastapi==0.116.1
|
||||
httpx==0.28.1
|
||||
numpy==2.2.6
|
||||
paddleocr==3.4.0
|
||||
paddlepaddle==3.2.2
|
||||
Pillow==11.3.0
|
||||
pytest==8.4.1
|
||||
pypdfium2==4.30.0
|
||||
python-multipart==0.0.20
|
||||
uvicorn==0.35.0
|
||||
133
ocr-service/tests/test_api.py
Normal file
133
ocr-service/tests/test_api.py
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parents[1]))
|
||||
|
||||
from app.main import create_app
|
||||
|
||||
|
||||
TOKEN = "unit-5-test-token"
|
||||
PDF = b"%PDF-1.4\nunit five\n%%EOF"
|
||||
|
||||
|
||||
def request_for(pdf: bytes = PDF, pages: list[int] | None = None) -> dict:
|
||||
return {
|
||||
"documentSha256": hashlib.sha256(pdf).hexdigest(),
|
||||
"pages": pages or [1, 2],
|
||||
"languages": ["es", "en"],
|
||||
"dpi": 200,
|
||||
"engine": "paddleocr",
|
||||
"engineVersion": "3.4.0",
|
||||
"runtimeVersion": "3.2.2",
|
||||
"configVersion": "ocr-v1",
|
||||
"returnLayout": True,
|
||||
}
|
||||
|
||||
|
||||
def key_for(request: dict) -> str:
|
||||
pages = json.dumps(request["pages"], separators=(",", ":")).encode()
|
||||
return f'{request["documentSha256"]}:ocr-v1:{hashlib.sha256(pages).hexdigest()}'
|
||||
|
||||
|
||||
def submit(client: TestClient, request: dict, pdf: bytes = PDF, key: str | None = None):
|
||||
return client.post(
|
||||
"/v1/jobs",
|
||||
headers={"Authorization": f"Bearer {TOKEN}", "Idempotency-Key": key or key_for(request)},
|
||||
files={
|
||||
"file": ("input.pdf", pdf, "application/pdf"),
|
||||
"request": (None, json.dumps(request), "application/json"),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(tmp_path: Path) -> TestClient:
|
||||
return TestClient(create_app(TOKEN, tmp_path / "jobs.db", engine_ready=True))
|
||||
|
||||
|
||||
def test_auth_rejects_missing_and_wrong_bearer_and_health_exposes_no_secret(client: TestClient):
|
||||
live = client.get("/health/live")
|
||||
ready = client.get("/health/ready")
|
||||
assert live.json() == {"status": "ok"}
|
||||
assert ready.json() == {"ready": True, "queueDepth": 0, "queueCapacity": 3, "concurrency": 1}
|
||||
for authorization in (None, "Bearer wrong-token"):
|
||||
headers = {"Idempotency-Key": key_for(request_for())}
|
||||
if authorization:
|
||||
headers["Authorization"] = authorization
|
||||
response = client.post(
|
||||
"/v1/jobs",
|
||||
headers=headers,
|
||||
files={"file": ("input.pdf", PDF, "application/pdf"), "request": (None, json.dumps(request_for()))},
|
||||
)
|
||||
assert response.status_code == 401
|
||||
assert response.json()["detail"]["retryable"] is False
|
||||
assert TOKEN not in response.text
|
||||
|
||||
unavailable = TestClient(create_app(TOKEN, ":memory:", engine_ready=False)).get("/health/ready")
|
||||
assert unavailable.status_code == 503
|
||||
assert unavailable.json()["ready"] is False
|
||||
|
||||
|
||||
def test_auth_submission_is_idempotent_and_conflicting_payload_is_terminal(client: TestClient):
|
||||
request = request_for()
|
||||
first = submit(client, request)
|
||||
repeated = submit(client, request)
|
||||
conflict = request_for(pages=[1])
|
||||
conflicting = submit(client, conflict, key=key_for(request))
|
||||
assert first.status_code == repeated.status_code == 202
|
||||
assert first.json() == repeated.json()
|
||||
assert first.json()["requestedPages"] == [1, 2]
|
||||
assert conflicting.status_code == 409
|
||||
assert conflicting.json()["detail"] == {
|
||||
"code": "IDEMPOTENCY_CONFLICT",
|
||||
"message": "The idempotency key is already bound to another request",
|
||||
"retryable": False,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value"),
|
||||
[("languages", ["en"]), ("dpi", 300), ("engine", "tesseract"), ("returnLayout", False)],
|
||||
)
|
||||
def test_auth_allowlist_rejects_client_selected_configuration(client: TestClient, field: str, value: object):
|
||||
request = request_for()
|
||||
request[field] = value
|
||||
response = submit(client, request, key="different-key")
|
||||
assert response.status_code == 400
|
||||
assert response.json()["detail"]["retryable"] is False
|
||||
|
||||
|
||||
def test_auth_limits_corrupt_pdf_and_integrity_mismatch_are_terminal(tmp_path: Path):
|
||||
client = TestClient(create_app(TOKEN, tmp_path / "limited.db", max_upload_bytes=8, engine_ready=True))
|
||||
regular = TestClient(create_app(TOKEN, tmp_path / "regular.db", engine_ready=True))
|
||||
oversized = submit(client, request_for(PDF), PDF)
|
||||
too_many = submit(regular, request_for(pages=list(range(1, 102))))
|
||||
corrupt = submit(regular, request_for(b"not-pdf"), b"not-pdf")
|
||||
wrong_hash = request_for()
|
||||
wrong_hash["documentSha256"] = "0" * 64
|
||||
mismatch = submit(regular, wrong_hash, key=key_for(wrong_hash))
|
||||
assert oversized.status_code == too_many.status_code == 413
|
||||
assert corrupt.status_code == mismatch.status_code == 422
|
||||
assert mismatch.json()["detail"]["code"] == "INTEGRITY_MISMATCH"
|
||||
for response in (oversized, too_many, corrupt, mismatch):
|
||||
assert response.json()["detail"]["retryable"] is False
|
||||
assert "Retry-After" not in response.headers
|
||||
|
||||
|
||||
def test_auth_queue_pressure_is_retryable_and_status_and_delete_are_authenticated(client: TestClient):
|
||||
accepted = [submit(client, request_for(pdf), pdf) for pdf in (PDF, PDF + b"1", PDF + b"2")]
|
||||
pressure = submit(client, request_for(PDF + b"3"), PDF + b"3")
|
||||
assert [response.status_code for response in accepted] == [202, 202, 202]
|
||||
assert pressure.status_code == 429
|
||||
assert pressure.json()["detail"]["retryable"] is True
|
||||
assert pressure.headers["Retry-After"] == "2"
|
||||
job_id = accepted[0].json()["jobId"]
|
||||
status = client.get(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"})
|
||||
assert status.json() == {"jobId": job_id, "status": "queued", "completedPages": 0, "totalPages": 2, "error": None}
|
||||
assert client.delete(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"}).status_code == 204
|
||||
assert client.delete(f"/v1/jobs/{job_id}", headers={"Authorization": f"Bearer {TOKEN}"}).status_code == 204
|
||||
147
ocr-service/tests/test_render.py
Normal file
147
ocr-service/tests/test_render.py
Normal file
|
|
@ -0,0 +1,147 @@
|
|||
import hashlib
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parents[1]))
|
||||
|
||||
from app.engine import EngineLine, PaddleOcrEngine
|
||||
from app.render import PdfRenderError, process_pdf, render_pdf_pages
|
||||
|
||||
|
||||
FIXTURE = Path(__file__).parents[2] / "tests" / "fixtures" / "ocr" / "native-three-pages.pdf"
|
||||
DOCUMENT_SHA256 = hashlib.sha256(FIXTURE.read_bytes()).hexdigest()
|
||||
|
||||
|
||||
class DeterministicEngine:
|
||||
def __init__(self) -> None:
|
||||
self.calls = 0
|
||||
|
||||
def recognize(self, _image: object) -> list[EngineLine]:
|
||||
self.calls += 1
|
||||
if self.calls % 2 == 1:
|
||||
return [
|
||||
EngineLine("FATo7", 0.98, (120, 340, 245, 372)),
|
||||
EngineLine("second line", 0.74, (80, 410, 300, 450)),
|
||||
]
|
||||
return [EngineLine("page three", 0.91, (50, 60, 250, 100))]
|
||||
|
||||
|
||||
def test_render_produces_selected_200_dpi_png_pages() -> None:
|
||||
pages = render_pdf_pages(FIXTURE.read_bytes(), [1, 3])
|
||||
|
||||
assert [(page.page, page.width, page.height) for page in pages] == [
|
||||
(1, 1700, 2200),
|
||||
(3, 1700, 2200),
|
||||
]
|
||||
assert all(page.png.startswith(b"\x89PNG\r\n\x1a\n") for page in pages)
|
||||
|
||||
|
||||
def test_render_rejects_invalid_pages_and_pixel_limit_before_image_creation() -> None:
|
||||
with pytest.raises(PdfRenderError, match="one-based"):
|
||||
render_pdf_pages(FIXTURE.read_bytes(), [0])
|
||||
with pytest.raises(PdfRenderError, match="25 megapixels"):
|
||||
render_pdf_pages(FIXTURE.read_bytes(), [1], max_pixels=1_000_000)
|
||||
|
||||
|
||||
def test_render_builds_repeatable_result_schema_with_deterministic_engine() -> None:
|
||||
def run() -> dict:
|
||||
return process_pdf(
|
||||
job_id="ocr_repeatable",
|
||||
document_sha256=DOCUMENT_SHA256,
|
||||
pdf=FIXTURE.read_bytes(),
|
||||
requested_pages=[1, 3],
|
||||
engine=DeterministicEngine(),
|
||||
processing_ms=lambda _page: 17,
|
||||
)
|
||||
|
||||
first = run()
|
||||
assert first == run()
|
||||
assert first["schemaVersion"] == "1"
|
||||
assert first["documentSha256"] == DOCUMENT_SHA256
|
||||
assert first["engine"] == {
|
||||
"name": "paddleocr",
|
||||
"version": "3.4.0",
|
||||
"runtime": "paddlepaddle-3.2.2",
|
||||
"device": "cpu",
|
||||
"configVersion": "ocr-v1",
|
||||
"dpi": 200,
|
||||
}
|
||||
assert first["pages"][0] == {
|
||||
"page": 1,
|
||||
"width": 1700,
|
||||
"height": 2200,
|
||||
"processingMs": 17,
|
||||
"text": "FATo7\nsecond line",
|
||||
"metrics": {
|
||||
"lineCount": 2,
|
||||
"nonWhitespaceCharacters": 15,
|
||||
"medianConfidence": 0.86,
|
||||
"p10Confidence": 0.74,
|
||||
"lowConfidenceLineRatio": 0.0,
|
||||
},
|
||||
"lines": [
|
||||
{
|
||||
"lineId": "p1-l01-120-340-245-372",
|
||||
"text": "FATo7",
|
||||
"confidence": 0.98,
|
||||
"bbox": [120, 340, 245, 372],
|
||||
},
|
||||
{
|
||||
"lineId": "p1-l02-80-410-300-450",
|
||||
"text": "second line",
|
||||
"confidence": 0.74,
|
||||
"bbox": [80, 410, 300, 450],
|
||||
},
|
||||
],
|
||||
}
|
||||
assert [page["page"] for page in first["pages"]] == [1, 3]
|
||||
|
||||
|
||||
def test_render_paddle_adapter_preserves_text_confidence_and_boxes() -> None:
|
||||
class Prediction:
|
||||
json = {
|
||||
"res": {
|
||||
"rec_texts": ["CBGO4a", "NSAvo6"],
|
||||
"rec_scores": [0.97, 0.83],
|
||||
"rec_boxes": [[10, 20, 110, 50], [15, 70, 130, 100]],
|
||||
}
|
||||
}
|
||||
|
||||
class Pipeline:
|
||||
def predict(self, _image: object) -> list[Prediction]:
|
||||
return [Prediction()]
|
||||
|
||||
lines = PaddleOcrEngine(pipeline=Pipeline()).recognize(object())
|
||||
|
||||
assert lines == [
|
||||
EngineLine("CBGO4a", 0.97, (10, 20, 110, 50)),
|
||||
EngineLine("NSAvo6", 0.83, (15, 70, 130, 100)),
|
||||
]
|
||||
|
||||
|
||||
def test_render_container_pins_cpu_runtime_models_and_single_worker() -> None:
|
||||
service_root = Path(__file__).parents[1]
|
||||
requirements = (service_root / "requirements.txt").read_text()
|
||||
dockerfile = (service_root / "Dockerfile").read_text()
|
||||
dockerignore = (service_root / "Dockerfile.dockerignore").read_text().splitlines()
|
||||
model_loader = (service_root / "app" / "models.py").read_text()
|
||||
|
||||
assert "paddleocr==3.4.0" in requirements
|
||||
assert "paddlepaddle==3.2.2" in requirements
|
||||
assert "pypdfium2==4.30.0" in requirements
|
||||
assert "RUN python -m app.models" in dockerfile
|
||||
assert 'OCR_JOBS_DB="/data/jobs/jobs.db"' in dockerfile
|
||||
assert '"--workers", "1"' in dockerfile
|
||||
assert "USER ocr" in dockerfile
|
||||
assert 'resource.cpu.max="3"' in dockerfile
|
||||
assert 'resource.memory.max="5GiB"' in dockerfile
|
||||
assert "PaddleOcrEngine" in model_loader
|
||||
assert dockerignore == [
|
||||
"**",
|
||||
"!ocr-service/",
|
||||
"!ocr-service/app/",
|
||||
"!ocr-service/app/**",
|
||||
"!ocr-service/requirements.txt",
|
||||
]
|
||||
0
openspec/changes/archive/.gitkeep
Normal file
0
openspec/changes/archive/.gitkeep
Normal file
428
openspec/changes/ocr-ingest-integration/apply-progress.md
Normal file
428
openspec/changes/ocr-ingest-integration/apply-progress.md
Normal file
|
|
@ -0,0 +1,428 @@
|
|||
# Apply Progress: OCR Ingest Integration
|
||||
|
||||
## Current State
|
||||
|
||||
- **Mode:** Strict TDD
|
||||
- **Delivery:** Feature-branch-chain; Units 1–13 implemented as bounded slices; production acceptance remains pending
|
||||
- **Completed tasks:** 1.1–1.4, 2.1–2.5, 3.1–3.4, 4.1–4.5, 5.1–5.4, 6.1–6.4, 7.1–7.3
|
||||
- **Overall task progress:** 29/30 complete
|
||||
|
||||
## Unit 1: Migration
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added `migrations/002_ocr_review.sql` with durable OCR jobs, auditable page records, atomic review correction targets, and a partial unique index for non-terminal OCR identities.
|
||||
- Added focused schema contract coverage in `tests/catalog/migration-002.test.ts`.
|
||||
- Preserved `migrations/001_knowledge_lifecycle.sql` and all repository logic unchanged.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 1.2 | `tests/catalog/migration-002.test.ts` | Schema contract | N/A (new files) | Valid RED: 4/4 failed with `ENOENT` because migration 002 did not exist. An earlier syntax-invalid run was corrected and is not counted as RED. | 4/4 passed with migration 002 present. | 4 independent schema behaviors cover jobs, pages, corrections, and pending identity uniqueness. | None needed; SQL already matches the minimal authoritative contract. |
|
||||
|
||||
### Test Summary
|
||||
|
||||
- **Total tests written:** 4
|
||||
- **Total tests passing:** 4
|
||||
- **Layers used:** Schema contract (4)
|
||||
- **Approval tests:** None — no existing production file was modified
|
||||
- **Pure functions created:** 0
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `npx --no-install tsx --test tests/catalog/migration-002.test.ts` — exit 0; 4 tests, 4 passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | N/A — the task forecast defines Unit 1 as schema-only, and the project has no PostgreSQL test harness or database-test dependency. Applying a live migration would require external database access outside this unit. |
|
||||
| Rollback boundary | Delete `migrations/002_ocr_review.sql` and `tests/catalog/migration-002.test.ts`, revert task 1.2 to pending, and remove this Unit 1 progress/history entry. No existing migration or repository behavior must be reverted. |
|
||||
|
||||
### Additional Validation
|
||||
|
||||
- `npm run check` — exit 0.
|
||||
- `git diff --check` — exit 0 after removing one trailing-space warning introduced in the history header.
|
||||
|
||||
### Deviations
|
||||
|
||||
None — the migration follows the proposal, specifications, design, and closed OCR persistence contract.
|
||||
|
||||
### Remaining Persistence Tasks
|
||||
|
||||
- [x] 1.1 Complete repository-level RED coverage for duplicate pending ingestion, lease recovery, and review transitions in Unit 2.
|
||||
- [x] 1.2 Add migration 002 schema and pending OCR identity unique index.
|
||||
- [x] 1.3 Add catalog repository candidates and leases in Unit 2.
|
||||
- [x] 1.4 Refactor persistence code after Unit 2 reaches green.
|
||||
|
||||
### Review Boundary
|
||||
|
||||
- **Start:** Migration 001 already supplies lifecycle states and parent catalog tables.
|
||||
- **End:** Migration 002 and its focused schema contract test are green.
|
||||
- **Out of scope:** Repository methods, dispatch behavior, runtime PostgreSQL migration, Unit 2, commits, pushes, PRs, deployment, and secrets.
|
||||
- **Rollback:** Remove only the two Unit 1 implementation files and their progress metadata.
|
||||
- **Authored review budget:** 225 additions/deletions, below the 400-line budget for this autonomous slice.
|
||||
|
||||
## Unit 2: Repository
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added pending OCR identity lookup, atomic queue claiming, expired-lease recovery, and guarded review transitions to `CatalogRepository`.
|
||||
- Added a focused fake-pool repository harness that covers duplicate reuse, empty queues, remote identity preservation, and invalid transitions.
|
||||
- Corrected the canonical test script so it executes both root-level and nested test suites.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 1.1 | `tests/catalog/repository-ocr.test.ts` | Unit/fake pool | 4/4 existing tests passed | 0/6 with methods absent; then 5/6 while each completion gate was missing | 6/6 passed after task 1.3 | Found/not-found candidates, claimed/empty queue, expired lease, valid/invalid transition | Covered by 1.4 |
|
||||
| 1.3 | `tests/catalog/repository-ocr.test.ts` | Unit/fake pool | 4/4 existing tests passed | Tests preceded production code | 6/6 passed | Multiple inputs and state paths exercised | Shared OCR row mapping extracted only after green |
|
||||
| 1.4 | `tests/catalog/repository-ocr.test.ts` | Approval refactor | 6/6 before refactor | N/A — behavior-preserving approval baseline | 6/6 after refactor | Existing six cases preserved | Reusable aliased-column projection removed SQL duplication |
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/catalog/repository-ocr.test.ts` — exit 0; 6 tests passed, 0 failed. `NODE_ENV=test npx --no-install tsx --test tests/catalog/*.test.ts` — exit 0; 10 tests passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | `NODE_ENV=test npx --no-install tsx --test tests/*.test.ts` — exit 0; 25 root tests passed, 0 failed. Canonical `npm test` — exit 0; 35 total tests passed, 0 failed, proving the 25 root and 10 nested tests execute together. |
|
||||
| Rollback boundary | Revert Unit 2 changes in `src/modules/catalog/repository.ts`, delete `tests/catalog/repository-ocr.test.ts`, and revert the `package.json` test script; Unit 1 migration remains intact. |
|
||||
|
||||
### Validation and Boundary
|
||||
|
||||
- Corrective rerun for failed evidence revision `sha256:9433c0cd6628f36f4c96a27d41c6d429f4303d038ffe928f297367c274f21c2f`; fresh evidence revision `sha256:a19119f84381447f54a6223e2a203acaf9d3ae08b6ae8f585e3bd52cd510909b`.
|
||||
- The failed canonical script expanded only nested tests after nested suites appeared. The minimal correction is `NODE_ENV=test tsx --test tests/*.test.ts tests/**/*.test.ts`.
|
||||
- Focused repository suite passed 6/6; all catalog suites passed 10/10; explicit root suites passed 25/25; canonical `npm test` passed 35/35; `npm run check` and `git diff --check` exited 0.
|
||||
- Native Unit 2 accounting was 511 changed lines. The maintainer approved `size:exception` exclusively for Unit 2; the corrective attempt has a separate 600-line remediation budget. All later units retain the normal 400-line budget.
|
||||
- No design deviations. No Unit 3 work was started.
|
||||
|
||||
### Corrective TDD Evidence
|
||||
|
||||
| Correction | Safety Net / RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|
|
||||
| Canonical test discovery | Failed evidence revision proved canonical `npm test` executed only 10 nested tests while the explicit root command executed 25 tests. | Updated only the package test argument list; canonical `npm test` passed 35/35. | Direct root 25/25 and nested catalog 10/10 results sum to and identify the canonical 35/35 run. | None needed — one package-script argument was added. |
|
||||
|
||||
## Unit 3: Extraction
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added a Node 22 fixture spike proving `pdf-parse` page callbacks backed by bundled PDF.js `2.0.550` preserve ordered pages, including a blank page.
|
||||
- Added `parsePdfPages` with one-based page identity, raw native text, stable SHA-256 hashes, and fail-closed complete-page accounting.
|
||||
- Preserved textual ingestion as data-only, excluded `CMakeLists.txt`, MDX, and shell lookalikes, and kept task 2.2 pending for Unit 4 detection/composition assertions.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 2.1 | `tests/parsers/pdf-pages.test.ts`; `scripts/spike-pdfjs.ts` | Runtime spike | 20/20 lifecycle service tests passed | Focused suite failed 0/1 because `parsePdfPages` did not exist; the spike and parser implementation were still absent. | Node 22.14.0 spike returned pages 1–3 exactly, including blank page 2. | Three pages prove non-empty, blank, and later-page ordering paths. | Replaced the over-budget direct dependency with the design-approved existing `pdf-parse` pagerender path; spike remained green. |
|
||||
| 2.3 | `tests/parsers/pdf-pages.test.ts` | Unit/integration fixture | 20/20 lifecycle service tests passed | Focused suite failed 0/1 on the missing export before production code. | Final focused run passed 5/5 after implementation. | Covered page hashes/blank page, native PDF composition, non-PDF rejection, data-only text parsing, and unsupported lookalikes. | Focused suite remained 5/5 after consolidating on the existing parser dependency. |
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/parsers/pdf-pages.test.ts` — exit 0; 5 passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | `NODE_ENV=test npx --no-install tsx scripts/spike-pdfjs.ts` — exit 0 on Node 22.14.0; PDF.js 2.0.550 returned three ordered pages with page 2 blank. |
|
||||
| Rollback boundary | Revert `src/modules/parsers/parser-registry.ts`; delete `scripts/spike-pdfjs.ts`, `tests/parsers/pdf-pages.test.ts`, and `tests/fixtures/ocr/native-three-pages.pdf`; revert tasks 2.1/2.3 and this Unit 3 progress/history entry. Units 1–2 remain intact. |
|
||||
|
||||
### Validation and Boundary
|
||||
|
||||
- Canonical `npm test` passed 40/40; `npm run check` and `git diff --check` exited 0.
|
||||
- Native authored slice before progress/history metadata: 225 additions/deletions; no package or lockfile delta remains.
|
||||
- Start: Unit 2 persistence is complete. End: per-page native extraction and parser threat routing are green.
|
||||
- Out of scope: detection thresholds, OCR selection, composition, risks, Unit 4 tasks 2.4/2.5, commits, pushes, PRs, deployment, and restricted paths.
|
||||
- No specification deviation. The design-authorized `pdf-parse` pagerender fallback was selected only after its Node 22 fixture spike passed.
|
||||
|
||||
## Unit 4: Detection and Composition
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added pure per-page native detection under `pdf-detection-v1`, exact OCR page selection, verified-blank classification, and fail-closed OCR quality gates.
|
||||
- Added deterministic candidate composition using exactly one source per page, ordered OCR lines, blank omission, canonical hashes, and unchanged risk tokens prioritized for review.
|
||||
- Completed the remaining task 2.2 scenarios by combining the Unit 3 data-only/parser threat coverage with Unit 4 mixed-PDF, detection, blank, hash, and risk RED assertions.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 2.2 | `tests/ocr/detection.test.ts`; `tests/parsers/pdf-pages.test.ts` | Unit/integration fixture | Unit 3 parser suite passed 5/5 before edits. | Focused Unit 4 suite failed before production with `ERR_MODULE_NOT_FOUND` for `src/modules/ocr/detection.js`. | Focused suite passed 5/5 after detection and composition were added. | Real generated mixed PDF selected pages 2 and 3; `.txt`/`.md` stayed data-only and CMake/MDX/shell lookalikes stayed unsupported. | None needed; routing remains separated between parser support and the pure PDF OCR planner. |
|
||||
| 2.4 | `tests/ocr/detection.test.ts` | Unit | N/A (new production file) | Threshold, mixed-page, blank, and quality assertions preceded production code. | Exact native, blank, and OCR quality boundaries passed in the 5/5 focused suite. | Every native and OCR threshold has a passing boundary plus a failing neighbor; visible ink with empty OCR is blocked. | Pure functions and named policy version were present at green; final focused suite remained 5/5. |
|
||||
| 2.5 | `tests/ocr/detection.test.ts` | Unit | N/A (new production file) | Ordered composition, blank omission, canonical hash, audit-source, and risk assertions preceded production code. | Deterministic composition and risk prioritization passed in the 5/5 focused suite. | Reordered pages, tied OCR boxes, blank middle page, repeated token, and four known ambiguous tokens exercise distinct paths. | Reused shared canonical JSON and SHA-256 utilities; final focused suite and type check remained green. |
|
||||
|
||||
### Test Summary
|
||||
|
||||
- **Tests:** 5 written and focused-passing; 45 canonical (unit 4, integration fixture 1)
|
||||
- **Approval tests:** None — new production files; **pure functions created:** 6
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/ocr/detection.test.ts` — exit 0; 5 tests passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | `NODE_ENV=test npx --no-install tsx --test --test-name-pattern "parsed mixed PDF" tests/ocr/detection.test.ts` — exit 0; 1 test passed, 0 failed. The harness generated a three-page PDF, parsed it through the Node 22 `pdf-parse` page callback, and selected only blank page 2 and insufficient page 3 for OCR. |
|
||||
| Rollback boundary | Delete `src/modules/ocr/detection.ts`, `src/modules/ocr/composition.ts`, and `tests/ocr/detection.test.ts`; revert tasks 2.2/2.4/2.5 and this Unit 4 progress/history entry. Units 1–3 and their parser threat coverage remain intact. |
|
||||
|
||||
### Validation and Boundary
|
||||
|
||||
- Canonical `npm test` passed 45/45; `npm run check`, tracked `git diff --check`, and no-index whitespace checks for all three new files exited 0.
|
||||
- Authored Unit 4 code and tests total 322 additions and 0 deletions, below the 400-line budget with no size exception.
|
||||
- Start: Unit 3 page extraction and parser threat routing are complete. End: detection, blank/quality gates, deterministic composition, hashes, and risk tokens are green.
|
||||
- Out of scope: OCR service, client, dispatcher, ingest routing, review APIs, commits, pushes, PRs, deployment, and restricted paths.
|
||||
- No specification or design deviation.
|
||||
|
||||
## Unit 5: OCR Service
|
||||
|
||||
### Implementation and Correction Summary
|
||||
|
||||
- Added pinned Unit 5 FastAPI test dependencies and retained the ignored `ocr-service/.venv` environment.
|
||||
- Added RED-first HTTP tests and a SQLite-backed private API for bearer auth, idempotency, allowlisting, upload/page limits, queue pressure, status, deletion, and health.
|
||||
- Preserved failed evidence `sha256:4a5c8ed8d8b302f7ff654d4f40fe0f13049bfdd29db4c76576f3847c0a9be15a`: its harness returned `401/400/400`; inspection showed `&` backgrounded the preceding shell AND-list, so request variables existed only in the child shell and authenticated curls sent an empty `request` form, reproduced as `400 INVALID_REQUEST`.
|
||||
- Corrected only the harness command boundary; the 381-line candidate implementation remained unchanged before progress persistence.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 3.1 | `ocr-service/tests/test_api.py` | HTTP integration | N/A (new service) | Collection failed with `ModuleNotFoundError: No module named 'app'`; failed localhost runtime evidence then supplied correction RED. | 8/8 focused tests and corrected curl `401/202/409` passed. | Auth variants, idempotency/conflict, allowlist, limits, pressure, integrity, and real multipart requests exercised distinct paths. | No production refactor: the defect was shell variable scope in the harness. |
|
||||
| 3.2 | `ocr-service/tests/test_api.py` | HTTP integration/SQLite | N/A (new service) | Same missing-service RED covered the API; failed harness blocked completion. | 8/8 focused tests and real Uvicorn harness passed. | Ready/not-ready health, queue capacity, status, deletion, and process cleanup exercised distinct paths. | Pure page hashing and centralized terminal errors retained; no correction needed. |
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
- Focused: `ocr-service/.venv/bin/python -m pytest ocr-service/tests -k auth` — exit 0; 8 passed, 0 failed (one dependency deprecation warning).
|
||||
- Runtime: local Uvicorn plus multipart curl — unauthorized `401` with `UNAUTHORIZED`, accepted `202` with queued identity, conflicting same key `409` with `IDEMPOTENCY_CONFLICT`.
|
||||
- Gates: `npm test` 45/45, `npm run check`, Python `compileall`, tracked and untracked whitespace checks — all exit 0/equivalent clean.
|
||||
- Cleanup: Uvicorn PID 268425 terminated and absent; temporary PDF, SQLite DB, and log removed; `.venv` retained.
|
||||
- Rollback: remove `.gitignore` Unit 5 entries and `ocr-service/`, then revert tasks 3.1/3.2 and this Unit 5 progress/history block; Units 1–4 remain intact.
|
||||
- Fresh evidence revision: `sha256:0d8b2c57bbab967473eaf99a7ca900168554e64254fa80197b3626b4570b58d0`; corrected final authored change count: 397 additions plus deletions, within 400.
|
||||
|
||||
## Unit 6: OCR Rendering and Image
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added deterministic 200 DPI PDF page rendering with a pre-render 25-megapixel guard, a PaddleOCR adapter, stable line identities, quality metrics, and the exact result schema.
|
||||
- Added a deterministic fake-driven test seam and preserved ambiguous OCR tokens without correction.
|
||||
- Added the private CPU image with PaddleOCR `3.4.0`, PaddlePaddle CPU `3.2.2`, baked Latin/mobile models, one Uvicorn worker, a non-root user, private job storage, healthcheck, and documented 3 CPU/5 GiB deployment limits.
|
||||
- Added a Dockerfile-specific deny-by-default build context so restricted and unrelated repository paths are never sent to the Docker daemon.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 3.3 | `ocr-service/tests/test_render.py` | Unit/integration fixture | Existing OCR API suite passed 8/8 before production edits. | Focused collection failed because `app.engine` did not exist. | Focused render suite passed 5/5; complete OCR suite passed 13/13. | Selected pages 1/3, invalid page and pixel-limit paths, repeatable fake results, and Paddle payload normalization exercise distinct behavior. | Named constants and pure metric/result transforms retained; focused suite remained 5/5. |
|
||||
| 3.4 | `ocr-service/tests/test_render.py` | Static/container runtime | Existing OCR API suite passed 8/8 before production edits. | Docker contract was absent; deny-by-default context triangulation then failed until `Dockerfile.dockerignore` existed. | Image built with baked models; offline constrained container rendered and validated PNG/result output. | Static pin/runtime assertions plus real model build and `--network none` execution cover configuration and runtime paths. | Added missing OpenCV runtime libraries after the first build exposed `libGL.so.1`; exact corrected image passed. |
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `ocr-service/.venv/bin/python -m pytest ocr-service/tests -k render -q` — exit 0; 5 passed, 8 deselected, 1 dependency deprecation warning. |
|
||||
| Runtime harness command/scenario and exact result | `docker build --file ocr-service/Dockerfile --tag rag-ocr-service:unit6 .` — corrected exact image exited 0 and baked both models. A `docker run --rm --network none --cpus 3 --memory 5g ...` harness loaded cached PaddleOCR 3.4.0/PaddlePaddle 3.2.2 models, rendered a 1700x2200 PNG (23,309 bytes), returned schema `1`, one page, and one OCR line; host Pillow reopened it as PNG 1700x2200. |
|
||||
| Rollback boundary | Remove `ocr-service/{Dockerfile,Dockerfile.dockerignore,app/engine.py,app/models.py,app/render.py,tests/test_render.py}` and revert Unit 6 changes in `app/main.py`, `requirements.txt`, and `README.md`; Units 1–5 remain intact. |
|
||||
|
||||
### Validation and Boundary
|
||||
|
||||
- RED was followed by GREEN and post-refactor reruns. One expected GREEN iteration corrected a test-side non-whitespace count from 16 to the mathematically correct 15.
|
||||
- The first image build failed on missing `libGL.so.1`; adding the required slim-image runtime libraries produced a successful build and runtime harness. A later optional connectivity-check refactor rebuild exhausted Docker storage; that unvalidated one-line refactor was reverted, the validated Dockerfile bytes were restored, and the image tag was removed during cleanup.
|
||||
- `npm test` passed 45/45; `npm run check`, Python `compileall`, tracked/untracked whitespace checks, and the full OCR pytest suite (13/13) passed.
|
||||
- Start: Unit 5 private queue/API is complete. End: Unit 6 render/engine/schema/container behavior is complete. Unit 7 client work was not started.
|
||||
- Native Unit 6 implementation/tests/docs are 397 authored additions plus deletions, within the 400-line budget; required SDD progress and workspace history metadata are administrative evidence outside that native slice.
|
||||
- No specification or design deviation. The Docker image is private by deployment contract; CPU/RAM labels document limits that the platform must enforce.
|
||||
|
||||
## Unit 7: OCR Client and Durable Artifacts
|
||||
|
||||
### Implementation Summary
|
||||
|
||||
- Added a private OCR HTTP client that preserves the exact idempotency key across one initial submission and at most two transient resubmissions, treats `429` as retryable queue pressure without immediate resubmission, and treats deterministic failures as terminal.
|
||||
- Added 2–15 second polling backoff plus strict acknowledgement, status, engine, job, document, ordered-page, metrics, line, and schema validation before a result can be persisted.
|
||||
- Added durable `0600` original and canonical manifest writes, UUIDv5 document artifact identities, lexical path containment, verifiable hashes, and age-gated orphan sweeping that preserves retained, recent, non-UUID, and symlink entries.
|
||||
- Left task 4.1 pending because Unit 7 proves only its retries/integrity clause; Unit 8 still owns flag-off `201`, OCR `202`/status, OCR-down fail-closed, and catalog-down `503` HTTP coverage.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 4.2 | `tests/ocr/client.test.ts` | Unit/runtime fetch stub | Existing OCR detection suite passed 5/5; production file was new. | Focused suite failed with `ERR_MODULE_NOT_FOUND` for `src/modules/ocr/client.js` before production code existed. | Final focused suite passed 6/6; runtime retry scenario passed 1/1. | Network then `503` produced exactly three same-key attempts with 2s/4s backoff; `429` and `422` each stopped after one attempt; malformed acknowledgement/schema/pages failed integrity; polling used 2s/4s. | Adapted multipart bytes to a type-safe `Uint8Array`; focused suite and type check remained green. |
|
||||
| 4.3 | `tests/ocr/client.test.ts` | Filesystem integration | Existing OCR detection suite passed 5/5; production file was new. | The same missing-module RED preceded artifact production code. | Private originals, canonical manifest, safe resolution, and sweep passed in the final 6/6 suite. | Two out-of-order documents proved canonical ordering/hashes; traversal was rejected; sweep removed only an old orphan while preserving retained/recent directories and a symlink. | A test-side assertion was corrected to inspect the retained dangling symlink with `lstat`; no production behavior changed, and the suite passed 6/6. |
|
||||
|
||||
### Test Summary
|
||||
|
||||
- **Tests written and passing:** 6 focused tests; unit/fetch-stub and filesystem integration layers.
|
||||
- **Approval tests:** None — both production modules are new.
|
||||
- **RED:** exit 1, module-not-found before either production file existed.
|
||||
- **GREEN/REFACTOR:** exit 0, 6 passed, 0 failed.
|
||||
|
||||
### Work Unit Evidence
|
||||
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/ocr/client.test.ts` — exit 0; 6 passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | `NODE_ENV=test npx --no-install tsx --test --test-name-pattern "runtime retry stub" tests/ocr/client.test.ts` — exit 0; 1 passed, 0 failed. Deterministic stub observed network failure, `503`, then `202`: exactly 3 submission attempts, one unchanged idempotency key, 2,000/4,000 ms backoffs, followed by a strictly validated two-page result. |
|
||||
| Rollback boundary | Remove `src/modules/ocr/client.ts`, `src/modules/ocr/artifacts.ts`, and `tests/ocr/client.test.ts`; revert tasks 4.2/4.3 and this Unit 7 progress/history metadata. Units 1–6 remain intact. |
|
||||
|
||||
### Validation and Boundary
|
||||
|
||||
- `npm test` passed 51/51; `npm run check`, `npm run build`, tracked whitespace, and no-index whitespace checks for all three new files passed.
|
||||
- Test-created `rag-ocr-artifacts-*` and `rag-ocr-sweep-*` directories were removed; no server, external OCR call, or persistent process was started; `ocr-service/.venv` was preserved.
|
||||
- Start: Unit 6 OCR rendering and container runtime are complete. End: Unit 7 client retries/integrity and durable artifacts/orphan sweep are implemented and green.
|
||||
- Out of scope and untouched: dispatcher, reconciler, ingest branch, app upload/status routes, Unit 8 tests, commits, pushes, PRs, native review, deployment, and restricted paths.
|
||||
- The implementation/test slice is 506 authored additions (client 174, artifacts 129, tests 203). This exceeds the 400-line default after one honest cohesive assessment; no code-golf or second slicing pass was attempted. The maintainer explicitly approved `size:exception`, including 585 total native-accounting lines with required metadata.
|
||||
- No specification or design deviation.
|
||||
|
||||
### Approved Size-Exception Corrective Context
|
||||
|
||||
- The native attempt was reset after failed evidence revision `sha256:dd224c3864adc8380dc7d9fd147cb448f5da8138277f3f346a35d7eeb3aca158`; the failure was budget-only, not a functional defect.
|
||||
- Corrective work unit `unit-7-size-exception-validation` uses attempt token `sha256:0109cfd8d3b37a5672c65a37700269b89cb80e4e357a39f68a920779a1715c39` and must settle against the failed revision through the parent.
|
||||
- Production and test files remain unchanged for the corrective rerun; only approval/reset evidence and fresh validation metadata may change.
|
||||
- Task 4.1 remains pending, while tasks 4.2 and 4.3 remain complete. Unit 8 was not started.
|
||||
|
||||
## Unit 8: Dispatcher and Runtime Orchestration
|
||||
- Completed multi-document progress, uniqueness-race reuse, exact lease recovery, persisted-key dispatch, default runtime wiring, authenticated status, and fail-closed retrieval; no activation or embeddings occur on OCR failure.
|
||||
|
||||
### Evidence Chain and Review Accounting
|
||||
- **Complete review slice:** Relative to committed HEAD `ac046e3`, Unit 8 contains 1,045 functional changed lines plus 41 required metadata lines, for 1,086 total changed lines. The complete slice is not within 400 lines, and no `size:exception` was approved or recorded for it.
|
||||
- **Interrupted preserved candidate:** Native failed/interrupted revision `sha256:182cc73954a8518d03bd6c3ef5f4141b352aa9b69fca4177447d76deaddad223` contained 815 changed lines.
|
||||
- **Auto-chain reset:** The maintainer-selected preflight strategy was `auto-chain` with `feature-branch-chain`. Native reset revision `sha256:c0db82cf3db8ae790d34e4c310b998f97bb36864c97c3aa84e4c87ab77d07c2a` preserved the interrupted candidate and scoped the next bounded recovery objective only to additional changes; it did not approve the complete Unit 8 slice as a size exception.
|
||||
- **Recovery delta:** After that reset, the bounded recovery added 350 functional changed lines plus 41 required metadata lines, for 391 total changed lines within the separate 400-line recovery objective.
|
||||
- **Passed recovery:** Fresh passed evidence revision `sha256:e424afe7b2f8efd547dcfd0baf64ec1b46ac344f39afd87259560a094d59163f` closes the interrupted → auto-chain reset → passed recovery chain.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 4.1 | `tests/ocr/{dispatcher,client}.test.ts` | HTTP/integration | Recovered baseline 7/7 passed; not labeled RED | New scenarios produced genuine 5/9 RED; catalog `/retrieve` already failed closed | Unit 8 9/9; client 6/6 | 201/202/status, OCR failure, 503 retrieval, retries/integrity | Consolidated persisted-key assertion; 9/9 |
|
||||
| 4.4 | `tests/ocr/dispatcher.test.ts` | Unit/integration | Recovered baseline 7/7 | Exact recovery failed by claiming unrelated work | 9/9 | Live/new queue separation plus periodic dispatch | Exact-job dispatch helper; 9/9 |
|
||||
| 4.5 | `tests/ocr/dispatcher.test.ts` | HTTP/filesystem | Recovered baseline 7/7 | Runtime wiring RED was 8/10 | 9/9 | Multipart upload, durable reload, duplicate race, mixed documents | Shared queue drain; 9/9 |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test | `NODE_ENV=test npx --no-install tsx --test tests/ocr/dispatcher.test.ts` — 9 passed, 0 failed. |
|
||||
| Focused repository OCR test | `NODE_ENV=test npx --no-install tsx --test tests/catalog/repository-ocr.test.ts` — 6 passed, 0 failed. |
|
||||
| Canonical Node test | `npm test` — 60 passed, 0 failed. |
|
||||
| Runtime harness | Bounded localhost curl — ingest `202/ocr_queued`; authenticated status `200/ocr_queued`; leaked child from initial shell cleanup was detected, terminated, and port/temp cleanup reverified. |
|
||||
| Rollback boundary | Revert Unit 8 deltas in `src/{app,config/env}.ts`, catalog/ingest/client modules, `src/modules/ocr/dispatcher.ts`, and its test; preserve Units 1–7. |
|
||||
|
||||
## Unit 9: Review and Indexing Core
|
||||
- Added authenticated review/approval routes with injectable service boundaries, immutable correction preparation, stale/duplicate conflict guards, and new/reusable activation CAS behavior. Rejection transition and playground UI remain Unit 10.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 5.1 | `tests/ocr/review.test.ts` | HTTP/unit | App suites 9/9 and 20/20 | Missing `indexing.js`; exit 1 | 5/5 | Unauthorized/premature, candidate/line/active staleness, duplicate targets, new/reusable races, rejected no-index | Assertions remained 5/5 |
|
||||
| 5.2 | `tests/ocr/review.test.ts` | HTTP/unit | 29/29 | Tests preceded `review.ts` | 5/5 | Audit view plus two corrections and four conflict paths | Clone prevents failed commits mutating loaded candidates; 5/5 |
|
||||
| 5.3 | `tests/ocr/review.test.ts` | Unit | N/A (new file) | Tests preceded `indexing.ts` | 5/5 | New/reusable true/false, both CAS races, rejected no-index | Named CAS error mapping; 5/5 |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test | `NODE_ENV=test npx --no-install tsx --test tests/ocr/review.test.ts` — exit 0; 5 passed, 0 failed. |
|
||||
| Runtime harness | Same runner with `--test-name-pattern "review HTTP rejects"` — exit 0; 1 passed; real localhost Express requests returned authenticated boundary outcomes without external services. |
|
||||
| Rollback boundary | Remove `src/modules/ocr/{review,indexing}.ts` and `tests/ocr/review.test.ts`; revert only the Unit 9 route additions in `src/app.ts` and Unit 9 metadata. Preserve Units 1–8. |
|
||||
|
||||
### Validation and Boundary
|
||||
- Canonical `NODE_ENV=test npm test` passed 65/65; `npm run check`, `npm run build`, tracked whitespace, and no-index whitespace checks passed.
|
||||
- Unit 9 adds 341 functional changed lines and 38 metadata changed lines, 379 total, within the 400-line budget. No live PostgreSQL, OpenAI, Qdrant, or OCR process was used; no test process remained.
|
||||
- Start: Unit 8 dispatch/status is complete. End: review/correction and indexing/CAS ports plus authenticated routes are green. Unit 10 task 5.4 remains unchecked.
|
||||
|
||||
## Unit 10: Rejection and Review UI
|
||||
- Added authenticated rejection with current candidate-hash and state guards, required audit fields, durable store completion before success, and no indexing or activation call.
|
||||
- Extended the existing static playground with authenticated candidate loading, protected page images, native/raw/candidate comparisons, line confidence/boxes, corrections, risks, approval, and rejection controls.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 5.4 | `tests/ocr/review.test.ts` | HTTP integration/static HTTP | Existing review suite passed 5/5 before edits. | Genuine RED passed only 5/7: rejection returned 404 instead of authenticated 401 and the playground lacked the review contract. A malformed-body edge then returned 409 instead of 400. | Focused suite passed 7/7 after minimal rejection and UI implementation. | Unauthorized, malformed, stale-hash, successful, repeated-state, durable-status, and static audit-field paths exercise distinct outcomes. | Shared UI decision-state helper extracted; focused suite remained 7/7. |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/ocr/review.test.ts` — exit 0; 7 passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | `NODE_ENV=test npx --no-install tsx --test --test-name-pattern "authenticated rejection" tests/ocr/review.test.ts` — exit 0; 1 passed, 0 failed. Real localhost Express requests returned 401/400/409/200, persisted the rejected audit state through the injected store, retrieved rejected status, and made zero indexing calls. |
|
||||
| Rollback boundary | Revert Unit 10 deltas in `src/app.ts`, `src/modules/ocr/review.ts`, `tests/ocr/review.test.ts`, and `public/playground/{index.html,app.js,styles.css}` plus Unit 10 metadata; preserve Units 1–9. |
|
||||
|
||||
### Validation and Boundary
|
||||
- Canonical `NODE_ENV=test npm test` passed 67/67; `npm run check` and `npm run build` exited 0.
|
||||
- Previously observed tracked `git diff --check` exited 0; no-index whitespace checks for Unit 10's untracked `src/modules/ocr/review.ts` and `tests/ocr/review.test.ts` exited 0.
|
||||
- Unit 10 adds 271 functional and 42 metadata changed lines, 313 total, below the 400-line budget. No live database, embedding, vector, OCR, or external service was used; no persistent process was started.
|
||||
- Start: Unit 9 review/indexing ports and approval routes are green. End: rejection isolation and the static review UI are green. Retention, deploy, OpenAPI, and E2E tasks remain untouched.
|
||||
- No specification or design deviation.
|
||||
|
||||
## Unit 11: Retention and Flag-Off Isolation
|
||||
|
||||
### Implementation Summary
|
||||
- Added a retention boundary that expires 30-day reviews before deletion, retains failed/rejected artifacts for 7 days, and retains superseded OCR artifacts for 30 days.
|
||||
- Added exact state/artifact CAS claims, active-pointer checks, activation exclusion after a deletion claim, and restart-safe `retention_deleting→retention_deleted` completion.
|
||||
- Wired retention into the existing exclusive reconciler cycle and made status/review/decision APIs return `404` without store access while OCR is disabled; native synchronous ingestion remains unchanged.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 6.1 | `tests/ocr/retention.test.ts` | Repository/HTTP/filesystem integration | Existing repository, dispatcher, and review suites passed 22/22. | Genuine missing-module RED failed before `retention.ts`; activation-race RED then failed 0/1 with “Missing expected rejection.” | Focused suite passed 6/6. | Exact TTL SQL, active pointer/state guards, flag-off `201/404`, and deletion-claim activation conflict exercise independent paths. | No production refactor was needed; two test-side expectations were corrected during GREEN without weakening behavior. |
|
||||
| 6.2 | `tests/ocr/retention.test.ts` | Filesystem/reconciler integration | Same 22/22 safety net. | The missing-module RED preceded all production retention code. | Focused suite passed 6/6; relevant regression set passed 28/28. | Review expiry, terminal deletion, active preservation, simulated restart, repeat no-op, and exclusive reconciler execution cover distinct paths. | Named store boundary keeps filesystem behavior isolated; final focused suite remained 6/6. |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/ocr/retention.test.ts` — exit 0; 6 passed, 0 failed. |
|
||||
| Runtime harness command/scenario and exact result | The focused suite used real temporary directories and localhost Express requests: interrupted deletion removed only the claimed directory, resumed to `retention_deleted`, repeated as a no-op, preserved the active directory, returned native `201`, and returned `404` for all disabled OCR candidate APIs without store access. |
|
||||
| Rollback boundary | Remove `src/modules/ocr/retention.ts` and `tests/ocr/retention.test.ts`; revert only Unit 11 seams in `src/app.ts`, catalog repository/reconciler, and explicit enabled-state setup in dispatcher/review tests. Preserve Units 1–10. |
|
||||
|
||||
### Validation and Boundary
|
||||
- Relevant regression suites passed 28/28 after explicitly enabling OCR in pre-existing enabled-path tests; the initial 19/22 run exposed only that test-fixture assumption.
|
||||
- Canonical `NODE_ENV=test npm test` passed 73/73; `npm run check`, `npm run build`, tracked whitespace, and no-index whitespace checks for all seven intended untracked files passed.
|
||||
- Temporary `rag-retention-*` directories were removed and no `tsx --test` or Node test process remained. No live database, OCR, embedding, vector, or external service was used; `ocr-service/.venv` was preserved.
|
||||
- Unit 11 adds 277 functional changed lines and 50 metadata changed lines, 327 total, below the 400-line budget.
|
||||
- Start: Unit 10 rejection/UI is complete. End: tasks 6.1–6.2 are complete. OpenAPI, deployment/configuration, E2E, production, commit, push, and PR work remain untouched.
|
||||
- No specification or design deviation.
|
||||
|
||||
## Unit 12: Contracts and Deployment Wiring
|
||||
|
||||
### Implementation Summary
|
||||
- Documented OCR `202` ingestion variants and authenticated status, review, correction/approval, and rejection routes with behavior-matched success and error responses.
|
||||
- Added progress, review evidence, optimistic correction, decision, and structured error schemas while preserving the legacy-disabled `202` variants.
|
||||
- Added private durable artifact defaults and OCR limits/timeouts; the root image now declares `/data/ingestions`, runs as `node`, and applies catalog migrations before startup without embedding secrets.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 6.3 | `tests/ocr/contracts-deploy.test.ts` | OpenAPI contract | Existing lifecycle/dispatcher/review/retention suites passed 42/42. | Genuine initial RED was 0/3: OCR `202` referenced the native schema and OCR routes/schemas were absent. | Focused suite passed 3/3. | A second RED required legacy-disabled `202` variants and accurate help authentication; final suite remained 3/3. | Shared schema references kept nested contracts reviewable; no behavior changed. |
|
||||
| 6.4 | `tests/ocr/contracts-deploy.test.ts` | Config/static container | Same 42/42 safety net. | Initial RED found the old relative artifact root, absent limits/timeouts, volume, non-root user, and migration-aware command. | Focused suite passed 3/3; sanitized Docker build and image inspection passed. | A second RED required explicit runtime artifact-root wiring; final suite remained 3/3. | No further refactor needed. |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test command and exact result | `NODE_ENV=test npx --no-install tsx --test tests/ocr/contracts-deploy.test.ts` — exit 0; 3 passed, 0 failed. |
|
||||
| Relevant regression command and exact result | Lifecycle, dispatcher, review, retention, and contract suites — exit 0; 45 passed, 0 failed. |
|
||||
| Runtime harness | N/A — Unit 12 is static-only by task forecast. A sanitized-context Docker build succeeded and inspection proved user `node`, durable volume, production artifact root, and migration-before-server command. |
|
||||
| Rollback boundary | Remove `tests/ocr/contracts-deploy.test.ts` and revert only Unit 12 deltas in `src/api/openapi.ts`, `src/config/env.ts`, and `Dockerfile`; preserve Units 1–11. |
|
||||
|
||||
### Validation and Boundary
|
||||
- Canonical `NODE_ENV=test npm test` passed 76/76; `npm run check`, `npm run build`, tracked whitespace, and all intended-untracked whitespace checks passed.
|
||||
- Docker build `rag-service:unit12-contract` succeeded from a sanitized 430.31 kB context; inspection matched the deployment contract. The image and temporary context were removed, no test process remained, and `ocr-service/.venv` was preserved.
|
||||
- Unit 12 adds 219 functional lines and 50 metadata lines, 269 total, below the 400-line budget.
|
||||
- Start: Unit 11 retention is complete. End: tasks 6.3–6.4 are complete. Unit 13 / all 7.x E2E and production work remains untouched.
|
||||
- `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` was read as authoritative and remained unchanged. No specification or design deviation.
|
||||
|
||||
## Unit 13: Local Deterministic E2E and Complete Gate
|
||||
|
||||
### Implementation Summary
|
||||
- Added a bounded localhost HTTP suite covering native `201`, scanned `202 → review → approve → active`, and exact mixed native/OCR page reporting.
|
||||
- Added resend identity reuse, fail-closed OCR exhaustion, and catalog-unavailable `503` coverage without external services.
|
||||
- Made no production-code changes; the new scenarios passed immediately as baseline/characterization evidence, so no RED was fabricated.
|
||||
|
||||
### TDD Cycle Evidence
|
||||
| Task | Test File | Layer | Safety Net | RED / Baseline | GREEN | TRIANGULATE | REFACTOR |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| 7.1 | `tests/ocr/e2e.test.ts` | Local HTTP E2E | Existing OCR suites passed 28/28. | Baseline/characterization: the test was written first and passed 2/2 immediately; no implementation gap existed at the local fake boundary. | Focused suite passed 2/2. | Native, scanned, corrected approval, post-activation status, and mixed page methods exercise distinct paths. | Test-only state naming improved; focused suite remained 2/2. |
|
||||
| 7.2 | `tests/ocr/e2e.test.ts` | Local HTTP/dispatcher E2E | Same 28/28 safety net. | Baseline/characterization: resend, OCR-down, and catalog-down assertions passed immediately; no production code was written. | Focused suite passed 2/2. | Duplicate resend, terminal dispatch failure, status `503`, and retrieval `503` cover independent outcomes. | None needed beyond the shared test harness. |
|
||||
| 7.3 | Gate commands below | Complete gate | Focused E2E passed 2/2. | N/A — verification-only task with no production behavior to implement. | All required gates passed. | Node type/build/test and offline Python tests provide independent stack evidence. | N/A — no production change. |
|
||||
|
||||
### Work Unit Evidence
|
||||
| Evidence | Result |
|
||||
|---|---|
|
||||
| Focused test | `NODE_ENV=test npx --no-install tsx --test tests/ocr/e2e.test.ts` — exit 0; 2 passed, 0 failed. |
|
||||
| Runtime harness | The focused suite started bounded ephemeral localhost Express servers and exercised real HTTP requests through ingest, status, review, approval, and retrieval routes; all servers closed through test cleanup. |
|
||||
| Complete gate | `npm run check`, `npm run build`, and `NODE_ENV=test npm test` exited 0; canonical Node passed 78/78. `PYTHONDONTWRITEBYTECODE=1 ocr-service/.venv/bin/python -m pytest ocr-service/tests -q` passed 13/13 offline with one dependency deprecation warning. |
|
||||
| Rollback boundary | Remove `tests/ocr/e2e.test.ts` and revert only tasks 7.1–7.3 plus this Unit 13 progress/history metadata; preserve Units 1–12. |
|
||||
|
||||
### Validation and Boundary
|
||||
- Unit 13 adds 183 functional lines. Required tasks/progress/history metadata adds 52 changed lines, for 235 total, below the 400-line budget.
|
||||
- Tracked and complete intended-untracked whitespace checks passed. No matching `tsx --test`, E2E Node, or OCR Uvicorn process remained after the corrected self-excluding process check.
|
||||
- Intended untracked inventory: `src/modules/ocr/{dispatcher,indexing,retention,review}.ts`; `tests/ocr/{contracts-deploy,dispatcher,e2e,retention,review}.test.ts`.
|
||||
- Tasks 7.1–7.3 are complete. Task 7.4 remains unchecked and untouched; no production flag, service, database, vector store, embedding provider, OCR endpoint, commit, push, or PR was used.
|
||||
95
openspec/changes/ocr-ingest-integration/design.md
Normal file
95
openspec/changes/ocr-ingest-integration/design.md
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
# Design: OCR Ingest Integration
|
||||
|
||||
## Technical Approach
|
||||
|
||||
Insert page-aware OCR before existing lifecycle indexing. Native PDFs retain synchronous `201`; OCR candidates persist originals, catalog rows, and jobs before returning `202`. PostgreSQL remains authoritative, Qdrant receives only reviewed content, and every failure preserves the active pointer.
|
||||
|
||||
## Architecture Decisions
|
||||
|
||||
| Decision | Choice and rationale |
|
||||
|---|---|
|
||||
| Extraction | Fixture-gate `pdfjs-dist`; use `pdf-parse` `pagerender` only if proven on Node 22. Aggregated-text slicing is forbidden. |
|
||||
| Execution | One in-process dispatcher with PostgreSQL leases; recovery needs no second Node deployment. |
|
||||
| OCR | Private FastAPI, one worker/replica, baked PaddleOCR 3.4.0/PaddlePaddle CPU 3.2.2 models. |
|
||||
| Ownership | RAG owns durable `0600` artifacts; OCR owns a 24-hour work copy; reviewed bytes are immutable. |
|
||||
|
||||
## Data Flow and Concurrency
|
||||
|
||||
`upload → disk stage → page detection → native 201 OR durable candidate/job → 202 → OCR → review_required → approve → index/verify → activation CAS`
|
||||
|
||||
`findPendingOcrVersion(identity)` is distinct from `findReusableVersion` and joins versions, documents, and jobs using:
|
||||
|
||||
```ts
|
||||
type PendingOcrIdentity={sourceId:string;originalManifestHash:Sha256;processingFingerprint:Sha256;metadataHash:Sha256}
|
||||
```
|
||||
|
||||
It accepts `source_content_hash IS NULL` and states `pending|indexing|review_required`. Migration 002 adds a matching partial unique index. Under `withSourceLock`, ingestion validates source type/reference, performs this lookup, then either returns the persisted version/jobs or preallocates `versionId`, durably stages/fsyncs originals, and transactionally inserts one candidate plus `UNIQUE(version_id,document_id)` jobs. A uniqueness loss reloads the winner; the losing staging directory is orphan-swept. All duplicates return the same `202/statusUrl` and persisted activation intent.
|
||||
|
||||
Approval starts one transaction locking source and candidate. Before artifact writes, corrections, or transition, it requires `request.expectedActiveVersionId === candidate.base_active_version_id === source.active_version_id`; otherwise `409 ACTIVE_VERSION_CHANGED` and zero mutation. Candidate/line hashes and unique targets are then validated atomically.
|
||||
|
||||
After final hashes, source-locked `findReusableVersion(finalContent,fingerprint,metadata)` runs. A match rejects the candidate: `DUPLICATE_REUSABLE_VERSION`. Persisted `candidate.activate_requested === false` preserves the active pointer; only true activates the `ready|superseded|active` version via expected-active CAS. CAS failure returns `409 ACTIVE_VERSION_CHANGED` and rolls back the reusable branch. Otherwise, approval persists corrections/hashes and indexes to `ready`. Only persisted `candidate.activate_requested === true` invokes Point-2 activation with the original expected-active CAS; false succeeds in `ready` without pointer change. CAS failure returns `409 ACTIVE_VERSION_CHANGED`, leaving the candidate `ready`.
|
||||
|
||||
Jobs use `queued→running→succeeded|failed`; expired leases return to `queued` with unchanged remote key/ID. Versions use `pending→indexing→review_required→indexing→ready→active`, plus fail/reject/purge paths. Retention uses version locks and `present→retention_deleting→retention_deleted`; active artifacts are never TTL-deleted.
|
||||
|
||||
## Interfaces / Contracts
|
||||
|
||||
`OcrAccepted` is the OCR `202`; all `/ingestions/*` routes require lifecycle bearer authentication.
|
||||
|
||||
```ts
|
||||
type OcrAccepted={accepted:true;sourceId:string;versionId:UUID;versionNumber:number;state:"indexing";phase:"ocr_queued";statusUrl:string;reviewUrl:null;activated:false}
|
||||
type IngestionStatus={sourceId:string;versionId:UUID;state:VersionState;phase:"native_extracting"|"ocr_queued"|"ocr_running"|"review_required"|"indexing"|"ready"|"active"|"failed"|"rejected";activated:boolean;documents:{documentId:string;state:"native_complete"|"ocr_queued"|"ocr_running"|"ocr_complete"|"failed";completedPages:number;totalPages:number;pages:{page:number;method:"native"|"ocr"|"blank";state:"pending"|"native_complete"|"ocr_queued"|"ocr_running"|"ocr_complete"|"blank"|"failed";errorCode?:string}[]}[];error:{code:string;message:string;retryable:boolean}|null;statusUrl:string;reviewUrl:string|null}
|
||||
type Review={versionId:UUID;candidateSha256:Sha256;baseActiveVersionId:UUID|null;documents:{documentId:string;pages:{page:number;imageUrl:string;nativeText:string;ocr:{text:string;lines:{lineId:string;text:string;confidence:number;bbox:number[]}[]};candidateText:string;differences:string[];risks:string[]}[]}[]}
|
||||
type Approve={candidateSha256:Sha256;expectedActiveVersionId:UUID|null;reviewedBy:string;corrections:{documentId:string;page:number;lineId:string;expectedLineSha256:Sha256;replacementText:string}[]}
|
||||
type Reject={candidateSha256:Sha256;reviewedBy:string;reason:string}
|
||||
```
|
||||
|
||||
Reject requires current candidate hash and `review_required`; success returns `{versionId,state:"rejected",activated:false}`. Approve returns `{versionId,state,activated,activatedVersionId?,errorCode?}`; hash/state/precondition conflicts are `409`.
|
||||
|
||||
Authenticated `POST /v1/jobs` uses multipart `file`/`request` and `Idempotency-Key=<documentSha256>:<configVersion>:<pagesSha256>`; `GET /v1/jobs/:id` returns status and `GET /v1/jobs/:id/result` returns result:
|
||||
|
||||
```ts
|
||||
type Submit={documentSha256:Sha256;pages:number[];languages:["es","en"];dpi:200;engine:"paddleocr";engineVersion:"3.4.0";runtimeVersion:"3.2.2";configVersion:"ocr-v1";returnLayout:true}
|
||||
type Ack={jobId:string;status:"queued";documentSha256:Sha256;requestedPages:number[];configVersion:"ocr-v1";createdAt:ISODate}
|
||||
type JobStatus={jobId:string;status:"queued"|"running"|"succeeded"|"failed";completedPages:number;totalPages:number;error:{code:string;message:string}|null}
|
||||
type Result={schemaVersion:"1";jobId:string;documentSha256:Sha256;engine:{name:"paddleocr";version:"3.4.0";runtime:"paddlepaddle-3.2.2";device:"cpu";configVersion:"ocr-v1";dpi:200};pages:{page:number;width:number;height:number;processingMs:number;text:string;metrics:Metrics;lines:OcrLine[]}[]}
|
||||
```
|
||||
|
||||
RAG verifies job/document/config/exact ordered page set/schema before persistence. `DELETE /v1/jobs/:id` is authenticated, idempotent `204`. Network/502/503 retry twice with the same key; `429` remains queued; polling backs off 2–15 seconds. 400/401/403/409/413/422/500, deterministic render failures, and integrity mismatches are terminal and fail closed.
|
||||
|
||||
## Files
|
||||
|
||||
Modify `src/modules/{ingest/service,parsers/parser-registry,catalog/repository,catalog/reconciler}.ts`, `src/{app,api/openapi,config/env}.ts`, shared types/IDs, playground, Docker/docs; add migration 002, `src/modules/ocr/{client,dispatcher,detection,composition,artifacts,review,indexing,retention}.ts`, `ocr-service/**`, fixtures, and tests.
|
||||
|
||||
## Requirement / Scenario Coverage
|
||||
|
||||
The 17 rows assign all 34 spec scenarios exactly once.
|
||||
|
||||
| Requirement | Component | RED scenarios |
|
||||
|---|---|---|
|
||||
| O1 Routing | ingest/artifacts | native-201; OCR-202 |
|
||||
| O2 Progress | repository/API | all-documents; one-fails |
|
||||
| O3 Dispatch | repository/dispatcher | duplicate-pending; lease-recovery |
|
||||
| O4 Availability | retrieval/dispatcher | OCR-down; catalog-down |
|
||||
| O5 Enablement | config/deploy | release-gate; rollback |
|
||||
| P1 Detection | parser/detection | textual; exact-mixed-pages |
|
||||
| P2 Quality | detection | verified-blank; visible-ink-blocked |
|
||||
| P3 Composition | composition | reproducible-candidate |
|
||||
| P4 Risks | composition/review | known-ambiguities |
|
||||
| P5 OCR service | client/service | idempotent; invalid/conflict; integrity-mismatch |
|
||||
| P6 Limits | service/client | unsafe-input; pressure; deterministic-failure |
|
||||
| V1 Review auth | API/review | inspect; unauthorized/premature |
|
||||
| V2 Corrections | review/repository | atomic-commit; stale-conflict |
|
||||
| V3 Approval | review/indexing | new-activate=true/false; active-race; reusable-activate=true/false |
|
||||
| V4 Rejection | review | reject-no-index |
|
||||
| V5 Retention | retention/artifacts | expiry; active-protected; resume-delete |
|
||||
| V6 Acceptance | production | FacturaTech-34-and-codes |
|
||||
|
||||
## Testing, Threats, Rollout
|
||||
|
||||
Strict TDD: Node unit/integration fixtures plus offline pytest; run `npm test`, check, build, then human production acceptance. Threat matrix: documentation-like routing is applicable—only `.pdf` enters OCR; `requirements.txt`/executable `.md` parse but never execute, while `CMakeLists.txt`, MDX, and `README.sh` remain unsupported; RED-test each. Git selection, commit, push, and PR-command rows are N/A (no VCS automation). Private HTTP is applicable: RED-test auth, allowlist, limits, retry classes, and integrity.
|
||||
|
||||
Deploy OCR ready → migration 002 → RAG with OCR disabled → native verification → enable. Rollback disables OCR; candidates remain invisible. Auto-chain cohesive code+test work units at ≤400 authored lines; report unavoidable `size:exception` after one split.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None.
|
||||
106
openspec/changes/ocr-ingest-integration/exploration.md
Normal file
106
openspec/changes/ocr-ingest-integration/exploration.md
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
# Exploration: OCR ingest integration (`ocr-ingest-integration`)
|
||||
|
||||
**Project:** rag-service — **Phase:** sdd-explore — **Date:** 2026-09-13
|
||||
**Store:** hybrid (OpenSpec file + Engram topic `sdd/ocr-ingest-integration/explore`)
|
||||
|
||||
## Current State
|
||||
|
||||
Point 2 (knowledge lifecycle) is complete, enforced in production (`KNOWLEDGE_LIFECYCLE_ENFORCED=true`, 7 cataloged sources, 22,605 active points, legacy batch migrated and verified 2026-09-13). `npm test` passes 25/25. The contract in `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` (section "Punto 3") is closed and authoritative: every decision below defers to it; this exploration only resolves what the contract leaves open.
|
||||
|
||||
Verified implementation facts (read from code, not docs):
|
||||
|
||||
| Area | Current behavior | OCR gap |
|
||||
|---|---|---|
|
||||
| Ingest | `IngestService.ingestWithLifecycle` is fully synchronous: attempt → source lock → discover → hash → reusable lookup → `createPendingVersion` → `markIndexing` → embed → upsert → count → `markReady` → `activateVersion`, all inside the HTTP request | Must become async-capable: durable original first, `202 + statusUrl` response when OCR is needed, `source_content_hash = null` until review completes |
|
||||
| PDF parsing | `parser-registry.parseDocument` uses `pdf-parse` over the whole file, returns aggregated text only; no page callback in our code (library has `pagerender` option but we never use it; bundled pdf.js is v1.10.100, very old) | Page-level extraction required; contract allows replacing `pdf-parse` if unreliable |
|
||||
| State machine | `SourceVersionState` already includes `review_required` and `rejected`; transitions `pending→indexing→ready→active`, `failed`, `purging→purged` implemented in `CatalogRepository`; **no transition writes into `review_required` today** | Need: `indexing→review_required` (OCR candidate complete), `review_required→indexing` (approve), `review_required→rejected`, plus `review_required`/`rejected` entries in `markPurging`-eligible states (already present), retention expiry transitions |
|
||||
| Catalog repo | `createPendingVersion` requires non-null `expectedPointCount` and writes all documents with `content_hash` pre-computed; `findReusableVersion` skips when `sourceContentHash` is null (correct for OCR candidates); `hasCompleteVersionDocuments` counts `content_hash IS NULL` as incomplete | OCR flow must create a pending version **before** content exists (chunk counts unknown until review) — needs a creation mode with `expectedPointCount=0`/null-safe path, or a separate "OCR candidate" insert path |
|
||||
| Reconciler | `runOnce` treats stale `pending/indexing` older than 30 min as orphans and either recovers to `ready` or fails them | Must not kill in-flight OCR versions: needs heartbeat awareness (jobs have `heartbeat_at`/`lease_expires_at` per contract) and must recover expired OCR leases by re-dispatching with the same remote idempotency key |
|
||||
| Uploads | `multer.memoryStorage()` in `src/app.ts`, no size limit; upload writes to `/tmp` with timestamp, deleted in `finally` — no durable original | Contract requires disk-based multer with limits and durable original under `/data/ingestions/<versionId>/` before processing |
|
||||
| Env/config | `env.ts` has PostgreSQL + lifecycle tokens only | Needs `OCR_SERVICE_URL`, `OCR_INTERNAL_TOKEN`, artifact volume path, limits (50 MiB, 100 pages, timeouts), retention intervals |
|
||||
| Migrations | `migrations/001_knowledge_lifecycle.sql` applied via `rag_schema_migrations` with checksums + advisory lock; migration runner sorts `^\d+_.*\.sql$` | `002_ocr_review.sql` fits the existing runner unchanged (tables `rag_ocr_jobs`, `rag_document_pages`, `rag_review_corrections` per contract) |
|
||||
| OpenAPI | `IngestResponse` is 201-only under lifecycle; `securitySchemes.bearerAuth` already defined and used by lifecycle endpoints | Needs `202` variant with `phase`/`statusUrl`, `/ingestions/:versionId` status route, review/approve/reject routes |
|
||||
| Playground | `public/playground/app.js` supports activate/expectedActiveVersionId; no review UI | Review view (page images, native vs OCR text, corrections) required by acceptance criteria |
|
||||
| Docker | Node runtime only; `CMD ["node", "dist/server.js"]` | Needs `/data/ingestions` volume; ocr-service gets its own Dockerfile (Python CPU, PaddleOCR 3.4.0 + PaddlePaddle 3.2.2 pinned, baked models) |
|
||||
| Artifacts | No filesystem artifact manager exists; `rag_version_documents.artifact_manifest_path` and `artifact_state`/`retention_due_at` columns exist but are never written (`retention_deleted` only set on purge) | Whole artifact subsystem (write/verify 0600, manifest, retention sweeper) is new |
|
||||
|
||||
## Affected Areas
|
||||
|
||||
- `src/modules/ingest/service.ts` — async OCR branch, durable originals, version creation before content
|
||||
- `src/modules/parsers/parser-registry.ts` — per-page PDF extraction API
|
||||
- `src/modules/catalog/repository.ts` — new state transitions, OCR-safe version creation, page/job/correction persistence
|
||||
- `src/modules/catalog/reconciler.ts` — lease-aware orphan handling, OCR job re-dispatch
|
||||
- `src/app.ts` — disk multer, 202 async responses, `/ingestions/*` + review endpoints
|
||||
- `src/api/openapi.ts`, `public/playground/*` — new contracts and review UI
|
||||
- `src/config/env.ts`, `migrations/002_ocr_review.sql`, `Dockerfile`, `docs/API_RAG.md`, `docs/INGESTA.md`
|
||||
- New: `src/modules/ocr/{client,detection,artifacts,review}.ts`, `ocr-service/` (Python)
|
||||
|
||||
## Open Decisions Resolved
|
||||
|
||||
| # | Decision | Resolution | Rationale |
|
||||
|---|---|---|---|
|
||||
| 1 | Page-level PDF extraction | **Replace `pdf-parse` with `pdfjs-dist` (legacy build) used directly**, wrapped in a new `parsePdfPages(buffer)` that returns `Array<{pageNumber, text}>`. Keep `pdf-parse` only if a spike proves its `pagerender` callback reliable on Node 22 — the contract permits either, but pdf.js v1.10.100 (2018) is too old to trust for the metrics contract. Fallback order: pdfjs-dist direct → `pdf-parse` with custom pagerender. Must be settled by a fixture-backed spike in Block 2 before anything downstream builds on it | Contract mandates "callback per pagina probado con fixture" and allows substitution. pdfjs-dist is the same engine, maintained, permissively licensed (Apache-2.0), and gives page objects with text items we need for metrics |
|
||||
| 2 | Job/page state machine & progress | **Reuse the contract schema exactly**: `rag_ocr_jobs` (one per version+document, `UNIQUE(version_id, document_id)`), `rag_document_pages` per page, dispatcher uses `FOR UPDATE SKIP LOCKED`, heartbeats. Progress = `completed_pages/totalPages` surfaced via `GET /ingestions/:versionId` from a single aggregate query over jobs+pages. No new states beyond contract | Schema is already specified in the contract (Punto 3 "Persistencia OCR"); inventing alternatives violates the closed decision |
|
||||
| 3 | Multi-document semantics | **Version-level candidate gate**: a version reaches `review_required` only when ALL documents' pages are native-sufficient or OCR-complete (contract quality gate is per-document AND per-version). One failed document fails the version (`OCR_QUALITY_BLOCKED` or the document's `error_code`). Mixed sources (md + pdf) in one folder ingest: only PDFs go through OCR; non-PDF docs keep native content_hash immediately, PDF docs get `content_hash` after review | The contract's gate ("el documento solo llega a review_required cuando…") is per-document; activation is version-level, so the join is forced |
|
||||
| 4 | Reconciler behavior | Extend `recoverOrphanedVersions` to distinguish: (a) versions in `indexing` with live OCR jobs (`lease_expires_at > now`) → leave alone; (b) expired leases → re-enqueue with same `remote_idempotency_key`, reset `next_attempt_at`; (c) stale native versions → existing behavior. The dispatcher itself (new module) owns lease renewal; reconciler only recovers dead leases | Contract: "Un reconciliador periodico reenvia con la misma clave remota y recupera remote_job_id sin duplicar OCR" |
|
||||
| 5 | Durable originals/artifacts | **Write original + manifest under `/data/ingestions/<versionId>/documents/<documentArtifactId>/` BEFORE `createPendingVersion`** (contract sequence: "se guarda el original durable; se crea la version…"). `documentArtifactId` = UUIDv5(document_id). If ingest dies between artifact write and version creation, a startup/periodic sweep deletes orphan version dirs (no catalog row = not referenceable). `artifact_state` column transitions: `none → present` at write, `present → retention_deleting → retention_deleted` by sweeper | Ordering is mandated by the contract's async section; sweep covers the one crash window |
|
||||
| 6 | Purge/retention | **Single daily sweeper** per contract: acquires the same version CAS/lock used by review and purge (`withVersionExclusiveLock`), sets `artifact_state='retention_deleting'`, deletes idempotently, marks `retention_deleted`. State TTLs: review_required 30d → `rejected`/`REVIEW_EXPIRED`; failed/rejected 7d → `purging→purged`; review images 7d post-decision; active keeps original+reviewed while active +30d. An active version is never TTL-deleted | All values are fixed by the contract's "Artefactos y retencion" section |
|
||||
| 7 | Worker topology | **In-process dispatcher inside the RAG Node service** (single instance, single EasyPanel app): a `setInterval`-driven poller that claims jobs with `FOR UPDATE SKIP LOCKED`, heartbeats, polls the OCR service, never blocks the ingest HTTP path beyond the initial enqueue. No separate Node worker deploy. The OCR service itself is one worker + one replica per contract | Adding a second Node worker would double deploy surface for zero benefit at current scale (≤3 queued jobs); contract fixes OCR service at 1 worker/1 replica, and the RAG API is the only writer |
|
||||
| 8 | OCR service/image boundary | **Exactly as contracted**: `ocr-service/` in-repo, FastAPI or plain HTTP Python app (implementation detail for design phase), PaddleOCR 3.4.0 + PaddlePaddle 3.2.2 CPU pinned, models baked at build, `/v1/jobs` multipart + Idempotency-Key, `/health/live` + `/health/ready`, internal network only, `OCR_INTERNAL_TOKEN` bearer, SQLite queue on ephemeral `/data/jobs` with 24h TTL, 50 MiB limit, config allowlist. RAG's `src/modules/ocr/client.ts` validates `documentSha256`, page sets, and result schema strictly | Closed decision in contract ("Despliegue OCR", "Contrato interno del servicio OCR") |
|
||||
| 9 | Durable uploads | Replace `multer.memoryStorage()` with disk storage into a temp staging dir, enforce 50 MiB + 100 pages + PDF sanity (reject encrypted/corrupt before full render), and move the original into the version artifact dir once identity is resolved | Contract: "Antes de habilitar OCR, sustituir multer.memoryStorage() por upload a disco con limite y limpieza garantizada" |
|
||||
| 10 | Representative fixtures | **Three committed fixtures** under `tests/fixtures/ocr/`: (1) `native.pdf` — text-rich, all pages pass detection (no OCR called); (2) `scanned.pdf` — image-only pages (2–3 pages, small) requiring OCR on all non-blank pages; (3) `mixed.pdf` — some native pages + some scanned pages, containing tokens shaped like the benchmark's risk identifiers (`CBG04a`-like). Plus unit fixtures: single-page blank image, oversized-rejection corpus. The real FacturaTech PDF stays out of the repo (client data); its 34-entry acceptance run happens in production per the contract's gate. Fixtures must be tiny (<200 KiB total) and generated with a script committed for reproducibility, so tests stay offline | `npm test` has no network; OCR service tests run inside its own Python test suite (pytest) — Node tests fake the OCR client. Contract acceptance criteria name the exact behaviors the fixtures must exercise |
|
||||
|
||||
## Approaches
|
||||
|
||||
1. **Contract-as-specified, in-process dispatcher (recommended)** — implement Punto 3 exactly as the contract defines: async ingest branch with 202, new tables 002, OCR client with retries/polling, artifacts manager, review endpoints, in-process job dispatcher.
|
||||
- Pros: zero contract amendments; all closed decisions honored; single deployable RAG change + one new OCR service; reconciler extension is small; rollback = feature-flag off.
|
||||
- Cons: dispatcher lifetime couples to RAG process (acceptable: single-instance deploy already assumed); ingest service grows substantially.
|
||||
- Effort: High (largest change in RAG history), but decomposable into 7 blocks.
|
||||
|
||||
2. **Separate Node worker process** — dispatcher runs as its own EasyPanel service consuming the same PostgreSQL queue.
|
||||
- Pros: isolates OCR polling latency; RAG API stays minimal.
|
||||
- Cons: violates the current single-service ops model; duplicate pool/lock code; two deploys to coordinate for what is currently ≤3 concurrent jobs; no requirement justifies it.
|
||||
- Effort: High+.
|
||||
|
||||
3. **Minimal sync OCR first, review later** — call OCR inline during ingest with a long timeout, add review workflow in a later change.
|
||||
- Pros: fewer moving parts initially.
|
||||
- Cons: directly violates the closed decision ("responde sin esperar los minutos de OCR", 202 async, review before activation); 245s benchmark runtime would hold HTTP connections; rejected.
|
||||
|
||||
**Recommendation: Approach 1.** It is the only one consistent with the closed contract, and the in-process dispatcher keeps deployment identical in shape to today's (RAG + private OCR service).
|
||||
|
||||
## Seven Work Blocks
|
||||
|
||||
Dependencies flow strictly downward; each block maps to a chained PR slice.
|
||||
|
||||
| # | Block | Contents | Depends on | Key risk |
|
||||
|---|---|---|---|---|
|
||||
| 1 | Persistence & state machine | `migrations/002_ocr_review.sql`; repo methods: create OCRCandidateVersion (null-safe point counts), `markReviewRequired`, `approve→indexing`, `reject`, lease claim/renew methods; state-machine unit tests over fake pool | — | Migration 002 is additive (new tables + `extraction_method` values already exist); zero impact on live 001 data. Rollback: drop 002 tables (safe — no production data references them until OCR ships) |
|
||||
| 2 | Page-level PDF extraction | pdfjs-dist spike + `parsePdfPages`; detection metrics (N/A/W/R thresholds); candidate composition (`--- Page N ---` separators); canonical JSON; detection unit tests with fixtures | 1 (types only) | Library choice must be settled by spike evidence early; if pdfjs-dist legacy build fails on Node 22, fall back to `pdf-parse` custom pagerender |
|
||||
| 3 | PaddleOCR service | `ocr-service/` FastAPI app: jobs (SQLite queue, idempotency, TTL), engine wrapper (PaddleOCR 3.4.0 pinned), render pipeline (200 DPI, 25 MP cap), auth, limits, health; pytest suite with a stub engine | none (parallel-safe; contract fully specifies it) | Image build size/time (Paddle ~2 GB); models must bake at build; readiness false until models load |
|
||||
| 4 | Orchestration & artifacts | `ocr/client.ts` (retries, backoff 2–15s, strict validation); `ocr/artifacts.ts` (0600 writes, manifest, artifactId); in-process dispatcher (SKIP LOCKED, heartbeats); async ingest branch with 202 + `GET /ingestions/:versionId`; durable disk multer upload; reconciler lease-awareness | 1, 2, 3 | Biggest integration surface: touches ingest, app, reconciler; must not regress native sync path (regression tests required) |
|
||||
| 5 | Review, corrections, indexing | Risk-token detection (regex + priority rules); `GET /ingestions/:versionId/review`; approve with `candidateSha256` + per-line `expectedLineSha256` corrections (all-or-nothing 409); duplicate-reusable check; post-approval chunk/embed/index/activate via existing point-2 path; playground review UI | 4 | Correction collision semantics (409 without partial application) must be exact; concurrency with activation changes |
|
||||
| 6 | Security, retention, deployment | Retention sweeper (TTLs, artifact_state CAS); `/health` OCR section; OpenAPI full update (`202`, `/ingestions/*`, review routes, bearer); Dockerfile volume + OCR compose config; docs (API_RAG, INGESTA, deploy); no-secret review | 4, 5 | Deployment ordering: OCR service first, RAG with OCR disabled, then enable — irreversible risk is low if ordering respected |
|
||||
| 7 | End-to-end validation | Fixture-driven e2e (native/scanned/mixed), idempotent-resend test, fail-closed test (OCR down), full static suite (`npm run check`, `npm run build`, Node tests + pytest), production validation plan for the real FacturaTech PDF | 1–6 | Production acceptance (34 entries) requires human review step; cannot be automated |
|
||||
|
||||
Boundary adjustments made (with evidence): the contract's own task order lists "añadir migracion OCR y contratos TypeScript" first and extraction second, but **Block 3 (OCR service) has no code dependency on Blocks 1–2** — the contract fully specifies its HTTP interface. Starting it in parallel is the only way to fit the overall work into review-budget-sized slices without artificial sequencing. This is an execution-order adjustment only; all contract sequencing gates (OCR deployed before RAG enables it) remain.
|
||||
|
||||
## Dependencies & Risk Map
|
||||
|
||||
- **Hard dependency chain:** 1 → 2 → 4 → 5 → 6 → 7; 3 is parallel, joins at 4.
|
||||
- **Irreversible operations:** only `migrations/002_ocr_review.sql` (additive DDL — new tables, no ALTER on existing ones; safe rollback = `DROP TABLE rag_ocr_jobs, rag_document_pages, rag_review_corrections` + `DELETE FROM rag_schema_migrations WHERE name='002_ocr_review.sql'`). Qdrant unchanged until approval-driven indexing, which uses existing versioned point IDs. Production artifact volume starts empty; retention never deletes active versions.
|
||||
- **Rollback boundary:** feature-flag `OCR_INGEST_ENABLED` (new env). Off = today's native-only sync behavior byte-for-byte (regression suite proves it). The 202 async path and `/ingestions/*` routes exist but return `503 OCR_DISABLED`-style errors when off — or routes can be conditionally mounted; design phase decides. Data written while enabled (candidates, jobs) is invisible to retrieval by construction (never `active` without review).
|
||||
- **Deploy ordering (contract-mandated):** OCR service private deploy + readiness check → RAG deploy with OCR disabled → verify native regressions → enable flag → run real PDF validation.
|
||||
- **Review-budget pressure (400 lines, auto-chain):** **High across the whole change.** Realistic estimate: Block 1 ~450–550 lines (SQL+repo+tests), Block 2 ~350–450, Block 3 ~600–800 (Python, reviewed separately from Node PRs), Block 4 ~700–900, Block 5 ~600–750, Block 6 ~350–450, Block 7 ~250–350. Every block except possibly Block 7 individually risks exceeding 400 changed lines; chained/stacked PRs are mandatory, and Blocks 4 and 5 may each need two slices (e.g., 4a client+artifacts, 4b dispatcher+ingest async).
|
||||
|
||||
## Risks
|
||||
|
||||
- **pdfjs-dist integration uncertainty on Node 22** (legacy vs modern build, worker config) — mitigate with a day-one spike in Block 2, contract already allows the `pdf-parse` fallback.
|
||||
- **OCR image build fragility**: PaddlePaddle 3.2.2 CPU wheels are large and platform-sensitive; baked models add build time. Mitigate: pin exact versions, build once, keep ocr-service deploy independent (contract requirement).
|
||||
- **Ingest regression on the native path**: the async branch rewrites the hottest service. Mitigate: strict TDD per config (strict_tdd: true), the 25 existing tests must stay green, plus new native-regression tests before any OCR wiring.
|
||||
- **State-machine drift from the contract**: the contract's diagram is normative; any deviation needs contract amendment FIRST. Mitigation: spec phase encodes each transition as Given/When/Then.
|
||||
- **Playground review UI scope creep**: image rendering, side-by-side diffs, corrections editor could balloon. Mitigation: MVP review UI = native/OCR text + corrections JSON input + image links; defer polish (the playground doc says it's an internal tool).
|
||||
- **VPS2 resource contention**: OCR service capped at 3 CPU/5 GiB of the 6 vCPU/11 GiB host; a 25-page/245s benchmark job can saturate one core for minutes. Mitigate: contract's queue depth 3, single worker, and 60s/page timeout are already sized for this.
|
||||
|
||||
## Ready for Proposal
|
||||
|
||||
**Yes.** Every open decision above is resolved within the closed contract; no human input is required to draft the proposal. The proposal must state: contract-conformant implementation (Approach 1), seven chained blocks as scoped, Block-3 parallelism, feature-flag rollback, and the deploy ordering. Contract amendments are NOT needed — the contract's Punto 3 already anticipates all resolved details; the only judgment calls (worker topology, fixture strategy, library spike order) are implementation-level and recorded here.
|
||||
65
openspec/changes/ocr-ingest-integration/proposal.md
Normal file
65
openspec/changes/ocr-ingest-integration/proposal.md
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
# Proposal: OCR Ingest Integration (`ocr-ingest-integration`)
|
||||
|
||||
**Project:** rag-service — **Phase:** sdd-propose — 2026-09-13
|
||||
**Store:** hybrid (OpenSpec + Engram `sdd/ocr-ingest-integration/proposal`)
|
||||
**Contract:** `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` Punto 3 (closed)
|
||||
|
||||
## Intent
|
||||
|
||||
Auditable OCR ingestion for scanned PDFs: durable originals, async processing behind a review gate, approval before OCR content enters retrieval. Native ingestion and Punto 2 must not regress.
|
||||
|
||||
## Scope
|
||||
|
||||
**In:** async OCR branch (`202 + statusUrl`, `GET /ingestions/:versionId`, version-level review gate); migration `002_ocr_review.sql` with transitions `indexing→review_required→indexing|rejected`; page-level PDF extraction (pdfjs-dist spike; `pdf-parse` fallback), detection; private `ocr-service/` (Python, PaddleOCR pinned, idempotent `/v1/jobs`); dispatcher, OCR client, artifacts, lease-aware reconciler; review endpoints (`candidateSha256`, corrections, 409), sweeper, review UI; `OCR_INGEST_ENABLED` flag, limits, deploy order, docs, fixtures.
|
||||
|
||||
**Out:** client PDFs in repo; UI polish; extra Node workers; contract amendments. FacturaTech acceptance: production.
|
||||
|
||||
## Capabilities
|
||||
|
||||
> `openspec/specs/` is empty — all new.
|
||||
|
||||
- `ocr-ingest-orchestration`: async ingestion, durable originals, job/page state machine, dispatcher, reconciliation
|
||||
- `ocr-processing`: page extraction, detection, composition, service contract
|
||||
- `ocr-review-workflow`: review gate, corrections, approval, retention
|
||||
|
||||
Modified: none.
|
||||
|
||||
## Approach
|
||||
|
||||
Contract-as-specified (exploration Approach 1): in-process Node dispatcher plus private PaddleOCR service. Blocks: 1) persistence/state, 2) page extraction, 3) OCR service (parallel; joins at 4), 4) orchestration/artifacts, 5) review/indexing, 6) security/retention/deploy, 7) e2e validation. Durable originals precede version creation; unapproved candidates stay out of retrieval. TDD.
|
||||
|
||||
## Affected Areas
|
||||
|
||||
| Area | Impact |
|
||||
|------|--------|
|
||||
| `src/modules/ingest/service.ts`, `parsers/parser-registry.ts` | Modified: async OCR branch; per-page `parsePdfPages` |
|
||||
| `src/modules/catalog/{repository,reconciler}.ts` | Modified: new transitions; lease recovery |
|
||||
| `src/app.ts`, `src/api/openapi.ts`, `public/playground/*` | Modified: disk multer; 202; `/ingestions/*`; review UI |
|
||||
| `src/modules/ocr/`, `ocr-service/`, `migrations/002_ocr_review.sql`, `src/config/env.ts`, `Dockerfile`, docs | New |
|
||||
|
||||
## Risks
|
||||
|
||||
| Risk | Likelihood | Mitigation |
|
||||
|------|------------|------------|
|
||||
| Native ingest regression | Medium | TDD; native-regression tests first |
|
||||
| pdfjs-dist on Node 22 | Medium | Day-one spike; contract fallback |
|
||||
| OCR image fragility | Medium | Pinned versions; baked models |
|
||||
| Review budget overrun | High | Auto-chain; split blocks 4–5 |
|
||||
|
||||
## Rollback Plan
|
||||
|
||||
`OCR_INGEST_ENABLED=false` restores native-only behavior. Migration 002 rollback: drop the new tables plus migration row (no production data references them). Qdrant untouched until approval; candidates invisible to retrieval.
|
||||
|
||||
## Dependencies
|
||||
|
||||
Contract Punto 3; Punto 2 lifecycle (in production). Deploy: OCR service → RAG (flag off) → native verify → enable → acceptance.
|
||||
|
||||
## Success Criteria
|
||||
|
||||
- [ ] Native PDFs ingest sync, zero behavior change; tests green
|
||||
- [ ] Scanned PDF: `202` → OCR → `review_required` → approve → active
|
||||
- [ ] Candidate never retrievable pre-approval (fail-closed)
|
||||
- [ ] Expired leases redispatched idempotently; corrections 409 atomic
|
||||
- [ ] Sweeper honors TTLs; active versions never deleted
|
||||
- [ ] `npm run check`, `npm run build`, `npm test`, pytest offline
|
||||
- [ ] FacturaTech: 34 entries human-approved
|
||||
|
|
@ -0,0 +1,94 @@
|
|||
# OCR Ingest Orchestration Specification
|
||||
|
||||
## Purpose
|
||||
|
||||
Define routing, lifecycle, recovery, and release behavior for OCR ingestion without changing native ingestion.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement: Native and OCR Routing
|
||||
|
||||
The system MUST preserve the synchronous Point 2 path when all PDF pages have sufficient native text. Before accepting OCR work, it MUST retain the original and manifest durably and create a non-active version with a null source content hash.
|
||||
|
||||
#### Scenario: Native PDF remains synchronous
|
||||
|
||||
- GIVEN a PDF whose pages satisfy native detection
|
||||
- WHEN it is ingested
|
||||
- THEN the system MUST return the existing `201` response and MUST NOT contact OCR
|
||||
- AND lifecycle and retrieval behavior MUST remain unchanged
|
||||
|
||||
#### Scenario: OCR work is accepted asynchronously
|
||||
|
||||
- GIVEN a valid scanned or mixed PDF requiring OCR
|
||||
- WHEN ingestion is accepted
|
||||
- THEN it MUST return `202` with source/version identity, `state: indexing`, `phase: ocr_queued`, and `statusUrl`
|
||||
- AND `reviewUrl` MUST be null and `activated` MUST be false
|
||||
|
||||
### Requirement: Version-Level Progress Gate
|
||||
|
||||
The system MUST report authenticated document/page progress using `native_extracting`, `ocr_queued`, `ocr_running`, `review_required`, `indexing`, `ready`, `active`, `failed`, or `rejected`. A version SHALL reach `review_required` only after every document is native-complete or OCR-complete.
|
||||
|
||||
#### Scenario: Multi-document candidate completes
|
||||
|
||||
- GIVEN a version with native and OCR documents
|
||||
- WHEN all non-blank pages pass their applicable extraction gates
|
||||
- THEN the version MUST become `review_required` with a review URL
|
||||
- AND no document MAY be omitted from reported progress
|
||||
|
||||
#### Scenario: One document fails
|
||||
|
||||
- GIVEN any requested page fails or violates integrity or quality gates
|
||||
- WHEN version progress is evaluated
|
||||
- THEN the version MUST become `failed` with an actionable error
|
||||
- AND it MUST produce no embeddings or partial activation
|
||||
|
||||
### Requirement: Idempotent Concurrent Dispatch
|
||||
|
||||
The system MUST maintain one job per version/document, prevent concurrent ownership, and reuse its remote idempotency key during recovery.
|
||||
|
||||
#### Scenario: Duplicate ingestion while pending
|
||||
|
||||
- GIVEN the same source and original have a non-terminal OCR version
|
||||
- WHEN ingestion is retried
|
||||
- THEN it MUST return that version and job without duplicates
|
||||
|
||||
#### Scenario: Lease recovery
|
||||
|
||||
- GIVEN one job has a live lease and another has an expired lease
|
||||
- WHEN dispatch or reconciliation runs concurrently
|
||||
- THEN the live job MUST remain untouched and the expired job MUST be redispatched once with its original key
|
||||
- AND an existing remote job MUST be recovered rather than duplicated
|
||||
|
||||
### Requirement: Fail-Closed Availability
|
||||
|
||||
OCR candidates MUST remain invisible until approved and active. OCR, catalog, embedding, or vector failures MUST preserve the previous active version.
|
||||
|
||||
#### Scenario: OCR is unavailable
|
||||
|
||||
- GIVEN OCR cannot complete after allowed transient retries
|
||||
- WHEN the job is processed
|
||||
- THEN the candidate MUST fail closed without activation or engine substitution
|
||||
- AND the prior active version MUST remain retrievable
|
||||
|
||||
#### Scenario: Catalog is unavailable during retrieval
|
||||
|
||||
- GIVEN active-version resolution is unavailable
|
||||
- WHEN retrieval is requested
|
||||
- THEN the system MUST return `503` and MUST NOT query without an active-version filter
|
||||
|
||||
### Requirement: Controlled Enablement and Rollback
|
||||
|
||||
OCR MUST be enabled only after its private service is ready, RAG is deployed disabled, and native behavior is verified. Disabling it MUST stop new OCR work while preserving native ingestion and candidate invisibility.
|
||||
|
||||
#### Scenario: Release gate passes
|
||||
|
||||
- GIVEN OCR is ready and `npm run check`, `npm run build`, `npm test`, and offline OCR tests pass
|
||||
- WHEN OCR is enabled after native production verification
|
||||
- THEN new eligible ingestions MAY use the OCR path
|
||||
|
||||
#### Scenario: Operational rollback
|
||||
|
||||
- GIVEN OCR processing or acceptance fails after deployment
|
||||
- WHEN OCR is disabled or the RAG release is rolled back
|
||||
- THEN native ingestion MUST continue and existing candidates MUST remain non-retrievable
|
||||
- AND the previous active corpus MUST require no embedding recomputation
|
||||
|
|
@ -0,0 +1,105 @@
|
|||
# OCR Processing Specification
|
||||
|
||||
## Purpose
|
||||
|
||||
Define safe OCR behavior.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement: Per-Page Native Detection
|
||||
|
||||
The system MUST evaluate PDF pages independently. Native text is sufficient only when `N >= 120`, `A >= 80`, `W >= 20`, and `R <= 0.01`, under `pdf-detection-v1`.
|
||||
|
||||
#### Scenario: Textual pages bypass OCR
|
||||
|
||||
- GIVEN every page meets native thresholds
|
||||
- WHEN extraction completes
|
||||
- THEN native text MUST be selected and no OCR job MAY be created
|
||||
|
||||
#### Scenario: Mixed PDF selects exact pages
|
||||
|
||||
- GIVEN only some pages fail any native threshold
|
||||
- WHEN detection completes
|
||||
- THEN OCR MUST receive those unique, ordered, one-based pages
|
||||
- AND `N == 0` MUST always request OCR inspection
|
||||
|
||||
### Requirement: Blank and OCR Quality Gates
|
||||
|
||||
A page MAY be blank only with `inkCoverage < 0.015` and fewer than 10 OCR characters. Non-blank OCR MUST meet `nonWhitespaceCharacters >= 40`, median `>= 0.80`, p10 `>= 0.50`, and low-confidence ratio `<= 0.20`.
|
||||
|
||||
#### Scenario: Verified blank contributes no text
|
||||
|
||||
- GIVEN both blank conditions hold
|
||||
- WHEN the page is classified
|
||||
- THEN it MUST be blank and contribute no candidate text
|
||||
|
||||
#### Scenario: Visible ink fails closed
|
||||
|
||||
- GIVEN visible ink and OCR fails the quality gate
|
||||
- WHEN processing completes
|
||||
- THEN the page MUST NOT be blank and the version MUST fail with `OCR_QUALITY_BLOCKED`
|
||||
|
||||
### Requirement: Deterministic Candidate Composition
|
||||
|
||||
The system MUST use native text or OCR lines, never both per page. It SHALL order lines by top, left, and original index; pages ascending; and separate non-blank pages with `--- Page N ---`.
|
||||
|
||||
#### Scenario: Candidate is reproducible
|
||||
|
||||
- GIVEN identical validated page records
|
||||
- WHEN the candidate is composed repeatedly
|
||||
- THEN canonical text, ordering, and hashes MUST be identical
|
||||
- AND raw OCR and native extraction MUST remain available for audit
|
||||
|
||||
### Requirement: Risk Tokens Remain Uncorrected
|
||||
|
||||
The system MUST mark `\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b` tokens and elevate ambiguous, differing, unique, or context-adjacent ones. It MUST NOT auto-correct casing or `O/0`, `I/1/l`, or `S/5`.
|
||||
|
||||
#### Scenario: Known code ambiguities are flagged
|
||||
|
||||
- GIVEN OCR emits `CBGO4a`, `FATo7`, `DSAuo8`, or `NSAvo6`
|
||||
- WHEN risk analysis runs
|
||||
- THEN each token MUST be shown for human review unchanged
|
||||
|
||||
### Requirement: Private Idempotent OCR Service
|
||||
|
||||
Private PaddleOCR `3.4.0` with PaddlePaddle CPU `3.2.2` MUST require bearer authorization, allowlisted jobs, and deterministic identity for identical hash, config, and pages. It MUST NOT be Internet-accessible.
|
||||
|
||||
#### Scenario: Idempotent submission
|
||||
|
||||
- GIVEN valid requests share a key and payload
|
||||
- WHEN both are submitted
|
||||
- THEN both MUST identify one job and result without duplicate processing
|
||||
|
||||
#### Scenario: Conflicting or invalid submission
|
||||
|
||||
- GIVEN a reused key conflicts, authentication fails, or configuration is disallowed
|
||||
- WHEN submission occurs
|
||||
- THEN OCR MUST return the contracted `409`, `401/403`, or `400` class
|
||||
|
||||
#### Scenario: Result integrity mismatch
|
||||
|
||||
- GIVEN returned integrity fields differ from the request
|
||||
- WHEN RAG validates the result
|
||||
- THEN processing MUST fail without a reviewable candidate
|
||||
|
||||
### Requirement: Operational Limits and Health
|
||||
|
||||
OCR MUST enforce 50 MiB, 100 pages, 200 DPI, 25 megapixels/page, one concurrent job, one replica, queue depth three, 60 seconds/page, 15 minutes total, and 3 CPU/5 GiB maximum. Readiness MUST await pinned models; polling SHALL back off 2–15 seconds.
|
||||
|
||||
#### Scenario: Unsafe input is rejected before render
|
||||
|
||||
- GIVEN a PDF is oversized, encrypted, corrupt, too long, or decompression-unsafe
|
||||
- WHEN it is accepted for processing
|
||||
- THEN it MUST be rejected before full render without changing the active version
|
||||
|
||||
#### Scenario: Queue or transient service pressure
|
||||
|
||||
- GIVEN the queue is full or OCR is transiently unavailable
|
||||
- WHEN RAG submits work
|
||||
- THEN `429` MUST remain retryable; connection/`502`/`503` failures MUST allow at most two same-key resubmissions
|
||||
|
||||
#### Scenario: Deterministic failure is not retried
|
||||
|
||||
- GIVEN OCR returns `400`, `401`, `403`, `413`, `422`, or deterministic render failure
|
||||
- WHEN RAG handles the response
|
||||
- THEN it MUST NOT retry or switch engines
|
||||
|
|
@ -0,0 +1,106 @@
|
|||
# OCR Review Workflow Specification
|
||||
|
||||
## Purpose
|
||||
|
||||
Define authenticated review, atomic correction, activation, and retention.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement: Mandatory Authenticated Review
|
||||
|
||||
OCR versions MUST remain `review_required` and non-retrievable until approved. Status, review, images, approval, and rejection MUST require the administrator token.
|
||||
|
||||
#### Scenario: Reviewer inspects a candidate
|
||||
|
||||
- GIVEN all documents passed the version gate
|
||||
- WHEN an authorized reviewer opens review
|
||||
- THEN it MUST show image, native text, raw OCR, boxes, confidence, differences, and risks
|
||||
|
||||
#### Scenario: Unauthorized or premature access
|
||||
|
||||
- GIVEN credentials fail or the version is not reviewable
|
||||
- WHEN review or decision is requested
|
||||
- THEN it MUST reject the request without exposing artifacts or changing state
|
||||
|
||||
### Requirement: Atomic Optimistic Corrections
|
||||
|
||||
Approval MUST validate `candidateSha256`, unique line identities, and every `expectedLineSha256` before changes. Reviewed text SHALL be immutable.
|
||||
|
||||
#### Scenario: Valid corrections commit together
|
||||
|
||||
- GIVEN hashes are current and correction targets are unique
|
||||
- WHEN approval is submitted
|
||||
- THEN all replacements MUST commit together with reviewer, time, and resulting hashes
|
||||
|
||||
#### Scenario: Stale or conflicting correction
|
||||
|
||||
- GIVEN any candidate, line, identity, or box is stale or duplicated
|
||||
- WHEN approval is submitted
|
||||
- THEN the system MUST return `409` without corrections or state transition
|
||||
|
||||
### Requirement: Approval and Activation Gate
|
||||
|
||||
Approval MUST cover the version, derive final hashes, and move `review_required` to `indexing` before chunking. Activation SHALL use Point 2 verification and the expected active version.
|
||||
|
||||
#### Scenario: Approved content becomes active
|
||||
|
||||
- GIVEN corrections are valid and the expected active version is current
|
||||
- WHEN indexing and point verification succeed
|
||||
- THEN the new version MUST become active atomically and the previous one superseded
|
||||
- AND only reviewed content MAY be retrieved
|
||||
|
||||
#### Scenario: Active version changes during review
|
||||
|
||||
- GIVEN another version activates during review
|
||||
- WHEN approved indexing completes
|
||||
- THEN the candidate MUST remain `ready`, return `409 ACTIVE_VERSION_CHANGED`, and not replace it
|
||||
|
||||
#### Scenario: Reusable content already exists
|
||||
|
||||
- GIVEN final content, fingerprint, and metadata match a reusable version
|
||||
- WHEN approval checks the active-version precondition
|
||||
- THEN the candidate MUST be `rejected` with `DUPLICATE_REUSABLE_VERSION`
|
||||
- AND only the existing version MAY be activated
|
||||
|
||||
### Requirement: Rejection and Failure Isolation
|
||||
|
||||
Rejection MUST produce `rejected`, preserve temporary audit evidence, and generate no embeddings. Failed or rejected candidates SHALL never alter the active version.
|
||||
|
||||
#### Scenario: Reviewer rejects candidate
|
||||
|
||||
- GIVEN a version is `review_required`
|
||||
- WHEN an authorized reviewer rejects it
|
||||
- THEN it MUST become `rejected` with audit data and no indexing
|
||||
|
||||
### Requirement: Retention Safety
|
||||
|
||||
Private artifacts MUST use `0600` permissions and verifiable hashes. The system MUST retain review 30 days, failed/rejected artifacts 7 days, review images 7 days post-decision, OCR copies until transfer or 24 hours, and active artifacts until 30 days after supersession. Cleanup MUST be idempotent and never delete active versions.
|
||||
|
||||
#### Scenario: Review expires
|
||||
|
||||
- GIVEN a version remains `review_required` for 30 days
|
||||
- WHEN daily retention runs exclusively
|
||||
- THEN it MUST become `rejected` with `REVIEW_EXPIRED` before removal
|
||||
|
||||
#### Scenario: Active artifacts are protected
|
||||
|
||||
- GIVEN an active version is past a TTL
|
||||
- WHEN retention runs concurrently with review or purge
|
||||
- THEN its original, reviewed text, raw OCR, and manifest MUST remain
|
||||
- AND cleanup MUST not escape the version directory or expose a public URL
|
||||
|
||||
#### Scenario: Interrupted cleanup resumes
|
||||
|
||||
- GIVEN deletion stopped after retention entered its deleting state
|
||||
- WHEN cleanup runs again
|
||||
- THEN deletion MUST resume safely and finish in the deleted state without affecting other versions
|
||||
|
||||
### Requirement: Human-Approved Production Acceptance
|
||||
|
||||
The release MUST prove corrected content after approval; confidence SHALL NOT authorize activation.
|
||||
|
||||
#### Scenario: FacturaTech acceptance succeeds
|
||||
|
||||
- GIVEN the real FacturaTech PDF has been reviewed and its 34 entries approved
|
||||
- WHEN the version becomes active and retrieval is queried
|
||||
- THEN all 34 entries and `CBG04a`, `FAT07`, `DSAU08`, and `NSAV06` MUST be returned exactly
|
||||
79
openspec/changes/ocr-ingest-integration/tasks.md
Normal file
79
openspec/changes/ocr-ingest-integration/tasks.md
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
# Tasks: OCR Ingest Integration
|
||||
|
||||
## Forecast
|
||||
|
||||
Estimate: 3,400–4,300 lines.
|
||||
|
||||
Decision needed before apply: No
|
||||
Chained PRs recommended: Yes
|
||||
Chain strategy: feature-branch-chain
|
||||
400-line budget risk: High
|
||||
|
||||
Tracker feature/ocr-ingest-integration is draft/no-merge and sole main target. PR1 targets tracker; children target predecessors. Coverage: 17 requirements/34 scenarios (O10/P12/V12).
|
||||
|
||||
### Units: command; harness; rollback
|
||||
|
||||
1 Migration: tsx --test tests/catalog/migration-002.test.ts; N/A schema-only; drop OCR schema
|
||||
2 Repository: tsx --test tests/catalog/repository-ocr.test.ts; fake pool; revert candidate/leases
|
||||
3 Extraction: tsx --test tests/parsers/pdf-pages.test.ts; PDF spike; revert parser/fixtures
|
||||
4 Detection: tsx --test tests/ocr/detection.test.ts; mixed PDF; revert transforms
|
||||
5 Service: pytest ocr-service/tests -k auth; curl 401/409; remove ocr-service
|
||||
6 Render: pytest ocr-service/tests -k render; Docker PNG; revert wrapper/image
|
||||
7 Client: tsx --test tests/ocr/client.test.ts; retry stub; revert client/artifacts
|
||||
8 Dispatcher: tsx --test tests/ocr/dispatcher.test.ts; curl 202/status; revert routing
|
||||
9 Review: tsx --test tests/ocr/review.test.ts; curl review/409; revert review/routes
|
||||
10 Approval: tsx --test tests/ocr/approval.test.ts; playground approval; revert indexing/UI
|
||||
11 Retention: tsx --test tests/ocr/retention.test.ts; fake-fs sweep; revert retention
|
||||
12 Config: npm run check; N/A static-only; revert config/docs
|
||||
13 E2E: npm test; production OCR flow; disable OCR flag
|
||||
|
||||
## 1 Persistence (Units 1–2; O2/O3)
|
||||
|
||||
- [x] 1.1 RED: duplicate pending, lease recovery, review transitions
|
||||
- [x] 1.2 GREEN: `migrations/002_ocr_review.sql` schema and unique index
|
||||
- [x] 1.3 GREEN: `src/modules/catalog/repository.ts` candidates and leases
|
||||
- [x] 1.4 REFACTOR after green
|
||||
|
||||
## 2 Extraction (Units 3–4; P1–P4)
|
||||
|
||||
- [x] 2.1 Spike `scripts/spike-pdfjs.ts` for Node 22 parser
|
||||
- [x] 2.2 RED: PDF-only routing; text files never execute; unsupported build/script formats; pages, blanks, hashes, risks
|
||||
- [x] 2.3 GREEN: `tests/fixtures/ocr/` and `src/modules/parsers/parser-registry.ts`
|
||||
- [x] 2.4 GREEN: `src/modules/ocr/detection.ts` thresholds
|
||||
- [x] 2.5 GREEN: `src/modules/ocr/composition.ts` text, hashes, risks
|
||||
|
||||
## 3 OCR Service (Units 5–6; P5/P6)
|
||||
|
||||
- [x] 3.1 RED HTTP: bearer 401; idempotency/conflict 409; allowlist; limits 413/422; pressure 429; no-retry; integrity mismatch
|
||||
- [x] 3.2 GREEN: `ocr-service/` API, queue, allowlist, limits, health
|
||||
- [x] 3.3 GREEN: PaddleOCR 3.4.0, 200 DPI, schema, stub
|
||||
- [x] 3.4 GREEN: `ocr-service/Dockerfile` pinned wheels/models/resources
|
||||
|
||||
## 4 Orchestration (Units 7–8; O1–O4)
|
||||
|
||||
- [x] 4.1 RED HTTP: flag-off 201; OCR 202/status; OCR-down fail-closed; catalog-down 503; retries/integrity
|
||||
- [x] 4.2 GREEN: `src/modules/ocr/client.ts` retries and integrity
|
||||
- [x] 4.3 GREEN: `src/modules/ocr/artifacts.ts` originals, manifest, sweep
|
||||
- [x] 4.4 GREEN: `src/modules/ocr/dispatcher.ts` leases; `src/modules/catalog/reconciler.ts`
|
||||
- [x] 4.5 GREEN: `src/modules/ingest/service.ts` branch; `src/app.ts` upload/status
|
||||
|
||||
## 5 Review (Units 9–10; V1–V4)
|
||||
|
||||
- [x] 5.1 RED: auth/premature access; atomic corrections/stale 409; new/reusable activation race; reject/no embeddings
|
||||
- [x] 5.2 GREEN: `src/modules/ocr/review.ts` view, corrections, conflicts
|
||||
- [x] 5.3 GREEN: `src/modules/ocr/indexing.ts` CAS; new/reusable activate_requested=true activates, false stays ready
|
||||
- [x] 5.4 GREEN: rejection and `public/playground/` review UI
|
||||
|
||||
## 6 Retention/Deploy (Units 11–12; O5/V5)
|
||||
|
||||
- [x] 6.1 RED: TTL/CAS/resume/active protection; flag-off native-only and candidate invisibility
|
||||
- [x] 6.2 GREEN: `src/modules/ocr/retention.ts`
|
||||
- [x] 6.3 GREEN: `src/api/openapi.ts` OCR/status/review contracts
|
||||
- [x] 6.4 GREEN: `src/config/env.ts`, `Dockerfile`; follow `docs/CONTRATO_CICLO_VIDA_Y_OCR.md` (read-only)
|
||||
|
||||
## 7 E2E (Unit 13; V6)
|
||||
|
||||
- [x] 7.1 E2E: native 201; scanned 202→review→approve→active; mixed
|
||||
- [x] 7.2 E2E: resend, OCR-down fail-closed, catalog-down 503
|
||||
- [x] 7.3 Gate: check, build, test, offline pytest
|
||||
- [ ] 7.4 Production: flag-off verify→enable→FacturaTech 34 entries; CBG04a/FAT07/DSAU08/NSAV06
|
||||
59
openspec/config.yaml
Normal file
59
openspec/config.yaml
Normal file
|
|
@ -0,0 +1,59 @@
|
|||
schema: spec-driven
|
||||
|
||||
context: |
|
||||
Project: rag-service (RAG knowledge service for enterprise AI tooling)
|
||||
Tech stack: Node.js 22 (ESM), TypeScript 5.8 strict, Express 4, tsx runner
|
||||
Architecture: modular service — src/modules/{ingest,process,parsers,retrieve,answer,catalog,embeddings,vectorstore,logs}, src/api (OpenAPI), src/shared
|
||||
Data stores: PostgreSQL (pg) + Qdrant (@qdrant/js-client-rest) + OpenAI embeddings
|
||||
Testing: node:test + node:assert/strict executed via `npm test` (NODE_ENV=test tsx --test tests/**/*.test.ts)
|
||||
Quality: `npm run check` (tsc --noEmit, strict), `npm run build` (tsc emit to dist/)
|
||||
Deploy: Docker + EasyPanel; migrations are manual scripts (migrate:lifecycle, migrate:legacy:lifecycle)
|
||||
No linter, formatter, or coverage tooling is configured.
|
||||
|
||||
strict_tdd: true
|
||||
|
||||
testing:
|
||||
projects:
|
||||
- path: ./
|
||||
stack: Node.js 22 + TypeScript 5.8 (ESM)
|
||||
test_command: npm test
|
||||
test_framework: node:test (node --test via tsx)
|
||||
layers:
|
||||
unit: true
|
||||
integration: true
|
||||
e2e: false
|
||||
coverage:
|
||||
available: false
|
||||
command: ""
|
||||
quality:
|
||||
linter: false
|
||||
type_checker: true
|
||||
type_checker_command: npm run check
|
||||
formatter: false
|
||||
|
||||
rules:
|
||||
proposal:
|
||||
- Include rollback plan for risky changes (migrations and Qdrant/PostgreSQL data operations)
|
||||
- Reference existing module boundaries under src/modules/
|
||||
specs:
|
||||
- Use Given/When/Then for scenarios
|
||||
- Use RFC 2119 keywords (MUST, SHALL, SHOULD, MAY)
|
||||
design:
|
||||
- Document architecture decisions with rationale
|
||||
- Keep module boundaries; parsers/ingest extensions must follow the parser-registry pattern
|
||||
tasks:
|
||||
- Group by phase, use hierarchical numbering
|
||||
- Keep tasks completable in one session
|
||||
- Respect review_budget_lines: 400 (auto-chain delivery when exceeded)
|
||||
apply:
|
||||
guidelines:
|
||||
- Follow existing code patterns (ESM imports with .js extensions, strict types)
|
||||
- Never read or expose .env* files, llaves, or backups/ contents
|
||||
tdd: true
|
||||
test_command: npm test
|
||||
verify:
|
||||
test_command: npm test
|
||||
build_command: npm run build
|
||||
coverage_threshold: 0
|
||||
archive:
|
||||
- Warn before merging destructive deltas
|
||||
0
openspec/specs/.gitkeep
Normal file
0
openspec/specs/.gitkeep
Normal file
|
|
@ -9,7 +9,7 @@
|
|||
"start": "node dist/server.js",
|
||||
"migrate:lifecycle": "node dist/modules/catalog/migrations.js",
|
||||
"migrate:legacy:lifecycle": "node dist/scripts/migrate-legacy-lifecycle.js",
|
||||
"test": "NODE_ENV=test tsx --test tests/**/*.test.ts",
|
||||
"test": "NODE_ENV=test tsx --test tests/*.test.ts tests/**/*.test.ts",
|
||||
"check": "tsc --noEmit -p tsconfig.json"
|
||||
},
|
||||
"dependencies": {
|
||||
|
|
|
|||
|
|
@ -10,6 +10,9 @@ const manualLogButton = document.getElementById("manualLogButton");
|
|||
const presetDocs = document.getElementById("presetDocs");
|
||||
const presetRagDocs = document.getElementById("presetRagDocs");
|
||||
const presetCode = document.getElementById("presetCode");
|
||||
const loadReviewButton = document.getElementById("loadReviewButton");
|
||||
const approveReviewButton = document.getElementById("approveReviewButton");
|
||||
const rejectReviewButton = document.getElementById("rejectReviewButton");
|
||||
|
||||
const healthResult = document.getElementById("healthResult");
|
||||
const ingestResult = document.getElementById("ingestResult");
|
||||
|
|
@ -24,6 +27,12 @@ const contextStatusText = document.getElementById("contextStatusText");
|
|||
const contextScopeText = document.getElementById("contextScopeText");
|
||||
const logsResult = document.getElementById("logsResult");
|
||||
const logCounterValue = document.getElementById("logCounterValue");
|
||||
const reviewCandidate = document.getElementById("reviewCandidate");
|
||||
const reviewResult = document.getElementById("reviewResult");
|
||||
const reviewVersionId = document.getElementById("reviewVersionId");
|
||||
const reviewToken = document.getElementById("reviewToken");
|
||||
const reviewedBy = document.getElementById("reviewedBy");
|
||||
const rejectionReason = document.getElementById("rejectionReason");
|
||||
|
||||
const ingestSourceType = document.getElementById("ingestSourceType");
|
||||
const ingestScopeMode = document.getElementById("ingestScopeMode");
|
||||
|
|
@ -69,6 +78,8 @@ let lastBootstrapMeta = null;
|
|||
let chatHistory = [];
|
||||
let availableScopes = [];
|
||||
let lastInteraction = null;
|
||||
let loadedReview = null;
|
||||
let reviewImageUrls = [];
|
||||
|
||||
let currentUploadType = null; // 'file' o 'folder'
|
||||
|
||||
|
|
@ -179,6 +190,72 @@ function request(url, payload, method = "POST") {
|
|||
});
|
||||
}
|
||||
|
||||
async function authorizedReviewRequest(path, payload) {
|
||||
const response = await fetch(`/ingestions/${encodeURIComponent(reviewVersionId.value.trim())}${path}`, {
|
||||
method: payload ? "POST" : "GET",
|
||||
headers: { Authorization: `Bearer ${reviewToken.value}`, ...(payload ? { "Content-Type": "application/json" } : {}) },
|
||||
body: payload ? JSON.stringify(payload) : undefined
|
||||
});
|
||||
const data = await response.json();
|
||||
if (!response.ok) throw new Error(`${data.code || response.status}: ${data.error || "Review request failed"}`);
|
||||
return data;
|
||||
}
|
||||
|
||||
function reviewText(name, value) {
|
||||
const element = document.createElement("pre");
|
||||
element.textContent = `${name}\n${Array.isArray(value) ? value.join("\n") : value || "(empty)"}`;
|
||||
return element;
|
||||
}
|
||||
|
||||
async function renderReview(candidate) {
|
||||
for (const url of reviewImageUrls) URL.revokeObjectURL(url);
|
||||
reviewImageUrls = [];
|
||||
reviewCandidate.replaceChildren();
|
||||
const identity = document.createElement("p");
|
||||
identity.textContent = `Candidate SHA-256: ${candidate.candidateSha256} · Base active version: ${candidate.baseActiveVersionId || "none"}`;
|
||||
reviewCandidate.append(identity);
|
||||
for (const document of candidate.documents) for (const page of document.pages) {
|
||||
const card = document.createElement("article");
|
||||
card.className = "review-page";
|
||||
const title = document.createElement("h3");
|
||||
title.textContent = `${document.documentId} · Page ${page.page}`;
|
||||
const image = document.createElement("img");
|
||||
image.className = "review-image";
|
||||
image.alt = `Source page ${page.page}`;
|
||||
card.append(title, image, reviewText("Native text", page.nativeText), reviewText("Raw OCR", page.ocr.text), reviewText("Candidate text", page.candidateText), reviewText("Differences", page.differences), reviewText("Risks", page.risks));
|
||||
for (const line of page.ocr.lines) {
|
||||
const label = document.createElement("label");
|
||||
label.className = "review-line";
|
||||
const detail = document.createElement("span");
|
||||
detail.textContent = `${line.lineId} · confidence ${line.confidence} · bbox ${line.bbox.join(", ")}`;
|
||||
const input = document.createElement("textarea");
|
||||
input.value = line.text;
|
||||
Object.assign(input.dataset, { original: line.text, documentId: document.documentId, page: String(page.page), lineId: line.lineId, expectedLineSha256: line.lineSha256 });
|
||||
label.append(detail, input);
|
||||
card.append(label);
|
||||
}
|
||||
reviewCandidate.append(card);
|
||||
void fetch(page.imageUrl, { headers: { Authorization: `Bearer ${reviewToken.value}` } }).then(async (response) => {
|
||||
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
||||
const url = URL.createObjectURL(await response.blob());
|
||||
reviewImageUrls.push(url);
|
||||
image.src = url;
|
||||
}).catch((error) => { image.alt = `Protected image unavailable: ${error}`; });
|
||||
}
|
||||
}
|
||||
|
||||
function reviewDecisionBase() {
|
||||
if (!loadedReview) throw new Error("Load a current candidate first");
|
||||
const reviewer = reviewedBy.value.trim();
|
||||
if (!reviewer) throw new Error("Reviewer identity is required");
|
||||
return { candidateSha256: loadedReview.candidateSha256, reviewedBy: reviewer };
|
||||
}
|
||||
|
||||
function setReviewDecisionEnabled(enabled) {
|
||||
approveReviewButton.disabled = !enabled;
|
||||
rejectReviewButton.disabled = !enabled;
|
||||
}
|
||||
|
||||
function renderBootstrapContext() {
|
||||
if (!lastBootstrapContext) {
|
||||
bootstrapContextResult.textContent = "Aun no hay bootstrap cargado.";
|
||||
|
|
@ -468,6 +545,54 @@ ingestButton.addEventListener("click", async () => {
|
|||
}
|
||||
});
|
||||
|
||||
loadReviewButton.addEventListener("click", async () => {
|
||||
reviewResult.textContent = "Loading authenticated review...";
|
||||
try {
|
||||
loadedReview = await authorizedReviewRequest("/review");
|
||||
await renderReview(loadedReview);
|
||||
setReviewDecisionEnabled(true);
|
||||
reviewResult.textContent = `Loaded review_required version ${loadedReview.versionId}.`;
|
||||
} catch (error) {
|
||||
loadedReview = null;
|
||||
setReviewDecisionEnabled(false);
|
||||
reviewCandidate.textContent = "No candidate loaded.";
|
||||
reviewResult.textContent = String(error);
|
||||
}
|
||||
});
|
||||
|
||||
approveReviewButton.addEventListener("click", async () => {
|
||||
reviewResult.textContent = "Submitting reviewed text...";
|
||||
try {
|
||||
const corrections = [...reviewCandidate.querySelectorAll(".review-line textarea")]
|
||||
.filter((input) => input.value !== input.dataset.original)
|
||||
.map((input) => ({
|
||||
documentId: input.dataset.documentId,
|
||||
page: Number(input.dataset.page),
|
||||
lineId: input.dataset.lineId,
|
||||
expectedLineSha256: input.dataset.expectedLineSha256,
|
||||
replacementText: input.value
|
||||
}));
|
||||
const result = await authorizedReviewRequest("/approve", { ...reviewDecisionBase(), expectedActiveVersionId: loadedReview.baseActiveVersionId, corrections });
|
||||
reviewResult.textContent = format(result);
|
||||
setReviewDecisionEnabled(false);
|
||||
} catch (error) {
|
||||
reviewResult.textContent = String(error);
|
||||
}
|
||||
});
|
||||
|
||||
rejectReviewButton.addEventListener("click", async () => {
|
||||
reviewResult.textContent = "Rejecting candidate...";
|
||||
try {
|
||||
const reason = rejectionReason.value.trim();
|
||||
if (!reason) throw new Error("Rejection reason is required");
|
||||
const result = await authorizedReviewRequest("/reject", { ...reviewDecisionBase(), reason });
|
||||
reviewResult.textContent = format(result);
|
||||
setReviewDecisionEnabled(false);
|
||||
} catch (error) {
|
||||
reviewResult.textContent = String(error);
|
||||
}
|
||||
});
|
||||
|
||||
cleanupButton.addEventListener("click", async () => {
|
||||
if (!cleanupScopeSelect.value) {
|
||||
cleanupResult.textContent = "Error: Debes seleccionar un scope primero.";
|
||||
|
|
|
|||
|
|
@ -29,11 +29,40 @@
|
|||
|
||||
<section class="tabs">
|
||||
<button class="tab-button active" data-tab="ingest">Ingesta</button>
|
||||
<button class="tab-button" data-tab="review">OCR Review</button>
|
||||
<button class="tab-button" data-tab="cleanup">Limpieza</button>
|
||||
<button class="tab-button" data-tab="bootstrap">Bootstrap</button>
|
||||
<button class="tab-button" data-tab="chat">Chat</button>
|
||||
</section>
|
||||
|
||||
<section id="tab-review" class="tab-panel">
|
||||
<article class="panel">
|
||||
<h2>OCR Candidate Review</h2>
|
||||
<p class="helper">Inspect every page and submit an authenticated approval or rejection. OCR content remains unavailable to retrieval until approval completes.</p>
|
||||
<div class="grid single-grid">
|
||||
<label>Version ID
|
||||
<input id="reviewVersionId" autocomplete="off" placeholder="OCR version UUID" />
|
||||
</label>
|
||||
<label>Administrator token
|
||||
<input id="reviewToken" type="password" autocomplete="off" />
|
||||
</label>
|
||||
<label>Reviewed by
|
||||
<input id="reviewedBy" autocomplete="off" placeholder="Reviewer identity" />
|
||||
</label>
|
||||
<label>Rejection reason
|
||||
<textarea id="rejectionReason" rows="2" placeholder="Required only when rejecting"></textarea>
|
||||
</label>
|
||||
</div>
|
||||
<div class="actions">
|
||||
<button id="loadReviewButton">Load candidate</button>
|
||||
<button id="approveReviewButton" disabled>Approve reviewed text</button>
|
||||
<button id="rejectReviewButton" class="danger" disabled>Reject candidate</button>
|
||||
</div>
|
||||
<div id="reviewCandidate" class="review-candidate" aria-live="polite">No candidate loaded.</div>
|
||||
<pre id="reviewResult">No review request submitted.</pre>
|
||||
</article>
|
||||
</section>
|
||||
|
||||
<section id="tab-ingest" class="tab-panel active">
|
||||
<article class="panel">
|
||||
<h2>Ingesta</h2>
|
||||
|
|
|
|||
|
|
@ -182,6 +182,23 @@ button.secondary {
|
|||
border: 1px solid var(--border);
|
||||
}
|
||||
|
||||
button.danger { background: var(--danger); color: #0b1020; }
|
||||
button:disabled { cursor: not-allowed; opacity: 0.45; }
|
||||
|
||||
.review-candidate { display: grid; gap: 16px; margin: 20px 0; }
|
||||
.review-page {
|
||||
display: grid;
|
||||
gap: 14px;
|
||||
padding: 18px;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 16px;
|
||||
background: #0b1020;
|
||||
}
|
||||
.review-page pre { margin: 0; min-height: 0; }
|
||||
.review-image { max-width: 100%; max-height: 520px; object-fit: contain; justify-self: start; }
|
||||
.review-line { display: grid; gap: 6px; }
|
||||
.review-line span { color: var(--muted); font-size: 12px; }
|
||||
|
||||
.actions, .checkbox {
|
||||
display: flex;
|
||||
gap: 12px;
|
||||
|
|
|
|||
36
scripts/spike-pdfjs.ts
Normal file
36
scripts/spike-pdfjs.ts
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { readFile } from "node:fs/promises";
|
||||
import pdf from "pdf-parse";
|
||||
|
||||
interface PdfPageData {
|
||||
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||
items: Array<{ str?: string }>;
|
||||
}>;
|
||||
}
|
||||
|
||||
const fixtureUrl = new URL("../tests/fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||
const input = process.argv[2] ? new URL(`file://${process.argv[2]}`) : fixtureUrl;
|
||||
const nodeMajor = Number.parseInt(process.versions.node.split(".")[0] ?? "0", 10);
|
||||
|
||||
assert.ok(nodeMajor >= 22, `Node 22 or newer is required; received ${process.versions.node}`);
|
||||
|
||||
const pages: Array<{ page: number; text: string }> = [];
|
||||
const bytes = Uint8Array.from(await readFile(input));
|
||||
const result = await pdf(bytes as Buffer, {
|
||||
version: "v2.0.550",
|
||||
pagerender: async (pageData: PdfPageData) => {
|
||||
const textContent = await pageData.getTextContent({ normalizeWhitespace: false, disableCombineTextItems: false });
|
||||
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||
pages.push({ page: pages.length + 1, text });
|
||||
return text;
|
||||
}
|
||||
});
|
||||
|
||||
assert.equal(result.numpages, pages.length);
|
||||
assert.deepEqual(pages, [
|
||||
{ page: 1, text: "Native page one." },
|
||||
{ page: 2, text: "" },
|
||||
{ page: 3, text: "Native page three." }
|
||||
]);
|
||||
|
||||
console.log(JSON.stringify({ node: process.versions.node, parser: `pdf-parse/pdfjs-${result.version}`, pages }));
|
||||
|
|
@ -30,6 +30,7 @@ export const openApiDocument = {
|
|||
{ name: "Discovery", description: "API contract and browser playground." },
|
||||
{ name: "Status", description: "Service capabilities and available resources." },
|
||||
{ name: "Ingestion", description: "Knowledge ingestion and cleanup." },
|
||||
{ name: "OCR Review", description: "Authenticated OCR progress, review, correction, and decisions." },
|
||||
{ name: "Lifecycle", description: "Knowledge source versions, activation, rollback, and purge." },
|
||||
{ name: "Retrieval", description: "Context retrieval and model-backed answers." },
|
||||
{ name: "Evaluation", description: "Evaluation log capture and review." }
|
||||
|
|
@ -222,7 +223,7 @@ export const openApiDocument = {
|
|||
},
|
||||
responses: {
|
||||
"201": jsonResponse("Lifecycle ingestion completed and committed.", ref("IngestResponse")),
|
||||
"202": jsonResponse("Legacy ingestion completed and accepted while lifecycle enforcement is disabled.", ref("IngestResponse")),
|
||||
"202": jsonResponse("OCR ingestion accepted, or legacy ingestion accepted while lifecycle enforcement is disabled.", { oneOf: [ref("OcrAccepted"), ref("IngestResponse")] }),
|
||||
"400": jsonResponse("Invalid lifecycle activation request.", ref("Error")),
|
||||
"409": jsonResponse("Concurrent active version change or reusable version in progress.", ref("Error")),
|
||||
"422": jsonResponse("Empty or unsupported source.", ref("Error")),
|
||||
|
|
@ -245,7 +246,7 @@ export const openApiDocument = {
|
|||
},
|
||||
responses: {
|
||||
"201": jsonResponse("Lifecycle upload ingested successfully.", ref("UploadIngestResponse")),
|
||||
"202": jsonResponse("Legacy upload ingested successfully while lifecycle enforcement is disabled.", ref("UploadIngestResponse")),
|
||||
"202": jsonResponse("OCR upload accepted, or legacy upload accepted while lifecycle enforcement is disabled.", { oneOf: [ref("OcrUploadAccepted"), ref("UploadIngestResponse")] }),
|
||||
"400": jsonResponse("The file field is missing.", ref("Error"), { ok: false, error: "Missing file upload" }),
|
||||
"409": jsonResponse("Concurrent active version change or reusable version in progress.", ref("Error")),
|
||||
"422": jsonResponse("Empty or unsupported source.", ref("Error")),
|
||||
|
|
@ -254,6 +255,60 @@ export const openApiDocument = {
|
|||
}
|
||||
}
|
||||
},
|
||||
"/ingestions/{versionId}": {
|
||||
get: {
|
||||
tags: ["OCR Review"], summary: "Get OCR ingestion progress", security: [{ bearerAuth: [] }],
|
||||
parameters: [{ name: "versionId", in: "path", required: true, schema: { type: "string", format: "uuid" } }],
|
||||
responses: {
|
||||
"200": jsonResponse("Current document and page progress.", ref("IngestionStatus")),
|
||||
"401": jsonResponse("Missing or invalid lifecycle admin token.", ref("Error")),
|
||||
"404": jsonResponse("OCR is disabled or the ingestion does not exist.", ref("Error")),
|
||||
"503": serverError, "500": serverError
|
||||
}
|
||||
}
|
||||
},
|
||||
"/ingestions/{versionId}/review": {
|
||||
get: {
|
||||
tags: ["OCR Review"], summary: "Inspect an OCR candidate", security: [{ bearerAuth: [] }],
|
||||
parameters: [{ name: "versionId", in: "path", required: true, schema: { type: "string", format: "uuid" } }],
|
||||
responses: {
|
||||
"200": jsonResponse("Review candidate with page evidence.", ref("OcrReview")),
|
||||
"401": jsonResponse("Missing or invalid lifecycle admin token.", ref("Error")),
|
||||
"404": jsonResponse("OCR is disabled or the candidate does not exist.", ref("Error")),
|
||||
"409": jsonResponse("The version is not awaiting review.", ref("Error")),
|
||||
"503": serverError, "500": serverError
|
||||
}
|
||||
}
|
||||
},
|
||||
"/ingestions/{versionId}/approve": {
|
||||
post: {
|
||||
tags: ["OCR Review"], summary: "Correct and approve an OCR candidate", security: [{ bearerAuth: [] }],
|
||||
parameters: [{ name: "versionId", in: "path", required: true, schema: { type: "string", format: "uuid" } }],
|
||||
requestBody: { required: true, content: jsonContent(ref("OcrApprovalRequest")) },
|
||||
responses: {
|
||||
"200": jsonResponse("Candidate indexed and settled according to activation intent.", ref("OcrDecisionResponse")),
|
||||
"401": jsonResponse("Missing or invalid lifecycle admin token.", ref("Error")),
|
||||
"404": jsonResponse("OCR is disabled or the candidate does not exist.", ref("Error")),
|
||||
"409": jsonResponse("Candidate, correction, state, or active-version precondition conflict.", ref("Error")),
|
||||
"503": serverError, "500": serverError
|
||||
}
|
||||
}
|
||||
},
|
||||
"/ingestions/{versionId}/reject": {
|
||||
post: {
|
||||
tags: ["OCR Review"], summary: "Reject an OCR candidate", security: [{ bearerAuth: [] }],
|
||||
parameters: [{ name: "versionId", in: "path", required: true, schema: { type: "string", format: "uuid" } }],
|
||||
requestBody: { required: true, content: jsonContent(ref("OcrRejectionRequest")) },
|
||||
responses: {
|
||||
"200": jsonResponse("Candidate rejected without indexing or activation.", ref("OcrDecisionResponse")),
|
||||
"400": jsonResponse("Required rejection fields are missing.", ref("Error")),
|
||||
"401": jsonResponse("Missing or invalid lifecycle admin token.", ref("Error")),
|
||||
"404": jsonResponse("OCR is disabled or the candidate does not exist.", ref("Error")),
|
||||
"409": jsonResponse("Candidate hash or review state conflict.", ref("Error")),
|
||||
"503": serverError, "500": serverError
|
||||
}
|
||||
}
|
||||
},
|
||||
"/cleanup": {
|
||||
post: {
|
||||
tags: ["Ingestion"],
|
||||
|
|
@ -402,7 +457,8 @@ export const openApiDocument = {
|
|||
required: ["ok", "error"],
|
||||
properties: {
|
||||
ok: { type: "boolean", const: false },
|
||||
error: { type: "string" }
|
||||
error: { type: "string" },
|
||||
code: { type: "string" }
|
||||
}
|
||||
},
|
||||
Scope: {
|
||||
|
|
@ -490,6 +546,77 @@ export const openApiDocument = {
|
|||
}
|
||||
]
|
||||
},
|
||||
OcrPhase: {
|
||||
type: "string",
|
||||
enum: ["native_extracting", "ocr_queued", "ocr_running", "review_required", "indexing", "ready", "active", "failed", "rejected"]
|
||||
},
|
||||
OcrAccepted: {
|
||||
type: "object",
|
||||
required: ["accepted", "sourceId", "versionId", "versionNumber", "state", "phase", "statusUrl", "reviewUrl", "activated"],
|
||||
properties: {
|
||||
accepted: { type: "boolean", const: true }, sourceId: { type: "string" }, versionId: { type: "string", format: "uuid" },
|
||||
versionNumber: { type: "integer" }, state: { type: "string", const: "indexing" }, phase: { type: "string", const: "ocr_queued" },
|
||||
statusUrl: { type: "string" }, reviewUrl: { type: "null" }, activated: { type: "boolean", const: false }
|
||||
}
|
||||
},
|
||||
OcrUploadAccepted: {
|
||||
allOf: [ref("OcrAccepted"), { type: "object", required: ["uploadedResource"], properties: { uploadedResource: { type: "string" } } }]
|
||||
},
|
||||
IngestionStatus: {
|
||||
type: "object",
|
||||
required: ["sourceId", "versionId", "state", "phase", "activated", "documents", "error", "statusUrl", "reviewUrl"],
|
||||
properties: {
|
||||
sourceId: { type: "string" }, versionId: { type: "string", format: "uuid" }, state: ref("SourceVersionState"), phase: ref("OcrPhase"), activated: { type: "boolean" },
|
||||
documents: { type: "array", items: ref("OcrProgressDocument") },
|
||||
error: { oneOf: [ref("OcrStatusError"), { type: "null" }] }, statusUrl: { type: "string" }, reviewUrl: { type: ["string", "null"] }
|
||||
}
|
||||
},
|
||||
OcrProgressDocument: {
|
||||
type: "object", required: ["documentId", "state", "completedPages", "totalPages", "pages"],
|
||||
properties: {
|
||||
documentId: { type: "string" }, state: { type: "string", enum: ["native_complete", "ocr_queued", "ocr_running", "ocr_complete", "failed"] },
|
||||
completedPages: { type: "integer" }, totalPages: { type: "integer" }, pages: { type: "array", items: ref("OcrProgressPage") }
|
||||
}
|
||||
},
|
||||
OcrProgressPage: {
|
||||
type: "object", required: ["page", "method", "state"],
|
||||
properties: { page: { type: "integer" }, method: { type: "string", enum: ["native", "ocr", "blank"] }, state: { type: "string", enum: ["pending", "native_complete", "ocr_queued", "ocr_running", "ocr_complete", "blank", "failed"] }, errorCode: { type: "string" } }
|
||||
},
|
||||
OcrStatusError: { type: "object", required: ["code", "message", "retryable"], properties: { code: { type: "string" }, message: { type: "string" }, retryable: { type: "boolean" } } },
|
||||
OcrLine: {
|
||||
type: "object", required: ["lineId", "text", "confidence", "bbox", "lineSha256"],
|
||||
properties: { lineId: { type: "string" }, text: { type: "string" }, confidence: { type: "number" }, bbox: { type: "array", items: { type: "number" }, minItems: 4, maxItems: 4 }, lineSha256: { type: "string", pattern: "^[a-f0-9]{64}$" } }
|
||||
},
|
||||
OcrReview: {
|
||||
type: "object",
|
||||
required: ["versionId", "sourceId", "state", "candidateSha256", "baseActiveVersionId", "currentActiveVersionId", "activateRequested", "processingFingerprint", "metadataHash", "documents"],
|
||||
properties: {
|
||||
versionId: { type: "string", format: "uuid" }, sourceId: { type: "string" }, state: { type: "string", const: "review_required" }, candidateSha256: { type: "string", pattern: "^[a-f0-9]{64}$" },
|
||||
baseActiveVersionId: { type: ["string", "null"], format: "uuid" }, currentActiveVersionId: { type: ["string", "null"], format: "uuid" }, activateRequested: { type: "boolean" }, processingFingerprint: { type: "string" }, metadataHash: { type: "string" },
|
||||
documents: { type: "array", items: ref("OcrReviewDocument") }
|
||||
}
|
||||
},
|
||||
OcrReviewDocument: { type: "object", required: ["documentId", "pages"], properties: { documentId: { type: "string" }, pages: { type: "array", items: ref("OcrReviewPage") } } },
|
||||
OcrReviewPage: {
|
||||
type: "object", required: ["page", "imageUrl", "nativeText", "ocr", "candidateText", "differences", "risks"],
|
||||
properties: { page: { type: "integer" }, imageUrl: { type: "string" }, nativeText: { type: "string" }, ocr: { type: "object", required: ["text", "lines"], properties: { text: { type: "string" }, lines: { type: "array", items: ref("OcrLine") } } }, candidateText: { type: "string" }, differences: { type: "array", items: { type: "string" } }, risks: { type: "array", items: { type: "string" } } }
|
||||
},
|
||||
OcrCorrection: {
|
||||
type: "object", required: ["documentId", "page", "lineId", "expectedLineSha256", "replacementText"],
|
||||
properties: { documentId: { type: "string" }, page: { type: "integer" }, lineId: { type: "string" }, expectedLineSha256: { type: "string", pattern: "^[a-f0-9]{64}$" }, replacementText: { type: "string" } }
|
||||
},
|
||||
OcrApprovalRequest: {
|
||||
type: "object", required: ["candidateSha256", "expectedActiveVersionId", "reviewedBy", "corrections"],
|
||||
properties: { candidateSha256: { type: "string", pattern: "^[a-f0-9]{64}$" }, expectedActiveVersionId: { type: ["string", "null"], format: "uuid" }, reviewedBy: { type: "string" }, corrections: { type: "array", items: ref("OcrCorrection") } }
|
||||
},
|
||||
OcrRejectionRequest: {
|
||||
type: "object", required: ["candidateSha256", "reviewedBy", "reason"],
|
||||
properties: { candidateSha256: { type: "string", pattern: "^[a-f0-9]{64}$" }, reviewedBy: { type: "string" }, reason: { type: "string" } }
|
||||
},
|
||||
OcrDecisionResponse: {
|
||||
type: "object", required: ["versionId", "state", "activated"],
|
||||
properties: { versionId: { type: "string", format: "uuid" }, state: { type: "string", enum: ["ready", "active", "rejected"] }, activated: { type: "boolean" }, activatedVersionId: { type: "string", format: "uuid" }, errorCode: { type: "string", enum: ["DUPLICATE_REUSABLE_VERSION"] } }
|
||||
},
|
||||
CleanupRequest: {
|
||||
type: "object",
|
||||
required: ["scope"],
|
||||
|
|
@ -877,7 +1004,7 @@ export function buildApiHelp() {
|
|||
openapiPurpose: "Consult this contract for complete parameters, request bodies, responses, errors, and examples for every endpoint.",
|
||||
playground: "/playground"
|
||||
},
|
||||
authentication: "None. The current API is publicly accessible; authentication is planned separately.",
|
||||
authentication: "Bearer authentication is required for lifecycle administration and all OCR candidate routes; other routes remain public.",
|
||||
endpoints
|
||||
};
|
||||
}
|
||||
|
|
|
|||
136
src/app.ts
136
src/app.ts
|
|
@ -1,10 +1,10 @@
|
|||
import express from "express";
|
||||
import multer from "multer";
|
||||
import type { Request } from "express";
|
||||
import { writeFile, unlink, mkdtemp, rm } from "node:fs/promises";
|
||||
import { readFile, unlink, mkdtemp, rm } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import { timingSafeEqual } from "node:crypto";
|
||||
import { randomUUID, timingSafeEqual } from "node:crypto";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import AdmZip from "adm-zip";
|
||||
import { buildApiHelp, openApiDocument } from "./api/openapi.js";
|
||||
|
|
@ -20,28 +20,66 @@ import { QdrantVectorStoreClient } from "./modules/vectorstore/client.js";
|
|||
import { getCatalogPool } from "./modules/catalog/client.js";
|
||||
import { KnowledgeLifecycleReconciler } from "./modules/catalog/reconciler.js";
|
||||
import { CatalogError, CatalogRepository } from "./modules/catalog/repository.js";
|
||||
import { OcrClient } from "./modules/ocr/client.js";
|
||||
import { OcrDispatcher } from "./modules/ocr/dispatcher.js";
|
||||
import { resolveArtifactPath } from "./modules/ocr/artifacts.js";
|
||||
import type { OcrReviewService } from "./modules/ocr/review.js";
|
||||
import type { OcrIndexingService } from "./modules/ocr/indexing.js";
|
||||
import { OcrRetentionService } from "./modules/ocr/retention.js";
|
||||
import type { ChatMessage, ChunkMode, RetrieveIntent, RetrieveScope } from "./shared/types/rag.js";
|
||||
import { sha256Hex } from "./shared/utils/ids.js";
|
||||
|
||||
type UploadRequest = Request & {
|
||||
file?: Express.Multer.File;
|
||||
};
|
||||
|
||||
export function createApp() {
|
||||
interface AppOptions {
|
||||
ingestService?: Pick<IngestService, "ingest" | "cleanup">;
|
||||
catalog?: CatalogRepository;
|
||||
ocrClient?: OcrClient;
|
||||
reviewService?: Pick<OcrReviewService, "view" | "approve" | "reject">;
|
||||
indexingService?: Pick<OcrIndexingService, "index">;
|
||||
startReconciler?: boolean;
|
||||
}
|
||||
|
||||
export function createApp(options: AppOptions = {}) {
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = path.dirname(__filename);
|
||||
const publicDir = path.resolve(__dirname, "../public");
|
||||
const app = express();
|
||||
const upload = multer({ storage: multer.memoryStorage() });
|
||||
const upload = multer({
|
||||
storage: multer.diskStorage({
|
||||
destination: os.tmpdir(),
|
||||
filename: (_request, file, callback) => callback(null, `${randomUUID()}${path.extname(file.originalname).toLowerCase()}`)
|
||||
}),
|
||||
limits: { fileSize: 50 * 1024 * 1024 }
|
||||
});
|
||||
const embeddingProvider = new OpenRouterEmbeddingProvider();
|
||||
const vectorStore = new QdrantVectorStoreClient();
|
||||
const catalogPool = getCatalogPool();
|
||||
const catalog = catalogPool ? new CatalogRepository(catalogPool) : undefined;
|
||||
const reconciler = new KnowledgeLifecycleReconciler(catalog, vectorStore);
|
||||
const catalog = Object.prototype.hasOwnProperty.call(options, "catalog")
|
||||
? options.catalog
|
||||
: catalogPool ? new CatalogRepository(catalogPool) : undefined;
|
||||
const ocr = { enabled: env.ocrIngestEnabled, artifactRoot: path.resolve(env.ocrArtifactRoot) };
|
||||
const ocrClient = ocr.enabled ? options.ocrClient ?? new OcrClient({ baseUrl: env.ocrServiceUrl, token: env.ocrInternalToken }) : undefined;
|
||||
const ocrDispatcher = catalog && ocrClient ? new OcrDispatcher(catalog, ocrClient, async (job) => {
|
||||
const versionDirectory = path.join(ocr.artifactRoot, job.versionId);
|
||||
const manifest = JSON.parse(await readFile(path.join(versionDirectory, "manifest.json"), "utf8")) as {
|
||||
documents?: Array<{ documentId: string; originalPath: string; originalSha256: string }>;
|
||||
};
|
||||
const document = manifest.documents?.find(({ documentId }) => documentId === job.documentId);
|
||||
if (!document) throw new Error("OCR artifact manifest does not contain the queued document");
|
||||
const bytes = await readFile(resolveArtifactPath(versionDirectory, document.originalPath));
|
||||
if (sha256Hex(bytes) !== document.originalSha256) throw new Error("OCR artifact integrity validation failed");
|
||||
return { bytes, documentSha256: document.originalSha256 };
|
||||
}) : undefined;
|
||||
const retention = catalog ? new OcrRetentionService(catalog, ocr.artifactRoot) : undefined;
|
||||
const reconciler = new KnowledgeLifecycleReconciler(catalog, vectorStore, ocrDispatcher, retention);
|
||||
const evaluationLogs = new EvaluationLogService(embeddingProvider);
|
||||
const ingestService = new IngestService(embeddingProvider, vectorStore, catalog);
|
||||
const ingestService = options.ingestService ?? new IngestService(embeddingProvider, vectorStore, catalog, ocr);
|
||||
const retrieveService = new RetrieveService(embeddingProvider, vectorStore, catalog);
|
||||
const answerService = new AnswerService(retrieveService);
|
||||
reconciler.start();
|
||||
if (options.startReconciler !== false) reconciler.start();
|
||||
|
||||
function sendError(res: express.Response, error: unknown, fallback: string) {
|
||||
const upstreamUnavailable = error instanceof Error && /(ECONNREFUSED|ETIMEDOUT|timeout|connection|database|postgres|qdrant)/i.test(error.message);
|
||||
|
|
@ -54,6 +92,10 @@ export function createApp() {
|
|||
res.status(statusCode).json(body);
|
||||
}
|
||||
|
||||
function dispatchAcceptedOcr(result: Awaited<ReturnType<IngestService["ingest"]>>): void {
|
||||
if ("phase" in result) void ocrDispatcher?.dispatchAvailable();
|
||||
}
|
||||
|
||||
function requireLifecycleAdmin(req: express.Request, res: express.Response): boolean {
|
||||
const header = req.header("authorization") ?? "";
|
||||
const token = header.startsWith("Bearer ") ? header.slice("Bearer ".length) : "";
|
||||
|
|
@ -71,6 +113,12 @@ export function createApp() {
|
|||
return true;
|
||||
}
|
||||
|
||||
function requireOcrEnabled(res: express.Response): boolean {
|
||||
if (ocr.enabled) return true;
|
||||
res.status(404).json({ ok: false, error: "Not found" });
|
||||
return false;
|
||||
}
|
||||
|
||||
function requireCatalog(res: express.Response): CatalogRepository | undefined {
|
||||
if (!catalog) {
|
||||
res.status(503).json({ ok: false, error: "Knowledge catalog is not configured", code: "CATALOG_UNAVAILABLE" });
|
||||
|
|
@ -201,6 +249,66 @@ export function createApp() {
|
|||
}
|
||||
});
|
||||
|
||||
app.get("/ingestions/:versionId", async (req, res) => {
|
||||
if (!requireOcrEnabled(res)) return;
|
||||
if (!requireLifecycleAdmin(req, res)) return;
|
||||
const repository = requireCatalog(res);
|
||||
if (!repository) return;
|
||||
try {
|
||||
const status = await repository.getIngestionStatus(String(req.params.versionId));
|
||||
if (!status) {
|
||||
res.status(404).json({ ok: false, error: "Ingestion not found", code: "INGESTION_NOT_FOUND" });
|
||||
return;
|
||||
}
|
||||
res.json(status);
|
||||
} catch (error) {
|
||||
sendError(res, error, "Unknown ingestion status error");
|
||||
}
|
||||
});
|
||||
|
||||
app.get("/ingestions/:versionId/review", async (req, res) => {
|
||||
if (!requireOcrEnabled(res)) return;
|
||||
if (!requireLifecycleAdmin(req, res)) return;
|
||||
if (!options.reviewService) {
|
||||
res.status(503).json({ ok: false, error: "OCR review service is not configured", code: "OCR_REVIEW_UNAVAILABLE" });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
res.json(await options.reviewService.view(String(req.params.versionId)));
|
||||
} catch (error) {
|
||||
sendError(res, error, "Unknown OCR review error");
|
||||
}
|
||||
});
|
||||
|
||||
app.post("/ingestions/:versionId/approve", async (req, res) => {
|
||||
if (!requireOcrEnabled(res)) return;
|
||||
if (!requireLifecycleAdmin(req, res)) return;
|
||||
if (!options.reviewService || !options.indexingService) {
|
||||
res.status(503).json({ ok: false, error: "OCR approval service is not configured", code: "OCR_REVIEW_UNAVAILABLE" });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const approved = await options.reviewService.approve(String(req.params.versionId), req.body);
|
||||
res.json(await options.indexingService.index(approved));
|
||||
} catch (error) {
|
||||
sendError(res, error, "Unknown OCR approval error");
|
||||
}
|
||||
});
|
||||
|
||||
app.post("/ingestions/:versionId/reject", async (req, res) => {
|
||||
if (!requireOcrEnabled(res)) return;
|
||||
if (!requireLifecycleAdmin(req, res)) return;
|
||||
if (!options.reviewService) {
|
||||
res.status(503).json({ ok: false, error: "OCR review service is not configured", code: "OCR_REVIEW_UNAVAILABLE" });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
res.json(await options.reviewService.reject(String(req.params.versionId), req.body));
|
||||
} catch (error) {
|
||||
sendError(res, error, "Unknown OCR rejection error");
|
||||
}
|
||||
});
|
||||
|
||||
app.get("/models/answer", async (_req, res) => {
|
||||
try {
|
||||
const models = await answerService.listAvailableAnswerModels();
|
||||
|
|
@ -226,7 +334,8 @@ export function createApp() {
|
|||
app.post("/ingest", async (req, res) => {
|
||||
try {
|
||||
const result = await ingestService.ingest(req.body);
|
||||
res.status(env.knowledgeLifecycleEnforced ? 201 : 202).json(result);
|
||||
dispatchAcceptedOcr(result);
|
||||
res.status("phase" in result ? 202 : env.knowledgeLifecycleEnforced ? 201 : 202).json(result);
|
||||
} catch (error) {
|
||||
sendError(res, error, "Unknown ingest error");
|
||||
}
|
||||
|
|
@ -322,7 +431,8 @@ export function createApp() {
|
|||
activate: req.body.activate === undefined ? true : Boolean(req.body.activate),
|
||||
expectedActiveVersionId: req.body.expectedActiveVersionId === undefined ? null : req.body.expectedActiveVersionId
|
||||
});
|
||||
res.status(201).json(result);
|
||||
dispatchAcceptedOcr(result);
|
||||
res.status("phase" in result ? 202 : 201).json(result);
|
||||
return;
|
||||
}
|
||||
res.status(202).json({
|
||||
|
|
@ -371,8 +481,7 @@ export function createApp() {
|
|||
|
||||
const isZipFolder = req.body.isZipFolder === "true";
|
||||
const tempDirBase = await os.tmpdir();
|
||||
tempFilePath = path.join(tempDirBase, `${Date.now()}-${req.file.originalname}`);
|
||||
await writeFile(tempFilePath, req.file.buffer);
|
||||
tempFilePath = req.file.path;
|
||||
|
||||
const tags = typeof req.body.tags === "string"
|
||||
? req.body.tags.split(",").map((entry: string) => entry.trim()).filter(Boolean)
|
||||
|
|
@ -408,7 +517,8 @@ export function createApp() {
|
|||
});
|
||||
}
|
||||
|
||||
res.status(env.knowledgeLifecycleEnforced ? 201 : 202).json({
|
||||
dispatchAcceptedOcr(result);
|
||||
res.status("phase" in result ? 202 : env.knowledgeLifecycleEnforced ? 201 : 202).json({
|
||||
...result,
|
||||
uploadedResource: req.file.originalname
|
||||
});
|
||||
|
|
|
|||
|
|
@ -52,5 +52,13 @@ export const env = {
|
|||
knowledgeLifecycleEnforced: booleanEnv("KNOWLEDGE_LIFECYCLE_ENFORCED", false),
|
||||
ingestWritesEnabled: booleanEnv("INGEST_WRITES_ENABLED", true),
|
||||
lifecycleReconcileIntervalMs: Number(process.env.LIFECYCLE_RECONCILE_INTERVAL_MS ?? 300000),
|
||||
lifecycleIndexingStaleTimeoutMs: Number(process.env.LIFECYCLE_INDEXING_STALE_TIMEOUT_MS ?? 1800000)
|
||||
lifecycleIndexingStaleTimeoutMs: Number(process.env.LIFECYCLE_INDEXING_STALE_TIMEOUT_MS ?? 1800000),
|
||||
ocrIngestEnabled: booleanEnv("OCR_INGEST_ENABLED", false),
|
||||
ocrServiceUrl: process.env.OCR_SERVICE_URL ?? "http://ocr-service:8000",
|
||||
ocrInternalToken: process.env.OCR_INTERNAL_TOKEN ?? "",
|
||||
ocrArtifactRoot: process.env.OCR_ARTIFACT_ROOT ?? "/data/ingestions",
|
||||
ocrMaxUploadBytes: Number(process.env.OCR_MAX_UPLOAD_BYTES ?? 50 * 1024 * 1024),
|
||||
ocrMaxPages: Number(process.env.OCR_MAX_PAGES ?? 100),
|
||||
ocrPageTimeoutMs: Number(process.env.OCR_PAGE_TIMEOUT_MS ?? 60_000),
|
||||
ocrTotalTimeoutMs: Number(process.env.OCR_TOTAL_TIMEOUT_MS ?? 15 * 60_000)
|
||||
} as const;
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
import { env } from "../../config/env.js";
|
||||
import type { VectorStoreClient } from "../vectorstore/client.js";
|
||||
import type { CatalogRepository } from "./repository.js";
|
||||
import type { OcrDispatcher } from "../ocr/dispatcher.js";
|
||||
import type { OcrRetentionService } from "../ocr/retention.js";
|
||||
|
||||
export interface ReconcilerStatus {
|
||||
ok: boolean;
|
||||
|
|
@ -9,6 +11,8 @@ export interface ReconcilerStatus {
|
|||
inconsistentSources: string[];
|
||||
orphanedVersionsRecovered?: number;
|
||||
orphanedVersionsFailed?: number;
|
||||
ocrLeasesRecovered?: number;
|
||||
ocrRetentionDeleted?: number;
|
||||
invariantViolations?: string[];
|
||||
}
|
||||
|
||||
|
|
@ -18,7 +22,9 @@ export class KnowledgeLifecycleReconciler {
|
|||
|
||||
constructor(
|
||||
private readonly catalog: CatalogRepository | undefined,
|
||||
private readonly vectorStore: VectorStoreClient
|
||||
private readonly vectorStore: VectorStoreClient,
|
||||
private readonly ocrDispatcher?: Pick<OcrDispatcher, "recoverExpiredLeases" | "dispatchAvailable">,
|
||||
private readonly ocrRetention?: Pick<OcrRetentionService, "runOnce">
|
||||
) {}
|
||||
|
||||
getStatus(): ReconcilerStatus {
|
||||
|
|
@ -49,6 +55,9 @@ export class KnowledgeLifecycleReconciler {
|
|||
const activeVersions = await catalog.resolveActiveVersions();
|
||||
const inconsistentSources: string[] = [];
|
||||
const invariantViolations = await catalog.validateActiveInvariant();
|
||||
const ocrLeasesRecovered = await this.ocrDispatcher?.recoverExpiredLeases() ?? 0;
|
||||
await this.ocrDispatcher?.dispatchAvailable();
|
||||
const retention = await this.ocrRetention?.runOnce();
|
||||
const orphanRecovery = await this.recoverOrphanedVersions(catalog);
|
||||
|
||||
for (const version of activeVersions) {
|
||||
|
|
@ -71,6 +80,8 @@ export class KnowledgeLifecycleReconciler {
|
|||
inconsistentSources,
|
||||
orphanedVersionsRecovered: orphanRecovery.ready,
|
||||
orphanedVersionsFailed: orphanRecovery.failed,
|
||||
ocrLeasesRecovered,
|
||||
ocrRetentionDeleted: (retention?.deleted ?? 0) + (retention?.resumed ?? 0),
|
||||
invariantViolations
|
||||
};
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,7 +1,10 @@
|
|||
import type { AvailableScope, ChunkMode, IngestSourceInput, SourceType, SourceVersionState, RetrieveScope } from "../../shared/types/rag.js";
|
||||
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||
import { withCatalogAdvisoryLock, withCatalogAdvisorySharedLocks, withCatalogTryAdvisoryLock, withTransaction, type PgPool, type PgPoolClient } from "./client.js";
|
||||
import { CatalogError } from "./errors.js";
|
||||
import { normalizeExpectedActiveVersion } from "./lifecycle.js";
|
||||
import type { OcrResult } from "../ocr/client.js";
|
||||
import type { OcrRetentionCandidate } from "../ocr/retention.js";
|
||||
|
||||
export interface CatalogDocumentInput {
|
||||
documentId: string;
|
||||
|
|
@ -12,9 +15,26 @@ export interface CatalogDocumentInput {
|
|||
mimeType: string;
|
||||
title: string;
|
||||
chunkCount: number;
|
||||
extractionMethod?: "native" | "ocr";
|
||||
artifactManifestPath?: string;
|
||||
}
|
||||
|
||||
export interface OcrPageInput {
|
||||
documentId: string;
|
||||
page: number;
|
||||
extractionMethod: "native" | "ocr";
|
||||
nativeTextHash: string;
|
||||
}
|
||||
|
||||
export interface OcrJobInput {
|
||||
documentId: string;
|
||||
remoteIdempotencyKey: string;
|
||||
requestedPages: number[];
|
||||
configVersion: "ocr-v1";
|
||||
}
|
||||
|
||||
export interface CreateVersionInput {
|
||||
versionId?: string;
|
||||
sourceId: string;
|
||||
sourceType: SourceType;
|
||||
sourceRef: string;
|
||||
|
|
@ -31,6 +51,8 @@ export interface CreateVersionInput {
|
|||
expectedDocumentCount: number;
|
||||
expectedPointCount: number;
|
||||
documents: CatalogDocumentInput[];
|
||||
ocrPages?: OcrPageInput[];
|
||||
ocrJobs?: OcrJobInput[];
|
||||
}
|
||||
|
||||
export interface CatalogVersionRow {
|
||||
|
|
@ -70,6 +92,36 @@ export interface OrphanedVersionCandidate {
|
|||
qdrantCollection: string;
|
||||
}
|
||||
|
||||
export interface PendingOcrIdentity {
|
||||
sourceId: string;
|
||||
originalManifestHash: string;
|
||||
processingFingerprint: string;
|
||||
metadataHash: string;
|
||||
}
|
||||
|
||||
export interface OcrJobRow {
|
||||
jobId: string;
|
||||
versionId: string;
|
||||
documentId: string;
|
||||
remoteJobId: string | null;
|
||||
remoteIdempotencyKey: string;
|
||||
state: "queued" | "running" | "succeeded" | "failed";
|
||||
requestedPages: number[];
|
||||
completedPages: number;
|
||||
configVersion: string;
|
||||
attemptCount: number;
|
||||
heartbeatAt: Date | null;
|
||||
leaseExpiresAt: Date | null;
|
||||
nextAttemptAt: Date | null;
|
||||
errorCode: string | null;
|
||||
errorDetail: string | null;
|
||||
}
|
||||
|
||||
export interface PendingOcrVersion {
|
||||
version: CatalogVersionRow;
|
||||
jobs: OcrJobRow[];
|
||||
}
|
||||
|
||||
type VersionDbRow = {
|
||||
version_id: string;
|
||||
source_id: string;
|
||||
|
|
@ -87,8 +139,41 @@ type VersionDbRow = {
|
|||
expected_point_count: string | number;
|
||||
verified_point_count: string | number;
|
||||
qdrant_collection: string;
|
||||
artifact_state?: "none" | "present" | "retention_deleting" | "retention_deleted";
|
||||
};
|
||||
|
||||
type OcrJobDbRow = {
|
||||
ocr_job_id: string;
|
||||
ocr_version_id: string;
|
||||
ocr_document_id: string;
|
||||
ocr_remote_job_id: string | null;
|
||||
ocr_remote_idempotency_key: string;
|
||||
ocr_state: OcrJobRow["state"];
|
||||
ocr_requested_pages: number[];
|
||||
ocr_completed_pages: string | number;
|
||||
ocr_config_version: string;
|
||||
ocr_attempt_count: string | number;
|
||||
ocr_heartbeat_at: Date | null;
|
||||
ocr_lease_expires_at: Date | null;
|
||||
ocr_next_attempt_at: Date | null;
|
||||
ocr_error_code: string | null;
|
||||
ocr_error_detail: string | null;
|
||||
};
|
||||
|
||||
const OCR_JOB_COLUMNS = [
|
||||
["job_id", "ocr_job_id"], ["version_id", "ocr_version_id"], ["document_id", "ocr_document_id"],
|
||||
["remote_job_id", "ocr_remote_job_id"], ["remote_idempotency_key", "ocr_remote_idempotency_key"],
|
||||
["state", "ocr_state"], ["requested_pages", "ocr_requested_pages"],
|
||||
["completed_pages", "ocr_completed_pages"], ["config_version", "ocr_config_version"],
|
||||
["attempt_count", "ocr_attempt_count"], ["heartbeat_at", "ocr_heartbeat_at"],
|
||||
["lease_expires_at", "ocr_lease_expires_at"], ["next_attempt_at", "ocr_next_attempt_at"],
|
||||
["error_code", "ocr_error_code"], ["error_detail", "ocr_error_detail"]
|
||||
] as const;
|
||||
|
||||
function ocrJobColumns(prefix = ""): string {
|
||||
return OCR_JOB_COLUMNS.map(([column, alias]) => `${prefix}${column} AS ${alias}`).join(", ");
|
||||
}
|
||||
|
||||
function toVersion(row: VersionDbRow): CatalogVersionRow {
|
||||
return {
|
||||
versionId: row.version_id,
|
||||
|
|
@ -110,6 +195,34 @@ function toVersion(row: VersionDbRow): CatalogVersionRow {
|
|||
};
|
||||
}
|
||||
|
||||
function toOcrJob(row: OcrJobDbRow): OcrJobRow {
|
||||
return {
|
||||
jobId: row.ocr_job_id,
|
||||
versionId: row.ocr_version_id,
|
||||
documentId: row.ocr_document_id,
|
||||
remoteJobId: row.ocr_remote_job_id,
|
||||
remoteIdempotencyKey: row.ocr_remote_idempotency_key,
|
||||
state: row.ocr_state,
|
||||
requestedPages: row.ocr_requested_pages,
|
||||
completedPages: Number(row.ocr_completed_pages),
|
||||
configVersion: row.ocr_config_version,
|
||||
attemptCount: Number(row.ocr_attempt_count),
|
||||
heartbeatAt: row.ocr_heartbeat_at,
|
||||
leaseExpiresAt: row.ocr_lease_expires_at,
|
||||
nextAttemptAt: row.ocr_next_attempt_at,
|
||||
errorCode: row.ocr_error_code,
|
||||
errorDetail: row.ocr_error_detail
|
||||
};
|
||||
}
|
||||
|
||||
function ingestionPhase(state: SourceVersionState, jobs: Array<OcrJobRow["state"] | null>): string {
|
||||
if (state !== "indexing") return state;
|
||||
if (jobs.some((job) => job === "failed")) return "failed";
|
||||
if (jobs.some((job) => job === "running")) return "ocr_running";
|
||||
if (jobs.some((job) => job === "queued")) return "ocr_queued";
|
||||
return "indexing";
|
||||
}
|
||||
|
||||
export class CatalogRepository {
|
||||
constructor(private readonly pool: PgPool) {}
|
||||
|
||||
|
|
@ -220,6 +333,243 @@ export class CatalogRepository {
|
|||
return result.rows[0] ? toVersion(result.rows[0]) : undefined;
|
||||
}
|
||||
|
||||
async findPendingOcrVersion(input: PendingOcrIdentity): Promise<PendingOcrVersion | undefined> {
|
||||
const result = await this.pool.query<VersionDbRow & OcrJobDbRow>(
|
||||
`SELECT v.*, ${ocrJobColumns("j.")}
|
||||
FROM rag_source_versions v
|
||||
JOIN rag_version_documents d ON d.version_id = v.version_id
|
||||
JOIN rag_ocr_jobs j ON j.version_id = d.version_id AND j.document_id = d.document_id
|
||||
WHERE v.source_id = $1
|
||||
AND v.original_manifest_hash = $2
|
||||
AND v.processing_fingerprint = $3
|
||||
AND v.metadata_hash = $4
|
||||
AND v.source_content_hash IS NULL
|
||||
AND v.state IN ('pending', 'indexing', 'review_required')
|
||||
ORDER BY j.document_id`,
|
||||
[input.sourceId, input.originalManifestHash, input.processingFingerprint, input.metadataHash]
|
||||
);
|
||||
return result.rows[0] ? { version: toVersion(result.rows[0]), jobs: result.rows.map(toOcrJob) } : undefined;
|
||||
}
|
||||
|
||||
async claimNextOcrJob(leaseMs: number): Promise<OcrJobRow | undefined> {
|
||||
return withTransaction(this.pool, async (client) => {
|
||||
const candidate = await client.query<{ job_id: string }>(
|
||||
`SELECT job_id FROM rag_ocr_jobs
|
||||
WHERE state = 'queued' AND (next_attempt_at IS NULL OR next_attempt_at <= now())
|
||||
ORDER BY created_at FOR UPDATE SKIP LOCKED LIMIT 1`
|
||||
);
|
||||
if (!candidate.rowCount) return undefined;
|
||||
const result = await client.query<OcrJobDbRow>(
|
||||
`UPDATE rag_ocr_jobs
|
||||
SET state = 'running', attempt_count = attempt_count + 1,
|
||||
started_at = COALESCE(started_at, now()), heartbeat_at = now(),
|
||||
lease_expires_at = now() + ($2::text || ' milliseconds')::interval
|
||||
WHERE job_id = $1
|
||||
RETURNING ${ocrJobColumns()}`,
|
||||
[candidate.rows[0].job_id, String(leaseMs)]
|
||||
);
|
||||
return result.rows[0] ? toOcrJob(result.rows[0]) : undefined;
|
||||
});
|
||||
}
|
||||
|
||||
async claimOcrJob(jobId: string, leaseMs: number): Promise<OcrJobRow | undefined> {
|
||||
const result = await this.pool.query<OcrJobDbRow>(
|
||||
`UPDATE rag_ocr_jobs
|
||||
SET state = 'running', attempt_count = attempt_count + 1,
|
||||
started_at = COALESCE(started_at, now()), heartbeat_at = now(),
|
||||
lease_expires_at = now() + ($2::text || ' milliseconds')::interval
|
||||
WHERE job_id = $1 AND state = 'queued' AND (next_attempt_at IS NULL OR next_attempt_at <= now())
|
||||
RETURNING ${ocrJobColumns()}`,
|
||||
[jobId, String(leaseMs)]
|
||||
);
|
||||
return result.rows[0] ? toOcrJob(result.rows[0]) : undefined;
|
||||
}
|
||||
|
||||
async recoverExpiredOcrLeases(): Promise<OcrJobRow[]> {
|
||||
const result = await this.pool.query<OcrJobDbRow>(
|
||||
`UPDATE rag_ocr_jobs
|
||||
SET state = 'queued', heartbeat_at = NULL, lease_expires_at = NULL, next_attempt_at = now()
|
||||
WHERE state = 'running' AND lease_expires_at <= now()
|
||||
RETURNING ${ocrJobColumns()}`
|
||||
);
|
||||
return result.rows.map(toOcrJob);
|
||||
}
|
||||
|
||||
async setOcrRemoteJob(jobId: string, remoteJobId: string, leaseMs: number): Promise<void> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_ocr_jobs SET remote_job_id = $2, heartbeat_at = now(),
|
||||
lease_expires_at = now() + ($3::text || ' milliseconds')::interval
|
||||
WHERE job_id = $1 AND state = 'running'`,
|
||||
[jobId, remoteJobId, String(leaseMs)]
|
||||
);
|
||||
if (result.rowCount !== 1) throw new CatalogError("OCR job lease was lost", 409, "OCR_LEASE_LOST");
|
||||
}
|
||||
|
||||
async requeueOcrJob(jobId: string, code: string, detail: string, delayMs: number): Promise<void> {
|
||||
await this.pool.query(
|
||||
`UPDATE rag_ocr_jobs SET state = 'queued', heartbeat_at = NULL, lease_expires_at = NULL,
|
||||
next_attempt_at = now() + ($4::text || ' milliseconds')::interval, error_code = $2, error_detail = $3
|
||||
WHERE job_id = $1 AND state = 'running'`,
|
||||
[jobId, code, detail.slice(0, 2000), String(delayMs)]
|
||||
);
|
||||
}
|
||||
|
||||
async completeOcrJob(jobId: string, result: OcrResult): Promise<boolean> {
|
||||
return withTransaction(this.pool, async (client) => {
|
||||
const identity = await client.query<{ version_id: string; document_id: string }>(
|
||||
"SELECT version_id, document_id FROM rag_ocr_jobs WHERE job_id = $1 AND state = 'running' FOR UPDATE",
|
||||
[jobId]
|
||||
);
|
||||
if (!identity.rowCount) throw new CatalogError("OCR job lease was lost", 409, "OCR_LEASE_LOST");
|
||||
const { version_id: versionId, document_id: documentId } = identity.rows[0];
|
||||
for (const page of result.pages) {
|
||||
await client.query(
|
||||
`UPDATE rag_document_pages SET ocr_text_hash = $4, candidate_text_hash = $4, metrics = $5
|
||||
WHERE version_id = $1 AND document_id = $2 AND page_number = $3 AND extraction_method = 'ocr'`,
|
||||
[versionId, documentId, page.page, sha256Hex(page.text), page.metrics]
|
||||
);
|
||||
}
|
||||
await client.query(
|
||||
`UPDATE rag_ocr_jobs SET state = 'succeeded', completed_pages = cardinality(requested_pages),
|
||||
heartbeat_at = now(), lease_expires_at = NULL, next_attempt_at = NULL,
|
||||
error_code = NULL, error_detail = NULL, completed_at = now()
|
||||
WHERE job_id = $1`,
|
||||
[jobId]
|
||||
);
|
||||
const pending = await client.query(
|
||||
"SELECT 1 FROM rag_ocr_jobs WHERE version_id = $1 AND state <> 'succeeded' LIMIT 1",
|
||||
[versionId]
|
||||
);
|
||||
return pending.rowCount === 0;
|
||||
});
|
||||
}
|
||||
|
||||
async failOcrJob(jobId: string, code: string, detail: string): Promise<void> {
|
||||
await this.pool.query(
|
||||
`UPDATE rag_ocr_jobs SET state = 'failed', lease_expires_at = NULL, next_attempt_at = NULL,
|
||||
error_code = $2, error_detail = $3, completed_at = now() WHERE job_id = $1 AND state = 'running'`,
|
||||
[jobId, code, detail.slice(0, 2000)]
|
||||
);
|
||||
}
|
||||
|
||||
async getIngestionStatus(versionId: string): Promise<Record<string, unknown> | undefined> {
|
||||
const version = await this.pool.query<{
|
||||
source_id: string;
|
||||
state: SourceVersionState;
|
||||
error_code: string | null;
|
||||
error_detail: string | null;
|
||||
}>("SELECT source_id, state, error_code, error_detail FROM rag_source_versions WHERE version_id = $1", [versionId]);
|
||||
if (!version.rowCount) return undefined;
|
||||
const documents = await this.pool.query<{
|
||||
document_id: string;
|
||||
index_state: "pending" | "indexing" | "ready" | "failed";
|
||||
job_state: OcrJobRow["state"] | null;
|
||||
completed_pages: string | number | null;
|
||||
requested_pages: number[] | null;
|
||||
}>(
|
||||
`SELECT d.document_id, d.index_state, j.state AS job_state, j.completed_pages, j.requested_pages
|
||||
FROM rag_version_documents d LEFT JOIN rag_ocr_jobs j
|
||||
ON j.version_id = d.version_id AND j.document_id = d.document_id
|
||||
WHERE d.version_id = $1 ORDER BY d.document_id`,
|
||||
[versionId]
|
||||
);
|
||||
const pages = await this.pool.query<{
|
||||
document_id: string;
|
||||
page_number: string | number;
|
||||
extraction_method: "native" | "ocr" | "blank";
|
||||
blocked_reason: string | null;
|
||||
}>(
|
||||
"SELECT document_id, page_number, extraction_method, blocked_reason FROM rag_document_pages WHERE version_id = $1 ORDER BY document_id, page_number",
|
||||
[versionId]
|
||||
);
|
||||
const row = version.rows[0];
|
||||
const phase = ingestionPhase(row.state, documents.rows.map(({ job_state }) => job_state));
|
||||
return {
|
||||
sourceId: row.source_id,
|
||||
versionId,
|
||||
state: row.state,
|
||||
phase,
|
||||
activated: row.state === "active",
|
||||
documents: documents.rows.map((document) => {
|
||||
const documentPages = pages.rows.filter((page) => page.document_id === document.document_id);
|
||||
const state = document.job_state === "failed" || document.index_state === "failed"
|
||||
? "failed"
|
||||
: document.job_state === "succeeded"
|
||||
? "ocr_complete"
|
||||
: document.job_state === "running"
|
||||
? "ocr_running"
|
||||
: document.job_state === "queued"
|
||||
? "ocr_queued"
|
||||
: "native_complete";
|
||||
return {
|
||||
documentId: document.document_id,
|
||||
state,
|
||||
completedPages: documentPages.filter((page) => page.extraction_method !== "ocr" && !page.blocked_reason).length
|
||||
+ Number(document.completed_pages ?? 0),
|
||||
totalPages: documentPages.length,
|
||||
pages: documentPages.map((page) => ({
|
||||
page: Number(page.page_number),
|
||||
method: page.extraction_method,
|
||||
state: page.blocked_reason
|
||||
? "failed"
|
||||
: page.extraction_method === "native"
|
||||
? "native_complete"
|
||||
: page.extraction_method === "blank"
|
||||
? "blank"
|
||||
: document.job_state === "succeeded"
|
||||
? "ocr_complete"
|
||||
: document.job_state === "running"
|
||||
? "ocr_running"
|
||||
: "ocr_queued",
|
||||
...(page.blocked_reason ? { errorCode: page.blocked_reason } : {})
|
||||
}))
|
||||
};
|
||||
}),
|
||||
error: row.error_code ? { code: row.error_code, message: row.error_detail ?? row.error_code, retryable: false } : null,
|
||||
statusUrl: `/ingestions/${versionId}`,
|
||||
reviewUrl: row.state === "review_required" ? `/ingestions/${versionId}/review` : null
|
||||
};
|
||||
}
|
||||
|
||||
async markReviewRequired(versionId: string): Promise<void> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions SET state = 'review_required'
|
||||
WHERE version_id = $1 AND state = 'indexing'
|
||||
AND EXISTS (SELECT 1 FROM rag_ocr_jobs WHERE version_id = $1)
|
||||
AND NOT EXISTS (SELECT 1 FROM rag_ocr_jobs WHERE version_id = $1 AND state <> 'succeeded')
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM rag_version_documents d
|
||||
WHERE d.version_id = $1 AND d.content_hash IS NULL
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM rag_ocr_jobs j
|
||||
WHERE j.version_id = d.version_id AND j.document_id = d.document_id AND j.state = 'succeeded'
|
||||
)
|
||||
)`,
|
||||
[versionId]
|
||||
);
|
||||
if (result.rowCount !== 1) throw new CatalogError("Version is not ready for OCR review", 409, "INVALID_VERSION_STATE");
|
||||
}
|
||||
|
||||
async markOcrReviewIndexing(versionId: string, reviewedBy: string): Promise<void> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions SET state = 'indexing', reviewed_at = now(), reviewed_by = $2, indexing_started_at = now()
|
||||
WHERE version_id = $1 AND state = 'review_required'`,
|
||||
[versionId, reviewedBy]
|
||||
);
|
||||
if (result.rowCount !== 1) throw new CatalogError("Version is not awaiting OCR review", 409, "INVALID_VERSION_STATE");
|
||||
}
|
||||
|
||||
async rejectOcrVersion(versionId: string, reviewedBy: string, reason: string): Promise<void> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions
|
||||
SET state = 'rejected', reviewed_at = now(), reviewed_by = $2, error_code = 'OCR_REJECTED', error_detail = $3,
|
||||
retention_due_at = now() + interval '7 days'
|
||||
WHERE version_id = $1 AND state = 'review_required'`,
|
||||
[versionId, reviewedBy, reason.slice(0, 2000)]
|
||||
);
|
||||
if (result.rowCount !== 1) throw new CatalogError("Version is not awaiting OCR review", 409, "INVALID_VERSION_STATE");
|
||||
}
|
||||
|
||||
async createPendingVersion(input: CreateVersionInput): Promise<CatalogVersionRow> {
|
||||
return withTransaction(this.pool, async (client) => {
|
||||
if (await this.isMaintenanceEnabled(client, true)) {
|
||||
|
|
@ -248,17 +598,19 @@ export class CatalogRepository {
|
|||
[input.sourceId]
|
||||
);
|
||||
const versionNumber = Number(versionNumberResult.rows[0].next_version_number);
|
||||
const versionResult = await client.query<VersionDbRow>(
|
||||
`INSERT INTO rag_source_versions(
|
||||
source_id, version_number, previous_version_id, state, original_manifest_hash,
|
||||
source_content_hash, processing_fingerprint, metadata_hash, tags, activate_requested,
|
||||
base_active_version_id, embedding_provider, embedding_model, embedding_dimensions,
|
||||
qdrant_collection, expected_document_count, expected_point_count
|
||||
) VALUES ($1, $2, $3, 'pending', $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16)
|
||||
RETURNING *`,
|
||||
[
|
||||
input.sourceId,
|
||||
versionNumber,
|
||||
const versionResult = await client.query<VersionDbRow>(
|
||||
`INSERT INTO rag_source_versions(
|
||||
version_id, source_id, version_number, previous_version_id, state, original_manifest_hash,
|
||||
source_content_hash, processing_fingerprint, metadata_hash, tags, activate_requested,
|
||||
base_active_version_id, embedding_provider, embedding_model, embedding_dimensions,
|
||||
qdrant_collection, expected_document_count, expected_point_count, artifact_state
|
||||
) VALUES (COALESCE($1::uuid, gen_random_uuid()), $2, $3, $4, 'pending', $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17,
|
||||
CASE WHEN $6::char(64) IS NULL THEN 'present' ELSE 'none' END)
|
||||
RETURNING *`,
|
||||
[
|
||||
input.versionId ?? null,
|
||||
input.sourceId,
|
||||
versionNumber,
|
||||
baseActiveVersionId,
|
||||
input.originalManifestHash,
|
||||
input.sourceContentHash,
|
||||
|
|
@ -270,34 +622,51 @@ export class CatalogRepository {
|
|||
input.embeddingProvider,
|
||||
input.embeddingModel,
|
||||
input.embeddingDimensions,
|
||||
input.qdrantCollection,
|
||||
input.expectedDocumentCount,
|
||||
input.expectedPointCount
|
||||
input.qdrantCollection,
|
||||
input.expectedDocumentCount,
|
||||
input.expectedPointCount
|
||||
]
|
||||
);
|
||||
|
||||
const version = toVersion(versionResult.rows[0]);
|
||||
for (const document of input.documents) {
|
||||
await client.query(
|
||||
`INSERT INTO rag_version_documents(
|
||||
version_id, document_id, document_key, original_hash, original_hash_kind,
|
||||
content_hash, mime_type, title, index_state, chunk_count
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, 'pending', $9)`,
|
||||
[
|
||||
await client.query(
|
||||
`INSERT INTO rag_version_documents(
|
||||
version_id, document_id, document_key, original_hash, original_hash_kind,
|
||||
content_hash, mime_type, title, extraction_method, index_state, chunk_count, artifact_manifest_path
|
||||
) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, 'pending', $10, $11)`,
|
||||
[
|
||||
version.versionId,
|
||||
document.documentId,
|
||||
document.documentKey,
|
||||
document.originalHash,
|
||||
document.originalHashKind,
|
||||
document.contentHash,
|
||||
document.mimeType,
|
||||
document.title,
|
||||
document.chunkCount
|
||||
]
|
||||
document.mimeType,
|
||||
document.title,
|
||||
document.extractionMethod ?? "native",
|
||||
document.chunkCount,
|
||||
document.artifactManifestPath ?? null
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
for (const page of input.ocrPages ?? []) {
|
||||
await client.query(
|
||||
`INSERT INTO rag_document_pages(version_id, document_id, page_number, extraction_method, native_text_hash)
|
||||
VALUES ($1, $2, $3, $4, $5)`,
|
||||
[version.versionId, page.documentId, page.page, page.extractionMethod, page.nativeTextHash]
|
||||
);
|
||||
}
|
||||
for (const job of input.ocrJobs ?? []) {
|
||||
await client.query(
|
||||
`INSERT INTO rag_ocr_jobs(version_id, document_id, remote_idempotency_key, state, requested_pages, config_version)
|
||||
VALUES ($1, $2, $3, 'queued', $4, $5)`,
|
||||
[version.versionId, job.documentId, job.remoteIdempotencyKey, job.requestedPages, job.configVersion]
|
||||
);
|
||||
}
|
||||
|
||||
return version;
|
||||
return version;
|
||||
});
|
||||
}
|
||||
|
||||
|
|
@ -335,7 +704,7 @@ export class CatalogRepository {
|
|||
}
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions
|
||||
SET state = 'failed', error_code = $2, error_detail = $3
|
||||
SET state = 'failed', error_code = $2, error_detail = $3, retention_due_at = now() + interval '7 days'
|
||||
WHERE version_id = $1 AND state IN ('pending', 'indexing')`,
|
||||
[versionId, code, detail.slice(0, 2000)]
|
||||
);
|
||||
|
|
@ -372,6 +741,9 @@ export class CatalogRepository {
|
|||
if (!["ready", "superseded", "active"].includes(version.rows[0].state)) {
|
||||
throw new CatalogError("Only ready or superseded versions can be activated", 409, "VERSION_NOT_ACTIVATABLE");
|
||||
}
|
||||
if (["retention_deleting", "retention_deleted"].includes(version.rows[0].artifact_state ?? "none")) {
|
||||
throw new CatalogError("Version artifacts are unavailable", 409, "VERSION_ARTIFACTS_UNAVAILABLE");
|
||||
}
|
||||
if (version.rows[0].state === "active") {
|
||||
if (currentActive !== versionId) {
|
||||
throw new CatalogError("Active version invariant is inconsistent", 503, "ACTIVE_VERSION_INVARIANT_FAILED");
|
||||
|
|
@ -381,7 +753,7 @@ export class CatalogRepository {
|
|||
|
||||
if (currentActive) {
|
||||
await client.query(
|
||||
"UPDATE rag_source_versions SET state = 'superseded', superseded_at = now() WHERE source_id = $1 AND version_id = $2",
|
||||
"UPDATE rag_source_versions SET state = 'superseded', superseded_at = now(), retention_due_at = now() + interval '30 days' WHERE source_id = $1 AND version_id = $2",
|
||||
[sourceId, currentActive]
|
||||
);
|
||||
}
|
||||
|
|
@ -497,6 +869,50 @@ export class CatalogRepository {
|
|||
await this.pool.query("UPDATE rag_sources SET needs_reingest = true WHERE source_id = $1", [sourceId]);
|
||||
}
|
||||
|
||||
async listOcrRetentionCandidates(now: Date): Promise<OcrRetentionCandidate[]> {
|
||||
const result = await this.pool.query<{ version_id: string; source_id: string; state: SourceVersionState; artifact_state: OcrRetentionCandidate["artifactState"] }>(
|
||||
`SELECT v.version_id, v.source_id, v.state, v.artifact_state FROM rag_source_versions v
|
||||
JOIN rag_sources s ON s.source_id = v.source_id
|
||||
WHERE v.artifact_state IN ('present', 'retention_deleting')
|
||||
AND s.active_version_id IS DISTINCT FROM v.version_id
|
||||
AND (v.artifact_state = 'retention_deleting'
|
||||
OR (v.state = 'review_required' AND v.created_at + interval '30 days' <= $1)
|
||||
OR (v.state IN ('failed', 'rejected') AND COALESCE(v.retention_due_at, v.reviewed_at + interval '7 days', v.created_at + interval '7 days') <= $1)
|
||||
OR (v.state = 'superseded' AND COALESCE(v.retention_due_at, v.superseded_at + interval '30 days') <= $1))
|
||||
ORDER BY v.created_at, v.version_id`, [now]
|
||||
);
|
||||
return result.rows.map((row) => ({ versionId: row.version_id, sourceId: row.source_id, state: row.state, artifactState: row.artifact_state }));
|
||||
}
|
||||
|
||||
async expireOcrReview(sourceId: string, versionId: string, now: Date): Promise<boolean> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions v SET state = 'rejected', reviewed_at = $3, error_code = 'REVIEW_EXPIRED',
|
||||
error_detail = 'OCR review expired after 30 days', retention_due_at = $3 + interval '7 days'
|
||||
FROM rag_sources s WHERE v.source_id = $1 AND v.version_id = $2 AND v.source_id = s.source_id
|
||||
AND v.state = 'review_required' AND v.created_at + interval '30 days' <= $3
|
||||
AND s.active_version_id IS DISTINCT FROM v.version_id`, [sourceId, versionId, now]
|
||||
);
|
||||
return result.rowCount === 1;
|
||||
}
|
||||
|
||||
async claimOcrRetentionDeletion(sourceId: string, versionId: string, state: SourceVersionState, artifactState: OcrRetentionCandidate["artifactState"]): Promise<boolean> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions v SET artifact_state = 'retention_deleting' FROM rag_sources s
|
||||
WHERE v.source_id = $1 AND v.version_id = $2 AND v.state = $3 AND v.artifact_state = $4
|
||||
AND v.source_id = s.source_id AND s.active_version_id IS DISTINCT FROM v.version_id`, [sourceId, versionId, state, artifactState]
|
||||
);
|
||||
return result.rowCount === 1;
|
||||
}
|
||||
|
||||
async completeOcrRetentionDeletion(sourceId: string, versionId: string, state: SourceVersionState): Promise<boolean> {
|
||||
const result = await this.pool.query(
|
||||
`UPDATE rag_source_versions v SET artifact_state = 'retention_deleted' FROM rag_sources s
|
||||
WHERE v.source_id = $1 AND v.version_id = $2 AND v.state = $3 AND v.artifact_state = 'retention_deleting'
|
||||
AND v.source_id = s.source_id AND s.active_version_id IS DISTINCT FROM v.version_id`, [sourceId, versionId, state]
|
||||
);
|
||||
return result.rowCount === 1;
|
||||
}
|
||||
|
||||
async rollback(sourceId: string, targetVersionId: string, expectedActiveVersionId: string): Promise<CatalogVersionRow> {
|
||||
return this.withSourceLock(sourceId, async () => this.withVersionExclusiveLock(targetVersionId, async () => {
|
||||
const version = await this.activateVersion(sourceId, targetVersionId, expectedActiveVersionId);
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import path from "node:path";
|
||||
import { readFile } from "node:fs/promises";
|
||||
import { parseDocument, isSupportedDocument } from "../parsers/parser-registry.js";
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { parseDocument, isSupportedDocument, parsePdfPages, type ParsedPdfPage } from "../parsers/parser-registry.js";
|
||||
import { chunkDocument, codeChunkingPolicy, documentalChunkingPolicy } from "../process/chunking.js";
|
||||
import type { EmbeddingProvider } from "../embeddings/provider.js";
|
||||
import type { VectorStoreClient } from "../vectorstore/client.js";
|
||||
|
|
@ -21,6 +22,8 @@ import {
|
|||
import { listFilesRecursively } from "../../shared/utils/files.js";
|
||||
import { env } from "../../config/env.js";
|
||||
import { CatalogError, type CatalogDocumentInput, type CatalogRepository } from "../catalog/repository.js";
|
||||
import { selectPdfPagesForOcr } from "../ocr/detection.js";
|
||||
import { stageOcrArtifacts } from "../ocr/artifacts.js";
|
||||
|
||||
type PreparedDocument = CatalogDocumentInput & {
|
||||
chunks: ReturnType<typeof chunkDocument>;
|
||||
|
|
@ -36,11 +39,38 @@ type OriginalDocument = {
|
|||
originalHash: string;
|
||||
};
|
||||
|
||||
interface OcrPlan {
|
||||
original: OriginalDocument;
|
||||
pages: ParsedPdfPage[];
|
||||
requestedPages: number[];
|
||||
contentHash: string;
|
||||
mimeType: string;
|
||||
title: string;
|
||||
}
|
||||
|
||||
export interface OcrAccepted {
|
||||
accepted: true;
|
||||
sourceId: string;
|
||||
versionId: string;
|
||||
versionNumber: number;
|
||||
state: "indexing";
|
||||
phase: "ocr_queued";
|
||||
statusUrl: string;
|
||||
reviewUrl: null;
|
||||
activated: false;
|
||||
}
|
||||
|
||||
interface OcrIngestOptions {
|
||||
enabled: boolean;
|
||||
artifactRoot: string;
|
||||
}
|
||||
|
||||
export class IngestService {
|
||||
constructor(
|
||||
private readonly embeddingProvider: EmbeddingProvider,
|
||||
private readonly vectorStore: VectorStoreClient,
|
||||
private readonly catalog?: CatalogRepository
|
||||
private readonly catalog?: CatalogRepository,
|
||||
private readonly ocr?: OcrIngestOptions
|
||||
) {}
|
||||
|
||||
async cleanup(scope: RetrieveScope): Promise<{ deleted: number }> {
|
||||
|
|
@ -51,7 +81,7 @@ export class IngestService {
|
|||
return { deleted: count };
|
||||
}
|
||||
|
||||
async ingest(source: IngestSourceInput): Promise<IngestResult> {
|
||||
async ingest(source: IngestSourceInput): Promise<IngestResult | OcrAccepted> {
|
||||
if (!env.ingestWritesEnabled) {
|
||||
throw new CatalogError("Ingest writes are disabled during knowledge catalog maintenance", 503, "INGEST_WRITES_DISABLED");
|
||||
}
|
||||
|
|
@ -135,7 +165,7 @@ export class IngestService {
|
|||
};
|
||||
}
|
||||
|
||||
private async ingestWithLifecycle(source: IngestSourceInput, catalog: CatalogRepository): Promise<IngestResult> {
|
||||
private async ingestWithLifecycle(source: IngestSourceInput, catalog: CatalogRepository): Promise<IngestResult | OcrAccepted> {
|
||||
const activate = source.activate ?? true;
|
||||
if (activate && source.expectedActiveVersionId === undefined) {
|
||||
throw new CatalogError("expectedActiveVersionId is required when activate is true", 400, "EXPECTED_ACTIVE_VERSION_REQUIRED");
|
||||
|
|
@ -158,16 +188,24 @@ export class IngestService {
|
|||
if (!originalManifestHash) {
|
||||
throw new CatalogError("Original manifest hash could not be calculated", 422, "ORIGINAL_MANIFEST_HASH_UNAVAILABLE");
|
||||
}
|
||||
const tags = source.tags ?? [];
|
||||
const embeddingDimensions = this.embeddingProvider.dimensions ?? 4096;
|
||||
const metadataHash = buildMetadataHash(tags, { sourceType: source.sourceType, sourceRef: source.sourceRef });
|
||||
if (this.ocr?.enabled) {
|
||||
const plans = await this.planOcr(originalDocuments);
|
||||
if (plans.some(({ requestedPages }) => requestedPages.length > 0)) {
|
||||
return await this.acceptOcr(source, sourceId, attemptId, originalManifestHash, plans, tags, metadataHash, embeddingDimensions, catalog);
|
||||
}
|
||||
}
|
||||
|
||||
const preparedDocuments = await this.prepareDocuments(originalDocuments);
|
||||
|
||||
if (preparedDocuments.length === 0) {
|
||||
throw new CatalogError("The source has no supported documents with useful content", 422, "EMPTY_SOURCE");
|
||||
}
|
||||
|
||||
const tags = source.tags ?? [];
|
||||
const sourceContentHash = hashOrderedPairs(preparedDocuments.map((document) => [document.documentKey, document.contentHash]));
|
||||
|
||||
const embeddingDimensions = this.embeddingProvider.dimensions ?? 4096;
|
||||
const processingFingerprint = buildProcessingFingerprint({
|
||||
parserVersion: "native-v1",
|
||||
normalizationPolicy: "bom-crlf-trim-final-lf-v1",
|
||||
|
|
@ -179,8 +217,6 @@ export class IngestService {
|
|||
embeddingModel: this.embeddingProvider.modelName,
|
||||
embeddingDimensions
|
||||
});
|
||||
const metadataHash = buildMetadataHash(tags, { sourceType: source.sourceType, sourceRef: source.sourceRef });
|
||||
|
||||
return await catalog.withSourceLock(sourceId, async () => {
|
||||
const reusable = await catalog.findReusableVersion({ sourceId, sourceContentHash, processingFingerprint, metadataHash });
|
||||
if (reusable?.state === "active") {
|
||||
|
|
@ -311,6 +347,146 @@ export class IngestService {
|
|||
}
|
||||
}
|
||||
|
||||
private async planOcr(originals: OriginalDocument[]): Promise<OcrPlan[]> {
|
||||
const plans: OcrPlan[] = [];
|
||||
for (const original of originals) {
|
||||
if (path.extname(original.filePath).toLowerCase() === ".pdf") {
|
||||
const pages = await parsePdfPages(original.filePath);
|
||||
plans.push({
|
||||
original,
|
||||
pages,
|
||||
requestedPages: selectPdfPagesForOcr(original.filePath, pages),
|
||||
contentHash: sha256Hex(normalizeContentForHash(pages.map(({ text }) => text).join("\n"))),
|
||||
mimeType: "application/pdf",
|
||||
title: path.basename(original.filePath)
|
||||
});
|
||||
continue;
|
||||
}
|
||||
const parsed = await parseDocument(original.filePath);
|
||||
plans.push({
|
||||
original,
|
||||
pages: [],
|
||||
requestedPages: [],
|
||||
contentHash: sha256Hex(normalizeContentForHash(parsed.content)),
|
||||
mimeType: parsed.mimeType,
|
||||
title: parsed.title
|
||||
});
|
||||
}
|
||||
return plans;
|
||||
}
|
||||
|
||||
private async acceptOcr(
|
||||
source: IngestSourceInput,
|
||||
sourceId: string,
|
||||
attemptId: string,
|
||||
originalManifestHash: string,
|
||||
plans: OcrPlan[],
|
||||
tags: string[],
|
||||
metadataHash: string,
|
||||
embeddingDimensions: number,
|
||||
catalog: CatalogRepository
|
||||
): Promise<OcrAccepted> {
|
||||
const processingFingerprint = buildProcessingFingerprint({
|
||||
parserVersion: "native-pages-v1+ocr-v1",
|
||||
detectionPolicyVersion: "pdf-detection-v1",
|
||||
normalizationPolicy: "bom-crlf-trim-final-lf-v1",
|
||||
chunking: { code: codeChunkingPolicy, documental: documentalChunkingPolicy },
|
||||
embeddingProvider: this.embeddingProvider.providerName,
|
||||
embeddingModel: this.embeddingProvider.modelName,
|
||||
embeddingDimensions
|
||||
});
|
||||
return catalog.withSourceLock(sourceId, async () => {
|
||||
const pending = await catalog.findPendingOcrVersion({ sourceId, originalManifestHash, processingFingerprint, metadataHash });
|
||||
if (pending) {
|
||||
await catalog.updateAttempt(attemptId, { state: "completed", versionId: pending.version.versionId, inputHash: originalManifestHash });
|
||||
return this.ocrAccepted(sourceId, pending.version.versionId, pending.version.versionNumber);
|
||||
}
|
||||
|
||||
const versionId = randomUUID();
|
||||
const staged = await stageOcrArtifacts({
|
||||
rootDirectory: this.ocr!.artifactRoot,
|
||||
versionId,
|
||||
createdAt: new Date().toISOString(),
|
||||
documents: plans.map(({ original }) => ({
|
||||
documentId: original.documentId,
|
||||
documentKey: original.documentKey,
|
||||
bytes: original.bytes
|
||||
}))
|
||||
});
|
||||
if (staged.originalManifestHash !== originalManifestHash) {
|
||||
throw new CatalogError("Durable OCR manifest does not match the source", 500, "OCR_ARTIFACT_INTEGRITY_FAILED");
|
||||
}
|
||||
let created;
|
||||
try {
|
||||
created = await catalog.createPendingVersion({
|
||||
versionId,
|
||||
sourceId,
|
||||
sourceType: source.sourceType,
|
||||
sourceRef: source.sourceRef,
|
||||
originalManifestHash,
|
||||
sourceContentHash: null,
|
||||
processingFingerprint,
|
||||
metadataHash,
|
||||
tags,
|
||||
activateRequested: source.activate ?? true,
|
||||
embeddingProvider: this.embeddingProvider.providerName,
|
||||
embeddingModel: this.embeddingProvider.modelName,
|
||||
embeddingDimensions,
|
||||
qdrantCollection: env.qdrantCollection,
|
||||
expectedDocumentCount: plans.length,
|
||||
expectedPointCount: 0,
|
||||
documents: plans.map(({ original, requestedPages, contentHash, mimeType, title }) => ({
|
||||
documentId: original.documentId,
|
||||
documentKey: original.documentKey,
|
||||
originalHash: original.originalHash,
|
||||
originalHashKind: "bytes",
|
||||
contentHash: requestedPages.length > 0 ? null : contentHash,
|
||||
mimeType,
|
||||
title,
|
||||
chunkCount: 0,
|
||||
extractionMethod: requestedPages.length > 0 ? "ocr" : "native",
|
||||
artifactManifestPath: staged.manifestPath
|
||||
})),
|
||||
ocrPages: plans.flatMap(({ original, pages, requestedPages }) => pages.map((page) => ({
|
||||
documentId: original.documentId,
|
||||
page: page.page,
|
||||
extractionMethod: requestedPages.includes(page.page) ? "ocr" as const : "native" as const,
|
||||
nativeTextHash: page.textSha256
|
||||
}))),
|
||||
ocrJobs: plans.filter(({ requestedPages }) => requestedPages.length > 0).map(({ original, requestedPages }) => ({
|
||||
documentId: original.documentId,
|
||||
remoteIdempotencyKey: `${original.originalHash}:ocr-v1:${sha256Hex(JSON.stringify(requestedPages))}`,
|
||||
requestedPages,
|
||||
configVersion: "ocr-v1"
|
||||
}))
|
||||
});
|
||||
} catch (error) {
|
||||
if ((error as { code?: unknown }).code !== "23505") throw error;
|
||||
const winner = await catalog.findPendingOcrVersion({ sourceId, originalManifestHash, processingFingerprint, metadataHash });
|
||||
if (!winner) throw error;
|
||||
await catalog.updateAttempt(attemptId, { state: "completed", versionId: winner.version.versionId, inputHash: originalManifestHash });
|
||||
return this.ocrAccepted(sourceId, winner.version.versionId, winner.version.versionNumber);
|
||||
}
|
||||
await catalog.markIndexing(created.versionId);
|
||||
await catalog.updateAttempt(attemptId, { state: "completed", versionId: created.versionId, inputHash: originalManifestHash });
|
||||
return this.ocrAccepted(sourceId, created.versionId, created.versionNumber);
|
||||
});
|
||||
}
|
||||
|
||||
private ocrAccepted(sourceId: string, versionId: string, versionNumber: number): OcrAccepted {
|
||||
return {
|
||||
accepted: true,
|
||||
sourceId,
|
||||
versionId,
|
||||
versionNumber,
|
||||
state: "indexing",
|
||||
phase: "ocr_queued",
|
||||
statusUrl: `/ingestions/${versionId}`,
|
||||
reviewUrl: null,
|
||||
activated: false
|
||||
};
|
||||
}
|
||||
|
||||
private async readOriginalDocuments(source: IngestSourceInput, sourceId: string, files: string[]): Promise<OriginalDocument[]> {
|
||||
const originals: OriginalDocument[] = [];
|
||||
for (const filePath of files) {
|
||||
|
|
|
|||
129
src/modules/ocr/artifacts.ts
Normal file
129
src/modules/ocr/artifacts.ts
Normal file
|
|
@ -0,0 +1,129 @@
|
|||
import { createHash, randomUUID } from "node:crypto";
|
||||
import { link, lstat, mkdir, open, readdir, rm, stat, unlink } from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { canonicalJson, hashOrderedPairs, sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
const UUID = /^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/iu;
|
||||
const UUID_NAMESPACE_URL = Buffer.from("6ba7b8119dad11d180b400c04fd430c8", "hex");
|
||||
|
||||
interface StageInput {
|
||||
rootDirectory: string;
|
||||
versionId: string;
|
||||
createdAt: string;
|
||||
documents: Array<{ documentId: string; documentKey: string; bytes: Buffer }>;
|
||||
}
|
||||
|
||||
export function resolveArtifactPath(versionDirectory: string, relativePath: string): string {
|
||||
if (path.isAbsolute(relativePath)) throw new Error("Artifact path escapes version directory");
|
||||
const root = path.resolve(versionDirectory);
|
||||
const resolved = path.resolve(root, relativePath);
|
||||
if (resolved === root || !resolved.startsWith(`${root}${path.sep}`)) throw new Error("Artifact path escapes version directory");
|
||||
return resolved;
|
||||
}
|
||||
|
||||
export async function stageOcrArtifacts(input: StageInput): Promise<{
|
||||
versionDirectory: string;
|
||||
manifestPath: string;
|
||||
manifestSha256: string;
|
||||
originalManifestHash: string;
|
||||
}> {
|
||||
assertUuid(input.versionId);
|
||||
if (input.documents.length === 0 || Number.isNaN(Date.parse(input.createdAt))) throw new TypeError("Artifact manifest input is invalid");
|
||||
const documentIds = new Set(input.documents.map(({ documentId }) => documentId));
|
||||
const documentKeys = new Set(input.documents.map(({ documentKey }) => documentKey));
|
||||
if (documentIds.size !== input.documents.length || documentKeys.size !== input.documents.length) throw new TypeError("Artifact documents must be unique");
|
||||
|
||||
await mkdir(input.rootDirectory, { recursive: true, mode: 0o700 });
|
||||
const versionDirectory = path.join(path.resolve(input.rootDirectory), input.versionId);
|
||||
await mkdir(versionDirectory, { mode: 0o700 });
|
||||
try {
|
||||
const documents = [];
|
||||
for (const document of [...input.documents].sort((left, right) => Buffer.compare(Buffer.from(left.documentKey), Buffer.from(right.documentKey)))) {
|
||||
const documentArtifactId = uuidV5(document.documentId);
|
||||
const relativeDirectory = path.posix.join("documents", documentArtifactId);
|
||||
const directory = resolveArtifactPath(versionDirectory, relativeDirectory);
|
||||
await mkdir(directory, { recursive: true, mode: 0o700 });
|
||||
const originalPath = path.posix.join(relativeDirectory, "original.pdf");
|
||||
await durableWrite(resolveArtifactPath(versionDirectory, originalPath), document.bytes);
|
||||
documents.push({
|
||||
documentId: document.documentId,
|
||||
documentKey: document.documentKey,
|
||||
documentArtifactId,
|
||||
originalPath,
|
||||
originalSha256: sha256Hex(document.bytes)
|
||||
});
|
||||
}
|
||||
const originalManifestHash = hashOrderedPairs(documents.map(({ documentKey, originalSha256 }) => [documentKey, originalSha256]))!;
|
||||
const manifest = { schemaVersion: "1", versionId: input.versionId, createdAt: input.createdAt, originalManifestHash, documents };
|
||||
const serialized = canonicalJson(manifest);
|
||||
const manifestPath = path.join(versionDirectory, "manifest.json");
|
||||
await durableWrite(manifestPath, Buffer.from(serialized));
|
||||
await syncDirectory(versionDirectory);
|
||||
await syncDirectory(path.resolve(input.rootDirectory));
|
||||
return { versionDirectory, manifestPath, manifestSha256: sha256Hex(serialized), originalManifestHash };
|
||||
} catch (error) {
|
||||
await rm(versionDirectory, { recursive: true, force: true });
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
export async function sweepOrphanArtifacts(input: {
|
||||
rootDirectory: string;
|
||||
retainedVersionIds: ReadonlySet<string>;
|
||||
olderThan: Date;
|
||||
}): Promise<string[]> {
|
||||
for (const versionId of input.retainedVersionIds) assertUuid(versionId);
|
||||
let entries;
|
||||
try {
|
||||
entries = await readdir(input.rootDirectory, { withFileTypes: true });
|
||||
} catch (error) {
|
||||
if ((error as NodeJS.ErrnoException).code === "ENOENT") return [];
|
||||
throw error;
|
||||
}
|
||||
const removed: string[] = [];
|
||||
for (const entry of entries.sort((left, right) => left.name.localeCompare(right.name))) {
|
||||
if (!entry.isDirectory() || !UUID.test(entry.name) || input.retainedVersionIds.has(entry.name)) continue;
|
||||
const candidate = path.join(path.resolve(input.rootDirectory), entry.name);
|
||||
const current = await lstat(candidate);
|
||||
if (!current.isDirectory() || current.mtimeMs >= input.olderThan.getTime()) continue;
|
||||
await rm(candidate, { recursive: true, force: true });
|
||||
removed.push(entry.name);
|
||||
}
|
||||
if (removed.length > 0) await syncDirectory(path.resolve(input.rootDirectory));
|
||||
return removed;
|
||||
}
|
||||
|
||||
async function durableWrite(target: string, bytes: Buffer): Promise<void> {
|
||||
const temporary = `${target}.${randomUUID()}.tmp`;
|
||||
const handle = await open(temporary, "wx", 0o600);
|
||||
try {
|
||||
await handle.writeFile(bytes);
|
||||
await handle.chmod(0o600);
|
||||
await handle.sync();
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
try {
|
||||
await link(temporary, target);
|
||||
} finally {
|
||||
await unlink(temporary).catch(() => undefined);
|
||||
}
|
||||
await syncDirectory(path.dirname(target));
|
||||
}
|
||||
|
||||
async function syncDirectory(directory: string): Promise<void> {
|
||||
const handle = await open(directory, "r");
|
||||
try { await handle.sync(); } finally { await handle.close(); }
|
||||
}
|
||||
|
||||
function uuidV5(value: string): string {
|
||||
const bytes = createHash("sha1").update(Buffer.concat([UUID_NAMESPACE_URL, Buffer.from(value)])).digest().subarray(0, 16);
|
||||
bytes[6] = (bytes[6]! & 0x0f) | 0x50;
|
||||
bytes[8] = (bytes[8]! & 0x3f) | 0x80;
|
||||
const hex = bytes.toString("hex");
|
||||
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
|
||||
}
|
||||
|
||||
function assertUuid(value: string): void {
|
||||
if (!UUID.test(value)) throw new TypeError("Version identity must be a UUID");
|
||||
}
|
||||
175
src/modules/ocr/client.ts
Normal file
175
src/modules/ocr/client.ts
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
const OCR_CONFIG = {
|
||||
languages: ["es", "en"],
|
||||
dpi: 200,
|
||||
engine: "paddleocr",
|
||||
engineVersion: "3.4.0",
|
||||
runtimeVersion: "3.2.2",
|
||||
configVersion: "ocr-v1",
|
||||
returnLayout: true
|
||||
} as const;
|
||||
const TRANSIENT_STATUSES = new Set([502, 503]);
|
||||
const JOB_STATUSES = new Set(["queued", "running", "succeeded", "failed"]);
|
||||
|
||||
export interface OcrAck {
|
||||
jobId: string;
|
||||
status: "queued";
|
||||
documentSha256: string;
|
||||
requestedPages: number[];
|
||||
configVersion: "ocr-v1";
|
||||
createdAt: string;
|
||||
}
|
||||
|
||||
export interface OcrJobStatus {
|
||||
jobId: string;
|
||||
status: "queued" | "running" | "succeeded" | "failed";
|
||||
completedPages: number;
|
||||
totalPages: number;
|
||||
error: { code: string; message: string } | null;
|
||||
}
|
||||
|
||||
export interface OcrResult {
|
||||
schemaVersion: "1";
|
||||
jobId: string;
|
||||
documentSha256: string;
|
||||
engine: { name: "paddleocr"; version: "3.4.0"; runtime: "paddlepaddle-3.2.2"; device: "cpu"; configVersion: "ocr-v1"; dpi: 200 };
|
||||
pages: Array<{
|
||||
page: number;
|
||||
width: number;
|
||||
height: number;
|
||||
processingMs: number;
|
||||
text: string;
|
||||
metrics: { lineCount: number; nonWhitespaceCharacters: number; medianConfidence: number; p10Confidence: number; lowConfidenceLineRatio: number };
|
||||
lines: Array<{ lineId: string; text: string; confidence: number; bbox: [number, number, number, number] }>;
|
||||
}>;
|
||||
}
|
||||
|
||||
export class OcrClientError extends Error {
|
||||
constructor(public readonly code: string, public readonly status: number | undefined, public readonly retryable: boolean) {
|
||||
super(`OCR request failed: ${code}`);
|
||||
}
|
||||
}
|
||||
|
||||
interface OcrClientOptions {
|
||||
baseUrl: string;
|
||||
token: string;
|
||||
fetch?: typeof globalThis.fetch;
|
||||
sleep?: (milliseconds: number) => Promise<void>;
|
||||
}
|
||||
|
||||
export class OcrClient {
|
||||
private readonly baseUrl: string;
|
||||
private readonly requestFetch: typeof globalThis.fetch;
|
||||
private readonly sleep: (milliseconds: number) => Promise<void>;
|
||||
|
||||
constructor(private readonly options: OcrClientOptions) {
|
||||
this.baseUrl = options.baseUrl.replace(/\/+$/u, "");
|
||||
this.requestFetch = options.fetch ?? globalThis.fetch;
|
||||
this.sleep = options.sleep ?? ((milliseconds) => new Promise((resolve) => setTimeout(resolve, milliseconds)));
|
||||
}
|
||||
|
||||
async submit(file: Buffer, expected: { documentSha256: string; pages: number[] }, persistedIdempotencyKey?: string): Promise<OcrAck> {
|
||||
assertExpected(expected);
|
||||
if (sha256Hex(file) !== expected.documentSha256) throw integrityError();
|
||||
const payload = { documentSha256: expected.documentSha256, pages: expected.pages, ...OCR_CONFIG };
|
||||
const idempotencyKey = persistedIdempotencyKey ?? `${expected.documentSha256}:ocr-v1:${sha256Hex(JSON.stringify(expected.pages))}`;
|
||||
if (!idempotencyKey.trim() || /[\r\n]/u.test(idempotencyKey)) throw new TypeError("OCR idempotency key is invalid");
|
||||
const value = await this.requestJson("/v1/jobs", () => {
|
||||
const form = new FormData();
|
||||
form.set("file", new Blob([new Uint8Array(file)], { type: "application/pdf" }), "original.pdf");
|
||||
form.set("request", JSON.stringify(payload));
|
||||
return { method: "POST", headers: this.headers({ "Idempotency-Key": idempotencyKey }), body: form };
|
||||
});
|
||||
if (!isObject(value)
|
||||
|| value.jobId === undefined
|
||||
|| value.status !== "queued"
|
||||
|| value.documentSha256 !== expected.documentSha256
|
||||
|| value.configVersion !== "ocr-v1"
|
||||
|| !sameNumbers(value.requestedPages, expected.pages)
|
||||
|| typeof value.createdAt !== "string"
|
||||
|| Number.isNaN(Date.parse(value.createdAt))) throw integrityError();
|
||||
return value as unknown as OcrAck;
|
||||
}
|
||||
|
||||
async getStatus(jobId: string): Promise<OcrJobStatus> {
|
||||
const value = await this.requestJson(`/v1/jobs/${encodeURIComponent(jobId)}`, () => ({ headers: this.headers() }));
|
||||
if (!isObject(value) || value.jobId !== jobId || typeof value.status !== "string" || !JOB_STATUSES.has(value.status)
|
||||
|| !isCount(value.completedPages) || !isCount(value.totalPages) || value.completedPages > value.totalPages
|
||||
|| !(value.error === null || (isObject(value.error) && typeof value.error.code === "string" && typeof value.error.message === "string"))) {
|
||||
throw integrityError();
|
||||
}
|
||||
return value as unknown as OcrJobStatus;
|
||||
}
|
||||
|
||||
async pollUntilTerminal(jobId: string): Promise<OcrJobStatus> {
|
||||
let delay = 2_000;
|
||||
while (true) {
|
||||
const status = await this.getStatus(jobId);
|
||||
if (status.status === "succeeded" || status.status === "failed") return status;
|
||||
await this.sleep(delay);
|
||||
delay = Math.min(delay * 2, 15_000);
|
||||
}
|
||||
}
|
||||
|
||||
async getResult(jobId: string, expected: { documentSha256: string; pages: number[] }): Promise<OcrResult> {
|
||||
assertExpected(expected);
|
||||
const value = await this.requestJson(`/v1/jobs/${encodeURIComponent(jobId)}/result`, () => ({ headers: this.headers() }));
|
||||
if (!validResult(value, jobId, expected)) throw integrityError();
|
||||
return value;
|
||||
}
|
||||
|
||||
private headers(additional: Record<string, string> = {}): Headers {
|
||||
return new Headers({ Authorization: `Bearer ${this.options.token}`, ...additional });
|
||||
}
|
||||
|
||||
private async requestJson(pathname: string, buildInit: () => RequestInit): Promise<unknown> {
|
||||
for (let attempt = 0; attempt < 3; attempt += 1) {
|
||||
let response: Response;
|
||||
try {
|
||||
response = await this.requestFetch(`${this.baseUrl}${pathname}`, buildInit());
|
||||
} catch {
|
||||
if (attempt < 2) { await this.sleep(2_000 * 2 ** attempt); continue; }
|
||||
throw new OcrClientError("OCR_NETWORK_ERROR", undefined, false);
|
||||
}
|
||||
if (response.ok) return response.json();
|
||||
const body = await response.json().catch(() => ({})) as Record<string, unknown>;
|
||||
const detail = isObject(body.detail) ? body.detail : body;
|
||||
const code = typeof detail.code === "string" ? detail.code : `OCR_HTTP_${response.status}`;
|
||||
if (TRANSIENT_STATUSES.has(response.status) && attempt < 2) { await this.sleep(2_000 * 2 ** attempt); continue; }
|
||||
throw new OcrClientError(code, response.status, response.status === 429);
|
||||
}
|
||||
throw new OcrClientError("OCR_RETRY_EXHAUSTED", undefined, false);
|
||||
}
|
||||
}
|
||||
|
||||
function assertExpected(expected: { documentSha256: string; pages: number[] }): void {
|
||||
if (!/^[a-f0-9]{64}$/u.test(expected.documentSha256)
|
||||
|| expected.pages.length === 0
|
||||
|| expected.pages.some((page, index) => !Number.isInteger(page) || page < 1 || (index > 0 && page <= expected.pages[index - 1]!))) {
|
||||
throw new TypeError("OCR request identity is invalid");
|
||||
}
|
||||
}
|
||||
|
||||
function validResult(value: unknown, jobId: string, expected: { documentSha256: string; pages: number[] }): value is OcrResult {
|
||||
if (!isObject(value) || value.schemaVersion !== "1" || value.jobId !== jobId || value.documentSha256 !== expected.documentSha256
|
||||
|| !isObject(value.engine) || canonicalEngine(value.engine) !== "paddleocr|3.4.0|paddlepaddle-3.2.2|cpu|ocr-v1|200"
|
||||
|| !Array.isArray(value.pages) || !sameNumbers(value.pages.map((page) => isObject(page) ? page.page : undefined), expected.pages)) return false;
|
||||
return value.pages.every((page) => isObject(page) && isCount(page.page) && positiveCount(page.width) && positiveCount(page.height)
|
||||
&& isCount(page.processingMs) && typeof page.text === "string" && isObject(page.metrics) && Array.isArray(page.lines)
|
||||
&& page.lines.every((line) => isObject(line) && typeof line.lineId === "string" && line.lineId.length > 0 && typeof line.text === "string"
|
||||
&& validRatio(line.confidence) && Array.isArray(line.bbox) && line.bbox.length === 4 && line.bbox.every(Number.isFinite))
|
||||
&& page.text === page.lines.map((line) => (line as Record<string, unknown>).text).join("\n")
|
||||
&& page.metrics.lineCount === page.lines.length && isCount(page.metrics.nonWhitespaceCharacters)
|
||||
&& validRatio(page.metrics.medianConfidence) && validRatio(page.metrics.p10Confidence) && validRatio(page.metrics.lowConfidenceLineRatio));
|
||||
}
|
||||
|
||||
function canonicalEngine(engine: Record<string, unknown>): string {
|
||||
return [engine.name, engine.version, engine.runtime, engine.device, engine.configVersion, engine.dpi].join("|");
|
||||
}
|
||||
function isObject(value: unknown): value is Record<string, unknown> { return typeof value === "object" && value !== null && !Array.isArray(value); }
|
||||
function isCount(value: unknown): value is number { return Number.isInteger(value) && Number(value) >= 0; }
|
||||
function positiveCount(value: unknown): value is number { return isCount(value) && value > 0; }
|
||||
function validRatio(value: unknown): value is number { return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1; }
|
||||
function sameNumbers(value: unknown, expected: number[]): boolean { return Array.isArray(value) && value.length === expected.length && value.every((entry, index) => entry === expected[index]); }
|
||||
function integrityError(): Error { return new Error("OCR response integrity validation failed"); }
|
||||
114
src/modules/ocr/composition.ts
Normal file
114
src/modules/ocr/composition.ts
Normal file
|
|
@ -0,0 +1,114 @@
|
|||
import { canonicalJson, sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
export interface CandidateLineInput {
|
||||
lineId: string;
|
||||
text: string;
|
||||
confidence: number;
|
||||
bbox: [number, number, number, number];
|
||||
}
|
||||
|
||||
export interface CandidatePageInput {
|
||||
page: number;
|
||||
method: "native" | "ocr" | "blank";
|
||||
nativeText: string;
|
||||
rawOcrText: string;
|
||||
lines: CandidateLineInput[];
|
||||
}
|
||||
|
||||
export interface CandidateLine extends CandidateLineInput {
|
||||
lineSha256: string;
|
||||
}
|
||||
|
||||
export interface CandidatePage {
|
||||
page: number;
|
||||
method: CandidatePageInput["method"];
|
||||
nativeText: string;
|
||||
rawOcrText: string;
|
||||
lines: CandidateLine[];
|
||||
candidateText: string;
|
||||
candidateTextSha256: string;
|
||||
risks: string[];
|
||||
}
|
||||
|
||||
const RISK_TOKEN = /\b[A-Za-z]{2,}[A-Za-z0-9_-]*\d[A-Za-z0-9_-]*\b/g;
|
||||
const CONTEXT_WORD = /(?:codigo|error|regla|sqlstate|estado|identificador)\s*[:#-]?\s*$/iu;
|
||||
|
||||
export function prioritizeRiskTokens(text: string, comparisonText = ""): string[] {
|
||||
const matches = [...text.matchAll(RISK_TOKEN)];
|
||||
const counts = new Map<string, number>();
|
||||
for (const match of matches) counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
|
||||
|
||||
const unique = new Map<string, { token: string; position: number; score: number }>();
|
||||
for (const match of matches) {
|
||||
const token = match[0];
|
||||
if (unique.has(token)) continue;
|
||||
const position = match.index ?? 0;
|
||||
const differs = !comparisonText.includes(token);
|
||||
const mixedCase = /[A-Z]/u.test(token) && /[a-z]/u.test(token);
|
||||
const ambiguous = /[Oo0Ii1lSs5]/u.test(token);
|
||||
const contextAdjacent = CONTEXT_WORD.test(text.slice(Math.max(0, position - 40), position));
|
||||
unique.set(token, {
|
||||
token,
|
||||
position,
|
||||
score: Number(differs) * 2 + Number(contextAdjacent) * 2 + Number(mixedCase) + Number(ambiguous) + Number(counts.get(token) === 1)
|
||||
});
|
||||
}
|
||||
return [...unique.values()]
|
||||
.sort((left, right) => right.score - left.score || left.position - right.position)
|
||||
.map(({ token }) => token);
|
||||
}
|
||||
|
||||
export function composeCandidate(inputPages: CandidatePageInput[]): {
|
||||
text: string;
|
||||
textSha256: string;
|
||||
candidateSha256: string;
|
||||
pages: CandidatePage[];
|
||||
} {
|
||||
const pageNumbers = new Set<number>();
|
||||
const pages = [...inputPages]
|
||||
.sort((left, right) => left.page - right.page)
|
||||
.map((input): CandidatePage => {
|
||||
if (!Number.isInteger(input.page) || input.page < 1 || pageNumbers.has(input.page)) {
|
||||
throw new Error("Candidate pages must be unique positive one-based integers");
|
||||
}
|
||||
pageNumbers.add(input.page);
|
||||
const lines = input.method === "ocr"
|
||||
? input.lines
|
||||
.map((line, originalIndex) => ({ line, originalIndex }))
|
||||
.sort((left, right) => left.line.bbox[1] - right.line.bbox[1]
|
||||
|| left.line.bbox[0] - right.line.bbox[0]
|
||||
|| left.originalIndex - right.originalIndex)
|
||||
.map(({ line }) => ({
|
||||
...line,
|
||||
bbox: [line.bbox[0], line.bbox[1], line.bbox[2], line.bbox[3]] as [number, number, number, number],
|
||||
lineSha256: sha256Hex(line.text)
|
||||
}))
|
||||
: [];
|
||||
const candidateText = input.method === "native"
|
||||
? input.nativeText
|
||||
: input.method === "ocr"
|
||||
? lines.map(({ text }) => text).join("\n")
|
||||
: "";
|
||||
return {
|
||||
page: input.page,
|
||||
method: input.method,
|
||||
nativeText: input.nativeText,
|
||||
rawOcrText: input.rawOcrText,
|
||||
lines,
|
||||
candidateText,
|
||||
candidateTextSha256: sha256Hex(candidateText),
|
||||
risks: input.method === "blank" ? [] : prioritizeRiskTokens(candidateText, input.nativeText)
|
||||
};
|
||||
});
|
||||
|
||||
const nonBlank = pages.filter(({ candidateText }) => candidateText.length > 0);
|
||||
const text = nonBlank.map(({ page, candidateText }, index) => index === 0
|
||||
? candidateText
|
||||
: `--- Page ${page} ---\n\n${candidateText}`).join("\n\n");
|
||||
return {
|
||||
text,
|
||||
textSha256: sha256Hex(text),
|
||||
candidateSha256: sha256Hex(canonicalJson(pages)),
|
||||
pages
|
||||
};
|
||||
}
|
||||
70
src/modules/ocr/detection.ts
Normal file
70
src/modules/ocr/detection.ts
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
import path from "node:path";
|
||||
|
||||
export const DETECTION_POLICY_VERSION = "pdf-detection-v1" as const;
|
||||
|
||||
export interface NativeTextMetrics {
|
||||
nonWhitespaceCharacters: number;
|
||||
alphanumericCharacters: number;
|
||||
wordCount: number;
|
||||
replacementControlRatio: number;
|
||||
}
|
||||
|
||||
export interface OcrQualityMetrics {
|
||||
nonWhitespaceCharacters: number;
|
||||
medianConfidence: number;
|
||||
p10Confidence: number;
|
||||
lowConfidenceLineRatio: number;
|
||||
}
|
||||
|
||||
const CONTROL_CHARACTER = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f-\u009f]/u;
|
||||
|
||||
export function computeNativeMetrics(text: string): NativeTextMetrics {
|
||||
const normalized = text.normalize("NFC");
|
||||
const characters = Array.from(normalized);
|
||||
const nonWhitespaceCharacters = characters.filter((character) => /\S/u.test(character) && !CONTROL_CHARACTER.test(character)).length;
|
||||
const replacementOrControl = characters.filter((character) => character === "\uFFFD" || CONTROL_CHARACTER.test(character)).length;
|
||||
return {
|
||||
nonWhitespaceCharacters,
|
||||
alphanumericCharacters: characters.filter((character) => /[A-Za-z0-9]/u.test(character)).length,
|
||||
wordCount: normalized.trim() ? normalized.trim().split(/\s+/u).length : 0,
|
||||
replacementControlRatio: nonWhitespaceCharacters === 0 ? 0 : replacementOrControl / nonWhitespaceCharacters
|
||||
};
|
||||
}
|
||||
|
||||
export function isNativeTextSufficient(metrics: NativeTextMetrics): boolean {
|
||||
return metrics.nonWhitespaceCharacters >= 120
|
||||
&& metrics.alphanumericCharacters >= 80
|
||||
&& metrics.wordCount >= 20
|
||||
&& metrics.replacementControlRatio <= 0.01;
|
||||
}
|
||||
|
||||
export function selectPdfPagesForOcr(
|
||||
filePath: string,
|
||||
pages: ReadonlyArray<{ page: number; text: string }>
|
||||
): number[] {
|
||||
if (path.extname(filePath).toLowerCase() !== ".pdf") return [];
|
||||
|
||||
const selected = new Set<number>();
|
||||
for (const { page, text } of pages) {
|
||||
if (!Number.isInteger(page) || page < 1) throw new Error("PDF pages must use positive one-based integers");
|
||||
if (!isNativeTextSufficient(computeNativeMetrics(text))) selected.add(page);
|
||||
}
|
||||
return [...selected].sort((left, right) => left - right);
|
||||
}
|
||||
|
||||
export function classifyOcrPage(input: {
|
||||
inkCoverage: number;
|
||||
metrics: OcrQualityMetrics;
|
||||
}): { method: "blank" } | { method: "ocr" } | { method: "blocked"; errorCode: "OCR_QUALITY_BLOCKED" } {
|
||||
const { inkCoverage, metrics } = input;
|
||||
if (inkCoverage < 0.015 && metrics.nonWhitespaceCharacters < 10) return { method: "blank" };
|
||||
if (
|
||||
metrics.nonWhitespaceCharacters >= 40
|
||||
&& metrics.medianConfidence >= 0.8
|
||||
&& metrics.p10Confidence >= 0.5
|
||||
&& metrics.lowConfidenceLineRatio <= 0.2
|
||||
) {
|
||||
return { method: "ocr" };
|
||||
}
|
||||
return { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" };
|
||||
}
|
||||
108
src/modules/ocr/dispatcher.ts
Normal file
108
src/modules/ocr/dispatcher.ts
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
import type { OcrJobRow } from "../catalog/repository.js";
|
||||
import { OcrClientError, type OcrClient, type OcrResult } from "./client.js";
|
||||
|
||||
export interface OcrDispatchStore {
|
||||
claimNextOcrJob(leaseMs: number): Promise<OcrJobRow | undefined>;
|
||||
claimOcrJob(jobId: string, leaseMs: number): Promise<OcrJobRow | undefined>;
|
||||
recoverExpiredOcrLeases(): Promise<OcrJobRow[]>;
|
||||
setOcrRemoteJob(jobId: string, remoteJobId: string, leaseMs: number): Promise<void>;
|
||||
requeueOcrJob(jobId: string, code: string, detail: string, delayMs: number): Promise<void>;
|
||||
completeOcrJob(jobId: string, result: OcrResult): Promise<boolean>;
|
||||
failOcrJob(jobId: string, code: string, detail: string): Promise<void>;
|
||||
markReviewRequired(versionId: string): Promise<void>;
|
||||
markFailed(versionId: string, code: string, detail: string): Promise<void>;
|
||||
}
|
||||
|
||||
export type OcrDispatchInput = { bytes: Buffer; documentSha256: string };
|
||||
export type OcrDispatchResult = "idle" | "pending" | "succeeded" | "failed";
|
||||
|
||||
export class OcrDispatcher {
|
||||
private activeDrain: Promise<number> | undefined;
|
||||
|
||||
constructor(
|
||||
private readonly store: OcrDispatchStore,
|
||||
private readonly client: OcrClient,
|
||||
private readonly loadInput: (job: OcrJobRow) => Promise<OcrDispatchInput>,
|
||||
private readonly leaseMs = 30_000
|
||||
) {}
|
||||
|
||||
async runOnce(): Promise<OcrDispatchResult> {
|
||||
const job = await this.store.claimNextOcrJob(this.leaseMs);
|
||||
if (!job) return "idle";
|
||||
|
||||
return this.dispatch(job);
|
||||
}
|
||||
|
||||
private async dispatch(job: OcrJobRow): Promise<OcrDispatchResult> {
|
||||
try {
|
||||
const input = await this.loadInput(job);
|
||||
let remoteJobId = job.remoteJobId;
|
||||
if (!remoteJobId) {
|
||||
const acknowledgement = await this.client.submit(input.bytes, {
|
||||
documentSha256: input.documentSha256,
|
||||
pages: job.requestedPages
|
||||
}, job.remoteIdempotencyKey);
|
||||
remoteJobId = acknowledgement.jobId;
|
||||
await this.store.setOcrRemoteJob(job.jobId, remoteJobId, this.leaseMs);
|
||||
}
|
||||
|
||||
const status = await this.client.getStatus(remoteJobId);
|
||||
if (status.status === "queued" || status.status === "running") {
|
||||
const delayMs = Math.min(2_000 * 2 ** Math.max(0, job.attemptCount - 1), 15_000);
|
||||
await this.store.requeueOcrJob(job.jobId, "OCR_PENDING", status.status, delayMs);
|
||||
return "pending";
|
||||
}
|
||||
if (status.status === "failed") {
|
||||
const code = status.error?.code ?? "OCR_REMOTE_FAILED";
|
||||
await this.fail(job, code, status.error?.message ?? "OCR processing failed");
|
||||
return "failed";
|
||||
}
|
||||
|
||||
const result = await this.client.getResult(remoteJobId, {
|
||||
documentSha256: input.documentSha256,
|
||||
pages: job.requestedPages
|
||||
});
|
||||
const versionComplete = await this.store.completeOcrJob(job.jobId, result);
|
||||
if (versionComplete) await this.store.markReviewRequired(job.versionId);
|
||||
return "succeeded";
|
||||
} catch (error) {
|
||||
const detail = error instanceof Error ? error.message : "Unknown OCR dispatch failure";
|
||||
if (error instanceof OcrClientError && error.retryable) {
|
||||
await this.store.requeueOcrJob(job.jobId, error.code, detail, 15_000);
|
||||
return "pending";
|
||||
}
|
||||
const code = error instanceof OcrClientError ? error.code : "OCR_DISPATCH_FAILED";
|
||||
await this.fail(job, code, detail);
|
||||
return "failed";
|
||||
}
|
||||
}
|
||||
|
||||
async recoverExpiredLeases(): Promise<number> {
|
||||
const recovered = await this.store.recoverExpiredOcrLeases();
|
||||
for (const job of recovered) {
|
||||
const claimed = await this.store.claimOcrJob(job.jobId, this.leaseMs);
|
||||
if (claimed) await this.dispatch(claimed);
|
||||
}
|
||||
return recovered.length;
|
||||
}
|
||||
|
||||
dispatchAvailable(): Promise<number> {
|
||||
if (this.activeDrain) return this.activeDrain;
|
||||
const drain = this.drainAvailable();
|
||||
this.activeDrain = drain;
|
||||
const clear = () => { if (this.activeDrain === drain) this.activeDrain = undefined; };
|
||||
void drain.then(clear, clear);
|
||||
return drain;
|
||||
}
|
||||
|
||||
private async drainAvailable(): Promise<number> {
|
||||
let processed = 0;
|
||||
while (await this.runOnce() !== "idle") processed += 1;
|
||||
return processed;
|
||||
}
|
||||
|
||||
private async fail(job: OcrJobRow, code: string, detail: string): Promise<void> {
|
||||
await this.store.failOcrJob(job.jobId, code, detail);
|
||||
await this.store.markFailed(job.versionId, code, detail);
|
||||
}
|
||||
}
|
||||
67
src/modules/ocr/indexing.ts
Normal file
67
src/modules/ocr/indexing.ts
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
import { CatalogError } from "../catalog/errors.js";
|
||||
|
||||
export interface ApprovedOcrCandidate {
|
||||
versionId: string;
|
||||
sourceId: string;
|
||||
state: "indexing" | "rejected";
|
||||
activateRequested: boolean;
|
||||
expectedActiveVersionId: string | null;
|
||||
reviewedText: string;
|
||||
reviewedTextSha256: string;
|
||||
processingFingerprint: string;
|
||||
metadataHash: string;
|
||||
}
|
||||
|
||||
export interface OcrIndexingStore {
|
||||
findReusableVersion(candidate: ApprovedOcrCandidate): Promise<{ versionId: string } | undefined>;
|
||||
indexReviewed(candidate: ApprovedOcrCandidate): Promise<number>;
|
||||
markReady(versionId: string, verifiedPointCount: number): Promise<void>;
|
||||
settleReusable(candidate: ApprovedOcrCandidate, reusableVersionId: string, activate: boolean): Promise<boolean>;
|
||||
activateVersion(sourceId: string, versionId: string, expectedActiveVersionId: string | null): Promise<string>;
|
||||
}
|
||||
|
||||
export type OcrIndexingResult = {
|
||||
versionId: string;
|
||||
state: "ready" | "active" | "rejected";
|
||||
activated: boolean;
|
||||
activatedVersionId?: string;
|
||||
errorCode?: "DUPLICATE_REUSABLE_VERSION";
|
||||
};
|
||||
|
||||
function activeRace(error: unknown): never {
|
||||
if (error instanceof CatalogError && error.statusCode === 409) {
|
||||
throw new CatalogError("Active version changed during OCR approval", 409, "ACTIVE_VERSION_CHANGED");
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
|
||||
export class OcrIndexingService {
|
||||
constructor(private readonly store: OcrIndexingStore) {}
|
||||
|
||||
async index(candidate: ApprovedOcrCandidate): Promise<OcrIndexingResult> {
|
||||
if (candidate.state !== "indexing") throw new CatalogError("Only approved OCR candidates can be indexed", 409, "INVALID_VERSION_STATE");
|
||||
const reusable = await this.store.findReusableVersion(candidate);
|
||||
if (reusable) {
|
||||
let activated: boolean;
|
||||
try {
|
||||
activated = await this.store.settleReusable(candidate, reusable.versionId, candidate.activateRequested);
|
||||
} catch (error) {
|
||||
return activeRace(error);
|
||||
}
|
||||
return {
|
||||
versionId: candidate.versionId, state: "rejected", activated,
|
||||
...(activated ? { activatedVersionId: reusable.versionId } : {}), errorCode: "DUPLICATE_REUSABLE_VERSION"
|
||||
};
|
||||
}
|
||||
|
||||
const verifiedPointCount = await this.store.indexReviewed(candidate);
|
||||
await this.store.markReady(candidate.versionId, verifiedPointCount);
|
||||
if (!candidate.activateRequested) return { versionId: candidate.versionId, state: "ready", activated: false };
|
||||
try {
|
||||
const activatedVersionId = await this.store.activateVersion(candidate.sourceId, candidate.versionId, candidate.expectedActiveVersionId);
|
||||
return { versionId: candidate.versionId, state: "active", activated: true, activatedVersionId };
|
||||
} catch (error) {
|
||||
return activeRace(error);
|
||||
}
|
||||
}
|
||||
}
|
||||
45
src/modules/ocr/retention.ts
Normal file
45
src/modules/ocr/retention.ts
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
import { rm } from "node:fs/promises";
|
||||
import type { SourceVersionState } from "../../shared/types/rag.js";
|
||||
import { resolveArtifactPath } from "./artifacts.js";
|
||||
|
||||
export interface OcrRetentionCandidate {
|
||||
versionId: string;
|
||||
sourceId: string;
|
||||
state: SourceVersionState;
|
||||
artifactState: "present" | "retention_deleting";
|
||||
}
|
||||
|
||||
export interface OcrRetentionStore {
|
||||
listOcrRetentionCandidates(now: Date): Promise<OcrRetentionCandidate[]>;
|
||||
withVersionTryLock<T>(versionId: string, handler: () => Promise<T>): Promise<T | undefined>;
|
||||
expireOcrReview(sourceId: string, versionId: string, now: Date): Promise<boolean>;
|
||||
claimOcrRetentionDeletion(sourceId: string, versionId: string, state: SourceVersionState, artifactState: OcrRetentionCandidate["artifactState"]): Promise<boolean>;
|
||||
completeOcrRetentionDeletion(sourceId: string, versionId: string, state: SourceVersionState): Promise<boolean>;
|
||||
}
|
||||
|
||||
export class OcrRetentionService {
|
||||
constructor(private readonly store: OcrRetentionStore, private readonly artifactRoot: string) {}
|
||||
|
||||
async runOnce(now = new Date()): Promise<{ expired: number; deleted: number; resumed: number; skipped: number }> {
|
||||
const result = { expired: 0, deleted: 0, resumed: 0, skipped: 0 };
|
||||
for (const candidate of await this.store.listOcrRetentionCandidates(now)) {
|
||||
if (candidate.state === "active") {
|
||||
result.skipped += 1;
|
||||
continue;
|
||||
}
|
||||
const outcome = await this.store.withVersionTryLock(candidate.versionId, async () => {
|
||||
if (candidate.state === "review_required") {
|
||||
return await this.store.expireOcrReview(candidate.sourceId, candidate.versionId, now) ? "expired" : "skipped";
|
||||
}
|
||||
const claimed = await this.store.claimOcrRetentionDeletion(candidate.sourceId, candidate.versionId, candidate.state, candidate.artifactState);
|
||||
if (!claimed) return "skipped";
|
||||
await rm(resolveArtifactPath(this.artifactRoot, candidate.versionId), { recursive: true, force: true });
|
||||
return await this.store.completeOcrRetentionDeletion(candidate.sourceId, candidate.versionId, candidate.state)
|
||||
? candidate.artifactState === "retention_deleting" ? "resumed" : "deleted"
|
||||
: "skipped";
|
||||
});
|
||||
result[outcome ?? "skipped"] += 1;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
}
|
||||
143
src/modules/ocr/review.ts
Normal file
143
src/modules/ocr/review.ts
Normal file
|
|
@ -0,0 +1,143 @@
|
|||
import { CatalogError } from "../catalog/errors.js";
|
||||
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
export interface OcrReviewLine {
|
||||
lineId: string;
|
||||
text: string;
|
||||
confidence: number;
|
||||
bbox: [number, number, number, number];
|
||||
lineSha256: string;
|
||||
}
|
||||
|
||||
export interface OcrReviewCandidate {
|
||||
versionId: string;
|
||||
sourceId: string;
|
||||
state: "review_required" | "indexing" | "rejected";
|
||||
candidateSha256: string;
|
||||
baseActiveVersionId: string | null;
|
||||
currentActiveVersionId: string | null;
|
||||
activateRequested: boolean;
|
||||
processingFingerprint: string;
|
||||
metadataHash: string;
|
||||
documents: Array<{ documentId: string; pages: Array<{
|
||||
page: number;
|
||||
imageUrl: string;
|
||||
nativeText: string;
|
||||
ocr: { text: string; lines: OcrReviewLine[] };
|
||||
candidateText: string;
|
||||
differences: string[];
|
||||
risks: string[];
|
||||
}> }>;
|
||||
}
|
||||
|
||||
export interface OcrCorrection {
|
||||
documentId: string;
|
||||
page: number;
|
||||
lineId: string;
|
||||
expectedLineSha256: string;
|
||||
replacementText: string;
|
||||
}
|
||||
|
||||
export interface OcrReviewStore {
|
||||
loadCandidate(versionId: string): Promise<OcrReviewCandidate | undefined>;
|
||||
commitApproval(input: {
|
||||
candidate: OcrReviewCandidate;
|
||||
candidateSha256: string;
|
||||
expectedActiveVersionId: string | null;
|
||||
reviewedBy: string;
|
||||
corrections: OcrCorrection[];
|
||||
reviewedText: string;
|
||||
reviewedTextSha256: string;
|
||||
}): Promise<void>;
|
||||
commitRejection(input: {
|
||||
candidate: OcrReviewCandidate;
|
||||
candidateSha256: string;
|
||||
reviewedBy: string;
|
||||
reason: string;
|
||||
}): Promise<void>;
|
||||
}
|
||||
|
||||
export interface ApprovedOcrReview {
|
||||
versionId: string;
|
||||
sourceId: string;
|
||||
state: "indexing";
|
||||
activateRequested: boolean;
|
||||
expectedActiveVersionId: string | null;
|
||||
reviewedText: string;
|
||||
reviewedTextSha256: string;
|
||||
processingFingerprint: string;
|
||||
metadataHash: string;
|
||||
}
|
||||
|
||||
function conflict(message: string, code = "REVIEW_CONFLICT"): CatalogError {
|
||||
return new CatalogError(message, 409, code);
|
||||
}
|
||||
|
||||
export class OcrReviewService {
|
||||
constructor(private readonly store: OcrReviewStore) {}
|
||||
|
||||
async view(versionId: string): Promise<OcrReviewCandidate> {
|
||||
const candidate = await this.store.loadCandidate(versionId);
|
||||
if (!candidate) throw new CatalogError("OCR review candidate not found", 404, "REVIEW_NOT_FOUND");
|
||||
if (candidate.state !== "review_required") throw conflict("Version is not awaiting OCR review", "INVALID_VERSION_STATE");
|
||||
return candidate;
|
||||
}
|
||||
|
||||
async approve(versionId: string, input: {
|
||||
candidateSha256: string;
|
||||
expectedActiveVersionId: string | null;
|
||||
reviewedBy: string;
|
||||
corrections: OcrCorrection[];
|
||||
}): Promise<ApprovedOcrReview> {
|
||||
const candidate = await this.view(versionId);
|
||||
if (input.candidateSha256 !== candidate.candidateSha256) throw conflict("Candidate hash is stale");
|
||||
if (input.expectedActiveVersionId !== candidate.baseActiveVersionId || input.expectedActiveVersionId !== candidate.currentActiveVersionId) {
|
||||
throw conflict("Active version changed during review", "ACTIVE_VERSION_CHANGED");
|
||||
}
|
||||
|
||||
const reviewedCandidate = structuredClone(candidate);
|
||||
const lines = new Map<string, OcrReviewLine>();
|
||||
for (const document of reviewedCandidate.documents) for (const page of document.pages) for (const line of page.ocr.lines) {
|
||||
const key = `${document.documentId}:${page.page}:${line.lineId}`;
|
||||
if (lines.has(key)) throw conflict("Candidate line identities are duplicated", "CORRECTION_CONFLICT");
|
||||
lines.set(key, line);
|
||||
}
|
||||
const targets = new Set<string>();
|
||||
for (const correction of input.corrections) {
|
||||
const key = `${correction.documentId}:${correction.page}:${correction.lineId}`;
|
||||
const line = lines.get(key);
|
||||
if (targets.has(key) || !line || line.lineSha256 !== correction.expectedLineSha256) {
|
||||
throw conflict("Correction target is stale or duplicated", "CORRECTION_CONFLICT");
|
||||
}
|
||||
targets.add(key);
|
||||
line.text = correction.replacementText;
|
||||
line.lineSha256 = sha256Hex(line.text);
|
||||
}
|
||||
for (const document of reviewedCandidate.documents) for (const page of document.pages) {
|
||||
if (page.ocr.lines.length > 0) page.ocr.text = page.candidateText = page.ocr.lines.map(({ text }) => text).join("\n");
|
||||
}
|
||||
const reviewedText = reviewedCandidate.documents.flatMap(({ pages }) => pages.filter(({ candidateText }) => candidateText).map(({ candidateText }) => candidateText)).join("\n\n");
|
||||
const reviewedTextSha256 = sha256Hex(reviewedText);
|
||||
await this.store.commitApproval({ candidate: reviewedCandidate, ...input, reviewedText, reviewedTextSha256 });
|
||||
return {
|
||||
versionId, sourceId: candidate.sourceId, state: "indexing", activateRequested: candidate.activateRequested,
|
||||
expectedActiveVersionId: input.expectedActiveVersionId, reviewedText, reviewedTextSha256,
|
||||
processingFingerprint: candidate.processingFingerprint, metadataHash: candidate.metadataHash
|
||||
};
|
||||
}
|
||||
|
||||
async reject(versionId: string, input: { candidateSha256: string; reviewedBy: string; reason: string }): Promise<{
|
||||
versionId: string; state: "rejected"; activated: false;
|
||||
}> {
|
||||
if (!input || typeof input.candidateSha256 !== "string" || typeof input.reviewedBy !== "string" || typeof input.reason !== "string") {
|
||||
throw new CatalogError("Candidate hash, reviewer, and rejection reason are required", 400, "INVALID_REJECTION");
|
||||
}
|
||||
const reviewedBy = input.reviewedBy?.trim();
|
||||
const reason = input.reason?.trim();
|
||||
if (!input.candidateSha256 || !reviewedBy || !reason) throw new CatalogError("Candidate hash, reviewer, and rejection reason are required", 400, "INVALID_REJECTION");
|
||||
const candidate = await this.view(versionId);
|
||||
if (input.candidateSha256 !== candidate.candidateSha256) throw conflict("Candidate hash is stale");
|
||||
await this.store.commitRejection({ candidate, candidateSha256: input.candidateSha256, reviewedBy, reason });
|
||||
return { versionId, state: "rejected", activated: false };
|
||||
}
|
||||
}
|
||||
|
|
@ -1,7 +1,9 @@
|
|||
import { readFile } from "node:fs/promises";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import path from "node:path";
|
||||
import pdf from "pdf-parse";
|
||||
import type { ChunkingMode } from "../process/chunking.js";
|
||||
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
export interface ParsedDocument {
|
||||
title: string;
|
||||
|
|
@ -10,6 +12,18 @@ export interface ParsedDocument {
|
|||
chunkMode: ChunkingMode;
|
||||
}
|
||||
|
||||
export interface ParsedPdfPage {
|
||||
page: number;
|
||||
text: string;
|
||||
textSha256: string;
|
||||
}
|
||||
|
||||
interface PdfPageData {
|
||||
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||
items: Array<{ str?: string }>;
|
||||
}>;
|
||||
}
|
||||
|
||||
const documentalExtensions = [".md", ".txt", ".pdf"] as const;
|
||||
const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const;
|
||||
const parserExtensions = [...documentalExtensions, ...codeExtensions] as const;
|
||||
|
|
@ -19,6 +33,9 @@ export function supportedParserExtensions(): string[] {
|
|||
}
|
||||
|
||||
export function isSupportedDocument(filePath: string): boolean {
|
||||
if (path.basename(filePath).toLowerCase() === "cmakelists.txt") {
|
||||
return false;
|
||||
}
|
||||
return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]);
|
||||
}
|
||||
|
||||
|
|
@ -43,16 +60,42 @@ function inferMimeType(extension: string, chunkMode: ChunkingMode): string {
|
|||
return "text/plain";
|
||||
}
|
||||
|
||||
export async function parsePdfPages(filePath: string | URL): Promise<ParsedPdfPage[]> {
|
||||
const resolvedPath = filePath instanceof URL ? fileURLToPath(filePath) : filePath;
|
||||
if (path.extname(resolvedPath).toLowerCase() !== ".pdf") {
|
||||
throw new Error("Only PDF documents support page extraction");
|
||||
}
|
||||
|
||||
const pages: ParsedPdfPage[] = [];
|
||||
const bytes = Uint8Array.from(await readFile(filePath));
|
||||
const result = await pdf(bytes as Buffer, {
|
||||
version: "v2.0.550",
|
||||
pagerender: async (pageData: PdfPageData) => {
|
||||
const textContent = await pageData.getTextContent({
|
||||
normalizeWhitespace: false,
|
||||
disableCombineTextItems: false
|
||||
});
|
||||
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||
pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) });
|
||||
return text;
|
||||
}
|
||||
});
|
||||
|
||||
if (result.numrender !== result.numpages || pages.length !== result.numpages) {
|
||||
throw new Error("PDF page extraction did not render every page");
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
|
||||
export async function parseDocument(filePath: string): Promise<ParsedDocument> {
|
||||
const extension = path.extname(filePath).toLowerCase();
|
||||
const chunkMode = inferChunkMode(filePath);
|
||||
|
||||
if (extension === ".pdf") {
|
||||
const buffer = await readFile(filePath);
|
||||
const result = await pdf(buffer);
|
||||
const pages = await parsePdfPages(filePath);
|
||||
return {
|
||||
title: path.basename(filePath),
|
||||
content: result.text.trim(),
|
||||
content: pages.map((page) => page.text).filter(Boolean).join("\n\n"),
|
||||
mimeType: inferMimeType(extension, chunkMode),
|
||||
chunkMode
|
||||
};
|
||||
|
|
|
|||
67
tests/catalog/migration-002.test.ts
Normal file
67
tests/catalog/migration-002.test.ts
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { readFile } from "node:fs/promises";
|
||||
import test from "node:test";
|
||||
|
||||
const migrationUrl = new URL("../../migrations/002_ocr_review.sql", import.meta.url);
|
||||
|
||||
async function readMigration(): Promise<string> {
|
||||
return readFile(migrationUrl, "utf8");
|
||||
}
|
||||
|
||||
test("migration 002 persists durable OCR jobs with idempotency and lease fields", async () => {
|
||||
const sql = await readMigration();
|
||||
|
||||
assert.match(sql, /CREATE TABLE IF NOT EXISTS rag_ocr_jobs/i);
|
||||
assert.match(sql, /remote_job_id text NULL/i);
|
||||
assert.match(sql, /remote_idempotency_key text NOT NULL/i);
|
||||
assert.match(sql, /requested_pages integer\[\] NOT NULL/i);
|
||||
assert.match(sql, /state text NOT NULL CHECK \(state IN \('queued', 'running', 'succeeded', 'failed'\)\)/i);
|
||||
assert.match(sql, /attempt_count integer NOT NULL DEFAULT 0/i);
|
||||
assert.match(sql, /heartbeat_at timestamptz NULL/i);
|
||||
assert.match(sql, /lease_expires_at timestamptz NULL/i);
|
||||
assert.match(sql, /next_attempt_at timestamptz NULL/i);
|
||||
assert.match(sql, /UNIQUE \(version_id, document_id\)/i);
|
||||
assert.match(
|
||||
sql,
|
||||
/FOREIGN KEY \(version_id, document_id\)\s+REFERENCES rag_version_documents\(version_id, document_id\) ON DELETE RESTRICT/i
|
||||
);
|
||||
});
|
||||
|
||||
test("migration 002 stores auditable page extraction hashes, metrics, and risks", async () => {
|
||||
const sql = await readMigration();
|
||||
|
||||
assert.match(sql, /CREATE TABLE IF NOT EXISTS rag_document_pages/i);
|
||||
assert.match(sql, /page_number integer NOT NULL CHECK \(page_number > 0\)/i);
|
||||
assert.match(sql, /extraction_method text NOT NULL CHECK \(extraction_method IN \('native', 'ocr', 'blank'\)\)/i);
|
||||
assert.match(sql, /native_text_hash char\(64\) NULL/i);
|
||||
assert.match(sql, /ocr_text_hash char\(64\) NULL/i);
|
||||
assert.match(sql, /candidate_text_hash char\(64\) NULL/i);
|
||||
assert.match(sql, /reviewed_text_hash char\(64\) NULL/i);
|
||||
assert.match(sql, /metrics jsonb NOT NULL DEFAULT '\{\}'/i);
|
||||
assert.match(sql, /risk_tokens jsonb NOT NULL DEFAULT '\[\]'/i);
|
||||
assert.match(sql, /blocked_reason text NULL/i);
|
||||
assert.match(sql, /PRIMARY KEY \(version_id, document_id, page_number\)/i);
|
||||
});
|
||||
|
||||
test("migration 002 makes review correction targets unique and page-bound", async () => {
|
||||
const sql = await readMigration();
|
||||
|
||||
assert.match(sql, /CREATE TABLE IF NOT EXISTS rag_review_corrections/i);
|
||||
assert.match(sql, /expected_line_hash char\(64\) NOT NULL/i);
|
||||
assert.match(sql, /replacement_text text NOT NULL/i);
|
||||
assert.match(sql, /reviewed_by text NOT NULL/i);
|
||||
assert.match(
|
||||
sql,
|
||||
/FOREIGN KEY \(version_id, document_id, page_number\)\s+REFERENCES rag_document_pages\(version_id, document_id, page_number\) ON DELETE RESTRICT/i
|
||||
);
|
||||
assert.match(sql, /UNIQUE \(version_id, document_id, page_number, line_id\)/i);
|
||||
});
|
||||
|
||||
test("migration 002 prevents duplicate non-terminal OCR identities", async () => {
|
||||
const sql = await readMigration();
|
||||
|
||||
assert.match(
|
||||
sql,
|
||||
/CREATE UNIQUE INDEX IF NOT EXISTS rag_one_pending_ocr_identity\s+ON rag_source_versions\(source_id, original_manifest_hash, processing_fingerprint, metadata_hash\)\s+WHERE source_content_hash IS NULL\s+AND state IN \('pending', 'indexing', 'review_required'\)/i
|
||||
);
|
||||
});
|
||||
147
tests/catalog/repository-ocr.test.ts
Normal file
147
tests/catalog/repository-ocr.test.ts
Normal file
|
|
@ -0,0 +1,147 @@
|
|||
import assert from "node:assert/strict";
|
||||
import test from "node:test";
|
||||
import { CatalogError, CatalogRepository } from "../../src/modules/catalog/repository.js";
|
||||
|
||||
const versionRow = {
|
||||
version_id: "version-1",
|
||||
source_id: "source-1",
|
||||
version_number: 2,
|
||||
previous_version_id: "version-0",
|
||||
state: "indexing",
|
||||
tags: [],
|
||||
source_content_hash: null,
|
||||
processing_fingerprint: "fingerprint",
|
||||
metadata_hash: "metadata",
|
||||
embedding_provider: "test",
|
||||
embedding_model: "test",
|
||||
embedding_dimensions: 3,
|
||||
expected_document_count: 1,
|
||||
expected_point_count: 0,
|
||||
verified_point_count: 0,
|
||||
qdrant_collection: "rag"
|
||||
};
|
||||
|
||||
const jobRow = {
|
||||
ocr_job_id: "job-1",
|
||||
ocr_document_id: "document-1",
|
||||
ocr_remote_job_id: "remote-1",
|
||||
ocr_remote_idempotency_key: "document-hash:ocr-v1:pages-hash",
|
||||
ocr_state: "running",
|
||||
ocr_requested_pages: [1, 3],
|
||||
ocr_completed_pages: 1,
|
||||
ocr_config_version: "ocr-v1",
|
||||
ocr_attempt_count: 2,
|
||||
ocr_heartbeat_at: new Date("2026-09-14T10:00:00Z"),
|
||||
ocr_lease_expires_at: new Date("2026-09-14T10:01:00Z"),
|
||||
ocr_next_attempt_at: null,
|
||||
ocr_error_code: null,
|
||||
ocr_error_detail: null
|
||||
};
|
||||
|
||||
function poolFor(handler: (sql: string, params?: unknown[]) => Promise<{ rowCount: number; rows: unknown[] }>) {
|
||||
const client = {
|
||||
query: async (sql: string, params?: unknown[]) => /^(BEGIN|COMMIT|ROLLBACK|SET CONSTRAINTS)/.test(sql.trim())
|
||||
? { rowCount: 0, rows: [] }
|
||||
: handler(sql, params),
|
||||
release() {}
|
||||
};
|
||||
return { query: handler, connect: async () => client };
|
||||
}
|
||||
|
||||
test("pending OCR lookup returns the persisted version and jobs for a duplicate ingestion", async () => {
|
||||
const pool = poolFor(async (sql, params) => {
|
||||
assert.match(sql, /JOIN rag_version_documents/);
|
||||
assert.match(sql, /JOIN rag_ocr_jobs/);
|
||||
assert.match(sql, /source_content_hash IS NULL/);
|
||||
assert.match(sql, /state IN \('pending', 'indexing', 'review_required'\)/);
|
||||
assert.deepEqual(params, ["source-1", "manifest", "fingerprint", "metadata"]);
|
||||
return { rowCount: 1, rows: [{ ...versionRow, ...jobRow }] };
|
||||
});
|
||||
const repository = new CatalogRepository(pool as never);
|
||||
|
||||
const candidate = await repository.findPendingOcrVersion({
|
||||
sourceId: "source-1",
|
||||
originalManifestHash: "manifest",
|
||||
processingFingerprint: "fingerprint",
|
||||
metadataHash: "metadata"
|
||||
});
|
||||
|
||||
assert.equal(candidate?.version.versionId, "version-1");
|
||||
assert.deepEqual(candidate?.jobs.map((job) => [job.jobId, job.remoteJobId, job.remoteIdempotencyKey]), [
|
||||
["job-1", "remote-1", "document-hash:ocr-v1:pages-hash"]
|
||||
]);
|
||||
});
|
||||
|
||||
test("pending OCR lookup returns undefined when only terminal candidates exist", async () => {
|
||||
const repository = new CatalogRepository(poolFor(async () => ({ rowCount: 0, rows: [] })) as never);
|
||||
const candidate = await repository.findPendingOcrVersion({
|
||||
sourceId: "source-1",
|
||||
originalManifestHash: "manifest",
|
||||
processingFingerprint: "fingerprint",
|
||||
metadataHash: "metadata"
|
||||
});
|
||||
assert.equal(candidate, undefined);
|
||||
});
|
||||
|
||||
test("lease claiming uses a locked queue row and keeps its remote identity", async () => {
|
||||
const calls: string[] = [];
|
||||
const pool = poolFor(async (sql, params) => {
|
||||
calls.push(sql);
|
||||
if (sql.includes("FOR UPDATE SKIP LOCKED")) {
|
||||
assert.match(sql, /state = 'queued'/);
|
||||
return { rowCount: 1, rows: [{ job_id: "job-1" }] };
|
||||
}
|
||||
assert.match(sql, /attempt_count = attempt_count \+ 1/);
|
||||
assert.deepEqual(params, ["job-1", "30000"]);
|
||||
return { rowCount: 1, rows: [jobRow] };
|
||||
});
|
||||
const repository = new CatalogRepository(pool as never);
|
||||
|
||||
const claimed = await repository.claimNextOcrJob(30_000);
|
||||
|
||||
assert.equal(calls.length, 2);
|
||||
assert.equal(claimed?.remoteJobId, "remote-1");
|
||||
assert.equal(claimed?.remoteIdempotencyKey, "document-hash:ocr-v1:pages-hash");
|
||||
});
|
||||
|
||||
test("lease claiming returns undefined when no queued job is eligible", async () => {
|
||||
const repository = new CatalogRepository(poolFor(async () => ({ rowCount: 0, rows: [] })) as never);
|
||||
assert.equal(await repository.claimNextOcrJob(30_000), undefined);
|
||||
});
|
||||
|
||||
test("expired leases return to queued without clearing remote recovery identity", async () => {
|
||||
const repository = new CatalogRepository(poolFor(async (sql) => {
|
||||
assert.match(sql, /state = 'running'/);
|
||||
assert.match(sql, /lease_expires_at <= now\(\)/);
|
||||
assert.doesNotMatch(sql, /remote_(job_id|idempotency_key)\s*=/);
|
||||
return { rowCount: 1, rows: [{ ...jobRow, ocr_state: "queued" }] };
|
||||
}) as never);
|
||||
|
||||
const recovered = await repository.recoverExpiredOcrLeases();
|
||||
|
||||
assert.deepEqual(recovered.map((job) => [job.state, job.remoteJobId, job.remoteIdempotencyKey]), [
|
||||
["queued", "remote-1", "document-hash:ocr-v1:pages-hash"]
|
||||
]);
|
||||
});
|
||||
|
||||
test("review transitions enforce indexing, approval, and rejection state guards", async () => {
|
||||
const successfulSql: string[] = [];
|
||||
const success = new CatalogRepository(poolFor(async (sql) => {
|
||||
successfulSql.push(sql);
|
||||
return { rowCount: 1, rows: [] };
|
||||
}) as never);
|
||||
await success.markReviewRequired("version-1");
|
||||
await success.markOcrReviewIndexing("version-1", "reviewer-1");
|
||||
await success.rejectOcrVersion("version-1", "reviewer-1", "Unreadable code");
|
||||
assert.match(successfulSql[0], /state = 'indexing'.*NOT EXISTS/s);
|
||||
assert.match(successfulSql[0], /FROM rag_version_documents/);
|
||||
assert.match(successfulSql[0], /rag_ocr_jobs WHERE version_id = \$1 AND state <> 'succeeded'/);
|
||||
assert.match(successfulSql[1], /state = 'review_required'/);
|
||||
assert.match(successfulSql[2], /state = 'review_required'/);
|
||||
|
||||
const blocked = new CatalogRepository(poolFor(async () => ({ rowCount: 0, rows: [] })) as never);
|
||||
await assert.rejects(
|
||||
blocked.markReviewRequired("version-2"),
|
||||
(error) => error instanceof CatalogError && error.code === "INVALID_VERSION_STATE"
|
||||
);
|
||||
});
|
||||
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
%PDF-1.4
|
||||
1 0 obj
|
||||
<< /Type /Catalog /Pages 2 0 R >>
|
||||
endobj
|
||||
2 0 obj
|
||||
<< /Type /Pages /Kids [4 0 R 5 0 R 6 0 R] /Count 3 >>
|
||||
endobj
|
||||
3 0 obj
|
||||
<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>
|
||||
endobj
|
||||
4 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 7 0 R >>
|
||||
endobj
|
||||
5 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 8 0 R >>
|
||||
endobj
|
||||
6 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 9 0 R >>
|
||||
endobj
|
||||
7 0 obj
|
||||
<< /Length 48 >>
|
||||
stream
|
||||
BT
|
||||
/F1 12 Tf
|
||||
72 720 Td
|
||||
(Native page one.) Tj
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
8 0 obj
|
||||
<< /Length 6 >>
|
||||
stream
|
||||
BT
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
9 0 obj
|
||||
<< /Length 50 >>
|
||||
stream
|
||||
BT
|
||||
/F1 12 Tf
|
||||
72 720 Td
|
||||
(Native page three.) Tj
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
xref
|
||||
0 10
|
||||
0000000000 65535 f
|
||||
0000000009 00000 n
|
||||
0000000058 00000 n
|
||||
0000000127 00000 n
|
||||
0000000197 00000 n
|
||||
0000000323 00000 n
|
||||
0000000449 00000 n
|
||||
0000000575 00000 n
|
||||
0000000672 00000 n
|
||||
0000000726 00000 n
|
||||
trailer
|
||||
<< /Size 10 /Root 1 0 R >>
|
||||
startxref
|
||||
825
|
||||
%%EOF
|
||||
203
tests/ocr/client.test.ts
Normal file
203
tests/ocr/client.test.ts
Normal file
|
|
@ -0,0 +1,203 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { access, lstat, mkdir, mkdtemp, readFile, stat, symlink, utimes } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import { OcrClient, OcrClientError } from "../../src/modules/ocr/client.js";
|
||||
import {
|
||||
resolveArtifactPath,
|
||||
stageOcrArtifacts,
|
||||
sweepOrphanArtifacts
|
||||
} from "../../src/modules/ocr/artifacts.js";
|
||||
import { canonicalJson, hashOrderedPairs, sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
const document = Buffer.from("%PDF-1.4\nunit-7\n%%EOF\n");
|
||||
const documentSha256 = sha256Hex(document);
|
||||
const jobId = "ocr_job-7";
|
||||
|
||||
function jsonResponse(status: number, body: unknown): Response {
|
||||
return new Response(JSON.stringify(body), { status, headers: { "content-type": "application/json" } });
|
||||
}
|
||||
|
||||
function ack(overrides: Record<string, unknown> = {}): Record<string, unknown> {
|
||||
return {
|
||||
jobId,
|
||||
status: "queued",
|
||||
documentSha256,
|
||||
requestedPages: [1, 3],
|
||||
configVersion: "ocr-v1",
|
||||
createdAt: "2026-09-14T10:00:00.000Z",
|
||||
...overrides
|
||||
};
|
||||
}
|
||||
|
||||
function result(overrides: Record<string, unknown> = {}): Record<string, unknown> {
|
||||
return {
|
||||
schemaVersion: "1",
|
||||
jobId,
|
||||
documentSha256,
|
||||
engine: {
|
||||
name: "paddleocr",
|
||||
version: "3.4.0",
|
||||
runtime: "paddlepaddle-3.2.2",
|
||||
device: "cpu",
|
||||
configVersion: "ocr-v1",
|
||||
dpi: 200
|
||||
},
|
||||
pages: [1, 3].map((page) => ({
|
||||
page,
|
||||
width: 1700,
|
||||
height: 2200,
|
||||
processingMs: 25,
|
||||
text: `page ${page}`,
|
||||
metrics: {
|
||||
lineCount: 1,
|
||||
nonWhitespaceCharacters: 5,
|
||||
medianConfidence: 0.95,
|
||||
p10Confidence: 0.95,
|
||||
lowConfidenceLineRatio: 0
|
||||
},
|
||||
lines: [{ lineId: `p${page}-l1`, text: `page ${page}`, confidence: 0.95, bbox: [1, 2, 3, 4] }]
|
||||
})),
|
||||
...overrides
|
||||
};
|
||||
}
|
||||
|
||||
test("runtime retry stub preserves identity, exact attempts, backoff, and result integrity", async () => {
|
||||
const calls: Array<{ url: string; key: string | null }> = [];
|
||||
const delays: number[] = [];
|
||||
const responses: Array<Response | Error> = [
|
||||
new TypeError("connection reset"),
|
||||
jsonResponse(503, { detail: { code: "ENGINE_UNAVAILABLE" } }),
|
||||
jsonResponse(202, ack()),
|
||||
jsonResponse(200, result())
|
||||
];
|
||||
const client = new OcrClient({
|
||||
baseUrl: "http://ocr.internal:8000/",
|
||||
token: "internal-test-token",
|
||||
fetch: (async (input, init) => {
|
||||
calls.push({ url: String(input), key: new Headers(init?.headers).get("Idempotency-Key") });
|
||||
const response = responses.shift();
|
||||
if (response instanceof Error) throw response;
|
||||
return response as Response;
|
||||
}) as typeof fetch,
|
||||
sleep: async (milliseconds) => { delays.push(milliseconds); }
|
||||
});
|
||||
|
||||
const submitted = await client.submit(document, { documentSha256, pages: [1, 3] });
|
||||
const completed = await client.getResult(jobId, { documentSha256, pages: [1, 3] });
|
||||
|
||||
assert.equal(submitted.jobId, jobId);
|
||||
assert.deepEqual(delays, [2_000, 4_000]);
|
||||
assert.equal(calls.filter(({ url }) => url.endsWith("/v1/jobs")).length, 3);
|
||||
assert.equal(new Set(calls.slice(0, 3).map(({ key }) => key)).size, 1);
|
||||
assert.equal(calls[0]?.key, `${documentSha256}:ocr-v1:${sha256Hex("[1,3]")}`);
|
||||
assert.deepEqual(completed.pages.map(({ page }) => page), [1, 3]);
|
||||
});
|
||||
|
||||
test("queue pressure and deterministic failures are surfaced without retries", async () => {
|
||||
for (const [status, retryable] of [[429, true], [422, false]] as const) {
|
||||
let attempts = 0;
|
||||
const client = new OcrClient({
|
||||
baseUrl: "http://ocr.internal:8000",
|
||||
token: "token",
|
||||
fetch: (async () => {
|
||||
attempts += 1;
|
||||
return jsonResponse(status, { detail: { code: status === 429 ? "QUEUE_FULL" : "UNSUPPORTED_PDF" } });
|
||||
}) as typeof fetch,
|
||||
sleep: async () => { throw new Error("must not sleep"); }
|
||||
});
|
||||
|
||||
await assert.rejects(
|
||||
client.submit(document, { documentSha256, pages: [1, 3] }),
|
||||
(error: unknown) => error instanceof OcrClientError && error.status === status && error.retryable === retryable
|
||||
);
|
||||
assert.equal(attempts, 1);
|
||||
}
|
||||
});
|
||||
|
||||
test("strict validation rejects mismatched acknowledgements and result contracts", async () => {
|
||||
for (const response of [
|
||||
jsonResponse(202, ack({ requestedPages: [3, 1] })),
|
||||
jsonResponse(200, result({ schemaVersion: "2" })),
|
||||
jsonResponse(200, result({ pages: [result().pages as unknown] }))
|
||||
]) {
|
||||
const client = new OcrClient({
|
||||
baseUrl: "http://ocr.internal:8000",
|
||||
token: "token",
|
||||
fetch: (async () => response) as typeof fetch,
|
||||
sleep: async () => undefined
|
||||
});
|
||||
const operation = response.status === 202
|
||||
? client.submit(document, { documentSha256, pages: [1, 3] })
|
||||
: client.getResult(jobId, { documentSha256, pages: [1, 3] });
|
||||
await assert.rejects(operation, /OCR response integrity validation failed/);
|
||||
}
|
||||
});
|
||||
|
||||
test("polling uses bounded exponential backoff until a terminal status", async () => {
|
||||
const delays: number[] = [];
|
||||
const states = ["queued", "running", "succeeded"] as const;
|
||||
const client = new OcrClient({
|
||||
baseUrl: "http://ocr.internal:8000",
|
||||
token: "token",
|
||||
fetch: (async () => jsonResponse(200, {
|
||||
jobId,
|
||||
status: states.shift(),
|
||||
completedPages: states.length === 0 ? 2 : 0,
|
||||
totalPages: 2,
|
||||
error: null
|
||||
})) as typeof fetch,
|
||||
sleep: async (milliseconds) => { delays.push(milliseconds); }
|
||||
});
|
||||
|
||||
assert.equal((await client.pollUntilTerminal(jobId)).status, "succeeded");
|
||||
assert.deepEqual(delays, [2_000, 4_000]);
|
||||
});
|
||||
|
||||
test("artifact staging writes private originals and a verifiable canonical manifest", async (context) => {
|
||||
const rootDirectory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-artifacts-"));
|
||||
context.after(async () => { await import("node:fs/promises").then(({ rm }) => rm(rootDirectory, { recursive: true, force: true })); });
|
||||
const versionId = "11111111-1111-4111-8111-111111111111";
|
||||
const staged = await stageOcrArtifacts({
|
||||
rootDirectory,
|
||||
versionId,
|
||||
createdAt: "2026-09-14T10:00:00.000Z",
|
||||
documents: [
|
||||
{ documentId: "doc:source:b", documentKey: "b.pdf", bytes: Buffer.from("%PDF-b") },
|
||||
{ documentId: "doc:source:a", documentKey: "a.pdf", bytes: Buffer.from("%PDF-a") }
|
||||
]
|
||||
});
|
||||
const persisted = JSON.parse(await readFile(staged.manifestPath, "utf8"));
|
||||
|
||||
assert.equal(staged.originalManifestHash, hashOrderedPairs([["b.pdf", sha256Hex("%PDF-b")], ["a.pdf", sha256Hex("%PDF-a")]]));
|
||||
assert.equal(staged.manifestSha256, sha256Hex(canonicalJson(persisted)));
|
||||
assert.deepEqual(persisted.documents.map((entry: { documentKey: string }) => entry.documentKey), ["a.pdf", "b.pdf"]);
|
||||
for (const entry of persisted.documents) {
|
||||
const originalPath = resolveArtifactPath(staged.versionDirectory, entry.originalPath);
|
||||
assert.equal((await stat(originalPath)).mode & 0o777, 0o600);
|
||||
assert.equal(sha256Hex(await readFile(originalPath)), entry.originalSha256);
|
||||
}
|
||||
assert.equal((await stat(staged.manifestPath)).mode & 0o777, 0o600);
|
||||
});
|
||||
|
||||
test("safe resolution blocks traversal and orphan sweep preserves retained, recent, and non-directory entries", async (context) => {
|
||||
const rootDirectory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-sweep-"));
|
||||
context.after(async () => { await import("node:fs/promises").then(({ rm }) => rm(rootDirectory, { recursive: true, force: true })); });
|
||||
const retained = "22222222-2222-4222-8222-222222222222";
|
||||
const orphan = "33333333-3333-4333-8333-333333333333";
|
||||
const recent = "44444444-4444-4444-8444-444444444444";
|
||||
await Promise.all([retained, orphan, recent].map((id) => mkdir(path.join(rootDirectory, id))));
|
||||
await utimes(path.join(rootDirectory, orphan), new Date(0), new Date(0));
|
||||
await symlink(path.join(rootDirectory, orphan), path.join(rootDirectory, "55555555-5555-4555-8555-555555555555"));
|
||||
|
||||
assert.throws(() => resolveArtifactPath(path.join(rootDirectory, retained), "../manifest.json"), /escapes version directory/);
|
||||
assert.deepEqual(await sweepOrphanArtifacts({
|
||||
rootDirectory,
|
||||
retainedVersionIds: new Set([retained]),
|
||||
olderThan: new Date("2026-09-14T09:00:00.000Z")
|
||||
}), [orphan]);
|
||||
await assert.rejects(access(path.join(rootDirectory, orphan)));
|
||||
await Promise.all([retained, recent].map((id) => access(path.join(rootDirectory, id))));
|
||||
assert.equal((await lstat(path.join(rootDirectory, "55555555-5555-4555-8555-555555555555"))).isSymbolicLink(), true);
|
||||
});
|
||||
71
tests/ocr/contracts-deploy.test.ts
Normal file
71
tests/ocr/contracts-deploy.test.ts
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { readFile } from "node:fs/promises";
|
||||
import { buildApiHelp, openApiDocument } from "../../src/api/openapi.js";
|
||||
import { env } from "../../src/config/env.js";
|
||||
|
||||
const api = openApiDocument as unknown as {
|
||||
paths: Record<string, Record<string, Record<string, unknown>>>;
|
||||
components: { schemas: Record<string, Record<string, unknown>> };
|
||||
};
|
||||
|
||||
test("OpenAPI describes authenticated OCR ingestion, status, review, approval, and rejection", () => {
|
||||
const ingestResponses = api.paths["/ingest"]!.post!.responses as Record<string, { content?: Record<string, { schema?: unknown }> }>;
|
||||
const uploadResponses = api.paths["/ingest/upload"]!.post!.responses as typeof ingestResponses;
|
||||
assert.deepEqual(ingestResponses["202"]!.content!["application/json"]!.schema, { oneOf: [{ $ref: "#/components/schemas/OcrAccepted" }, { $ref: "#/components/schemas/IngestResponse" }] });
|
||||
assert.deepEqual(uploadResponses["202"]!.content!["application/json"]!.schema, { oneOf: [{ $ref: "#/components/schemas/OcrUploadAccepted" }, { $ref: "#/components/schemas/UploadIngestResponse" }] });
|
||||
|
||||
for (const [path, method] of [
|
||||
["/ingestions/{versionId}", "get"],
|
||||
["/ingestions/{versionId}/review", "get"],
|
||||
["/ingestions/{versionId}/approve", "post"],
|
||||
["/ingestions/{versionId}/reject", "post"]
|
||||
]) {
|
||||
const operation = api.paths[path]![method]!;
|
||||
assert.deepEqual(operation.security, [{ bearerAuth: [] }], `${method.toUpperCase()} ${path} must require bearer auth`);
|
||||
const responses = operation.responses as Record<string, unknown>;
|
||||
assert.ok("200" in responses && "401" in responses && "404" in responses && "503" in responses);
|
||||
}
|
||||
|
||||
const approve = api.paths["/ingestions/{versionId}/approve"]!.post!;
|
||||
const reject = api.paths["/ingestions/{versionId}/reject"]!.post!;
|
||||
assert.ok("409" in (approve.responses as Record<string, unknown>));
|
||||
assert.ok("400" in (reject.responses as Record<string, unknown>));
|
||||
assert.ok("409" in (reject.responses as Record<string, unknown>));
|
||||
assert.deepEqual((approve.requestBody as { content: Record<string, { schema: unknown }> }).content["application/json"]!.schema,
|
||||
{ $ref: "#/components/schemas/OcrApprovalRequest" });
|
||||
assert.deepEqual((reject.requestBody as { content: Record<string, { schema: unknown }> }).content["application/json"]!.schema,
|
||||
{ $ref: "#/components/schemas/OcrRejectionRequest" });
|
||||
assert.match(buildApiHelp().authentication, /Bearer.*OCR/u);
|
||||
});
|
||||
|
||||
test("OpenAPI OCR schemas preserve progress, review evidence, corrections, and decision states", () => {
|
||||
const schemas = api.components.schemas;
|
||||
assert.deepEqual(schemas.OcrPhase!.enum, ["native_extracting", "ocr_queued", "ocr_running", "review_required", "indexing", "ready", "active", "failed", "rejected"]);
|
||||
assert.deepEqual(schemas.OcrAccepted!.required, ["accepted", "sourceId", "versionId", "versionNumber", "state", "phase", "statusUrl", "reviewUrl", "activated"]);
|
||||
assert.deepEqual(schemas.IngestionStatus!.required, ["sourceId", "versionId", "state", "phase", "activated", "documents", "error", "statusUrl", "reviewUrl"]);
|
||||
assert.deepEqual(schemas.OcrReview!.required, ["versionId", "sourceId", "state", "candidateSha256", "baseActiveVersionId", "currentActiveVersionId", "activateRequested", "processingFingerprint", "metadataHash", "documents"]);
|
||||
const approvalProperties = schemas.OcrApprovalRequest!.properties as Record<string, { items?: unknown }>;
|
||||
assert.deepEqual(approvalProperties.corrections!.items, { $ref: "#/components/schemas/OcrCorrection" });
|
||||
assert.deepEqual(schemas.OcrDecisionResponse!.properties, {
|
||||
versionId: { type: "string", format: "uuid" }, state: { type: "string", enum: ["ready", "active", "rejected"] }, activated: { type: "boolean" },
|
||||
activatedVersionId: { type: "string", format: "uuid" }, errorCode: { type: "string", enum: ["DUPLICATE_REUSABLE_VERSION"] }
|
||||
});
|
||||
});
|
||||
|
||||
test("OCR deployment defaults and container wiring match the private durable contract", async () => {
|
||||
assert.deepEqual({
|
||||
root: env.ocrArtifactRoot,
|
||||
uploadBytes: (env as unknown as Record<string, unknown>).ocrMaxUploadBytes,
|
||||
pages: (env as unknown as Record<string, unknown>).ocrMaxPages,
|
||||
pageTimeoutMs: (env as unknown as Record<string, unknown>).ocrPageTimeoutMs,
|
||||
totalTimeoutMs: (env as unknown as Record<string, unknown>).ocrTotalTimeoutMs
|
||||
}, { root: "/data/ingestions", uploadBytes: 50 * 1024 * 1024, pages: 100, pageTimeoutMs: 60_000, totalTimeoutMs: 15 * 60_000 });
|
||||
|
||||
const dockerfile = await readFile(new URL("../../Dockerfile", import.meta.url), "utf8");
|
||||
assert.match(dockerfile, /VOLUME \["\/data\/ingestions"\]/u);
|
||||
assert.match(dockerfile, /ENV NODE_ENV=production OCR_ARTIFACT_ROOT=\/data\/ingestions/u);
|
||||
assert.match(dockerfile, /USER node/u);
|
||||
assert.match(dockerfile, /dist\/modules\/catalog\/migrations\.js.*dist\/server\.js/u);
|
||||
assert.doesNotMatch(dockerfile, /(?:COPY|ADD)\s+\.env|ENV\s+(?:OCR_INTERNAL_TOKEN|LIFECYCLE_ADMIN_TOKEN)=/u);
|
||||
});
|
||||
138
tests/ocr/detection.test.ts
Normal file
138
tests/ocr/detection.test.ts
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import { isSupportedDocument, parsePdfPages } from "../../src/modules/parsers/parser-registry.js";
|
||||
import {
|
||||
DETECTION_POLICY_VERSION,
|
||||
classifyOcrPage,
|
||||
computeNativeMetrics,
|
||||
isNativeTextSufficient,
|
||||
selectPdfPagesForOcr
|
||||
} from "../../src/modules/ocr/detection.js";
|
||||
import { composeCandidate, prioritizeRiskTokens } from "../../src/modules/ocr/composition.js";
|
||||
import { canonicalJson, sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
function buildPdf(pageTexts: string[]): Buffer {
|
||||
const fontId = 3 + pageTexts.length * 2;
|
||||
const objects = [
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
`<< /Type /Pages /Kids [${pageTexts.map((_, index) => `${3 + index * 2} 0 R`).join(" ")}] /Count ${pageTexts.length} >>`
|
||||
];
|
||||
for (const [index, text] of pageTexts.entries()) {
|
||||
const pageId = 3 + index * 2;
|
||||
const contentId = pageId + 1;
|
||||
const escaped = text.replace(/([\\()])/g, "\\$1");
|
||||
const stream = text ? `BT /F1 10 Tf 40 760 Td (${escaped}) Tj ET` : "";
|
||||
objects.push(
|
||||
`<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 ${fontId} 0 R >> >> /Contents ${contentId} 0 R >>`,
|
||||
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`
|
||||
);
|
||||
}
|
||||
objects.push("<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>");
|
||||
|
||||
let pdf = "%PDF-1.4\n";
|
||||
const offsets = [0];
|
||||
objects.forEach((object, index) => {
|
||||
offsets.push(Buffer.byteLength(pdf));
|
||||
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||
});
|
||||
const xref = Buffer.byteLength(pdf);
|
||||
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
|
||||
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
|
||||
return Buffer.from(pdf);
|
||||
}
|
||||
|
||||
const sufficientText = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
|
||||
|
||||
test("native detection applies every pdf-detection-v1 boundary per page", () => {
|
||||
const boundary = { nonWhitespaceCharacters: 120, alphanumericCharacters: 80, wordCount: 20, replacementControlRatio: 0.01 };
|
||||
|
||||
assert.equal(DETECTION_POLICY_VERSION, "pdf-detection-v1");
|
||||
assert.deepEqual(
|
||||
[boundary, { ...boundary, nonWhitespaceCharacters: 119 }, { ...boundary, alphanumericCharacters: 79 }, { ...boundary, wordCount: 19 }, { ...boundary, replacementControlRatio: 0.0101 }]
|
||||
.map(isNativeTextSufficient),
|
||||
[true, false, false, false, false]
|
||||
);
|
||||
assert.deepEqual(computeNativeMetrics("Árbol 12\nword<72>\u0001"), {
|
||||
nonWhitespaceCharacters: 12,
|
||||
alphanumericCharacters: 10,
|
||||
wordCount: 3,
|
||||
replacementControlRatio: 2 / 12
|
||||
});
|
||||
});
|
||||
|
||||
test("a parsed mixed PDF selects only unique ordered insufficient pages and non-PDFs never route to OCR", async () => {
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-mixed-pdf-"));
|
||||
const filePath = path.join(directory, "mixed.PDF");
|
||||
try {
|
||||
await writeFile(filePath, buildPdf([sufficientText, "", "short scanned proxy"]));
|
||||
const pages = await parsePdfPages(filePath);
|
||||
|
||||
assert.deepEqual(selectPdfPagesForOcr(filePath, [...pages].reverse()), [2, 3]);
|
||||
for (const nonPdf of ["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"]) {
|
||||
assert.deepEqual(selectPdfPagesForOcr(nonPdf, pages), []);
|
||||
}
|
||||
assert.deepEqual(["requirements.txt", "executable.md", "CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument), [true, true, false, false, false]);
|
||||
} finally {
|
||||
await rm(directory, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("OCR blank and quality gates fail closed at exact thresholds", () => {
|
||||
const passing = { nonWhitespaceCharacters: 40, medianConfidence: 0.8, p10Confidence: 0.5, lowConfidenceLineRatio: 0.2 };
|
||||
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.0149, metrics: { ...passing, nonWhitespaceCharacters: 9 } }), { method: "blank" });
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.015, metrics: { ...passing, nonWhitespaceCharacters: 0 } }), { method: "blocked", errorCode: "OCR_QUALITY_BLOCKED" });
|
||||
assert.deepEqual(classifyOcrPage({ inkCoverage: 0.5, metrics: passing }), { method: "ocr" });
|
||||
for (const metrics of [
|
||||
{ ...passing, nonWhitespaceCharacters: 39 },
|
||||
{ ...passing, medianConfidence: 0.799 },
|
||||
{ ...passing, p10Confidence: 0.499 },
|
||||
{ ...passing, lowConfidenceLineRatio: 0.201 }
|
||||
]) {
|
||||
assert.equal(classifyOcrPage({ inkCoverage: 0.5, metrics }).method, "blocked");
|
||||
}
|
||||
});
|
||||
|
||||
test("candidate composition orders pages and OCR lines, omits blanks, and hashes canonical records", () => {
|
||||
const input = [
|
||||
{
|
||||
page: 3,
|
||||
method: "ocr" as const,
|
||||
nativeText: "codigo CBG04a FAT07 DSAU08 NSAV06",
|
||||
rawOcrText: "raw service text",
|
||||
lines: [
|
||||
{ lineId: "line-fat", text: "FATo7", confidence: 0.98, bbox: [20, 50, 30, 60] as [number, number, number, number] },
|
||||
{ lineId: "line-code", text: "codigo CBGO4a", confidence: 0.95, bbox: [10, 50, 15, 60] as [number, number, number, number] },
|
||||
{ lineId: "line-other", text: "DSAuo8 NSAvo6", confidence: 0.9, bbox: [10, 50, 18, 60] as [number, number, number, number] }
|
||||
]
|
||||
},
|
||||
{ page: 1, method: "native" as const, nativeText: "Native first page.", rawOcrText: "ignored OCR", lines: [] },
|
||||
{ page: 2, method: "blank" as const, nativeText: "", rawOcrText: "", lines: [] }
|
||||
];
|
||||
const first = composeCandidate(input);
|
||||
const second = composeCandidate(structuredClone(input));
|
||||
|
||||
assert.equal(first.text, "Native first page.\n\n--- Page 3 ---\n\ncodigo CBGO4a\nDSAuo8 NSAvo6\nFATo7");
|
||||
assert.equal(first.textSha256, sha256Hex(first.text));
|
||||
assert.deepEqual(first, second);
|
||||
assert.deepEqual(first.pages.map(({ page, candidateText, risks }) => ({ page, candidateText, risks })), [
|
||||
{ page: 1, candidateText: "Native first page.", risks: [] },
|
||||
{ page: 2, candidateText: "", risks: [] },
|
||||
{ page: 3, candidateText: "codigo CBGO4a\nDSAuo8 NSAvo6\nFATo7", risks: ["CBGO4a", "DSAuo8", "NSAvo6", "FATo7"] }
|
||||
]);
|
||||
assert.equal(first.candidateSha256, sha256Hex(canonicalJson(first.pages)));
|
||||
assert.equal(first.pages[2]?.lines[0]?.lineSha256, sha256Hex("codigo CBGO4a"));
|
||||
assert.equal(first.pages[0]?.rawOcrText, "ignored OCR");
|
||||
assert.equal(first.pages[2]?.nativeText, "codigo CBG04a FAT07 DSAU08 NSAV06");
|
||||
});
|
||||
|
||||
test("risk tokens are preserved unchanged and elevated by ambiguity, difference, uniqueness, and context", () => {
|
||||
assert.deepEqual(
|
||||
prioritizeRiskTokens("AB12 AB12 codigo CBGO4a FATo7 DSAuo8 NSAvo6", "AB12 CBG04a FAT07 DSAU08 NSAV06"),
|
||||
["CBGO4a", "FATo7", "DSAuo8", "NSAvo6", "AB12"]
|
||||
);
|
||||
});
|
||||
397
tests/ocr/dispatcher.test.ts
Normal file
397
tests/ocr/dispatcher.test.ts
Normal file
|
|
@ -0,0 +1,397 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import { createApp } from "../../src/app.js";
|
||||
import { env } from "../../src/config/env.js";
|
||||
import { CatalogError } from "../../src/modules/catalog/errors.js";
|
||||
import { CatalogRepository } from "../../src/modules/catalog/repository.js";
|
||||
import { KnowledgeLifecycleReconciler } from "../../src/modules/catalog/reconciler.js";
|
||||
import { OcrDispatcher } from "../../src/modules/ocr/dispatcher.js";
|
||||
import { IngestService } from "../../src/modules/ingest/service.js";
|
||||
import type { EmbeddingProvider } from "../../src/modules/embeddings/provider.js";
|
||||
import type { VectorStoreClient } from "../../src/modules/vectorstore/client.js";
|
||||
import { sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
function setEnvFlag(name: keyof typeof env, value: unknown): void {
|
||||
(env as unknown as Record<string, unknown>)[name] = value;
|
||||
}
|
||||
|
||||
function buildPdf(text: string): Buffer {
|
||||
const stream = text ? `BT /F1 10 Tf 40 760 Td (${text.replace(/([\\()])/g, "\\$1")}) Tj ET` : "";
|
||||
const objects = [
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
`<< /Length ${Buffer.byteLength(stream)} >>\nstream\n${stream}\nendstream`,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>"
|
||||
];
|
||||
let pdf = "%PDF-1.4\n";
|
||||
const offsets = [0];
|
||||
objects.forEach((object, index) => {
|
||||
offsets.push(Buffer.byteLength(pdf));
|
||||
pdf += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||
});
|
||||
const xref = Buffer.byteLength(pdf);
|
||||
pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||
pdf += offsets.slice(1).map((offset) => `${String(offset).padStart(10, "0")} 00000 n \n`).join("");
|
||||
pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`;
|
||||
return Buffer.from(pdf);
|
||||
}
|
||||
|
||||
function embeddingProvider(): EmbeddingProvider {
|
||||
return {
|
||||
providerName: "test-provider",
|
||||
modelName: "test-model",
|
||||
dimensions: 3,
|
||||
async embed(input) { return input.map(() => [0.1, 0.2, 0.3]); }
|
||||
};
|
||||
}
|
||||
|
||||
function vectorStore(): VectorStoreClient {
|
||||
return {
|
||||
kind: "fake",
|
||||
async upsert() {},
|
||||
async countVersionPoints() { return 0; }
|
||||
} as VectorStoreClient;
|
||||
}
|
||||
|
||||
test("OCR-disabled textual PDFs retain the native synchronous path", async (context) => {
|
||||
const previousLifecycle = env.knowledgeLifecycleEnforced;
|
||||
setEnvFlag("knowledgeLifecycleEnforced", true);
|
||||
context.after(() => setEnvFlag("knowledgeLifecycleEnforced", previousLifecycle));
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-disabled-"));
|
||||
context.after(() => rm(directory, { recursive: true, force: true }));
|
||||
const filePath = path.join(directory, "native.pdf");
|
||||
const sufficient = Array.from({ length: 24 }, (_, index) => `Alpha${index} beta${index}`).join(" ");
|
||||
await writeFile(filePath, buildPdf(sufficient));
|
||||
const calls: string[] = [];
|
||||
const catalog = {
|
||||
async beginAttempt() { return "attempt-1"; },
|
||||
async updateAttempt() {},
|
||||
async withSourceLock(_sourceId: string, handler: () => Promise<unknown>) { return handler(); },
|
||||
async findReusableVersion() { return { versionId: "native-1", versionNumber: 1, state: "active", previousVersionId: null, expectedDocumentCount: 1 }; },
|
||||
async assertActiveVersionPrecondition() { calls.push("native"); }
|
||||
};
|
||||
const service = new IngestService(embeddingProvider(), vectorStore(), catalog as never, {
|
||||
enabled: false,
|
||||
artifactRoot: directory
|
||||
});
|
||||
|
||||
const result = await service.ingest({ sourceType: "file", sourceRef: "native.pdf", readPath: filePath, activate: true, expectedActiveVersionId: null });
|
||||
|
||||
assert.equal(result.state, "active");
|
||||
assert.equal("phase" in result, false);
|
||||
assert.deepEqual(calls, ["native"]);
|
||||
});
|
||||
|
||||
test("OCR-required PDFs are durably accepted once and duplicate pending ingestion reuses status identity", async (context) => {
|
||||
const previousLifecycle = env.knowledgeLifecycleEnforced;
|
||||
setEnvFlag("knowledgeLifecycleEnforced", true);
|
||||
context.after(() => setEnvFlag("knowledgeLifecycleEnforced", previousLifecycle));
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-accept-"));
|
||||
context.after(() => rm(directory, { recursive: true, force: true }));
|
||||
const filePath = path.join(directory, "scanned.pdf");
|
||||
await writeFile(filePath, buildPdf(""));
|
||||
let createdInput: Record<string, unknown> | undefined;
|
||||
let pending: Record<string, unknown> | undefined;
|
||||
const catalog = {
|
||||
async beginAttempt() { return "attempt-1"; },
|
||||
async updateAttempt() {},
|
||||
async withSourceLock(_sourceId: string, handler: () => Promise<unknown>) { return handler(); },
|
||||
async findPendingOcrVersion() { return pending; },
|
||||
async createPendingVersion(input: Record<string, unknown>) {
|
||||
createdInput = input;
|
||||
pending = { version: { versionId: input.versionId, versionNumber: 4, state: "indexing" }, jobs: [{ jobId: "job-1" }] };
|
||||
return (pending as { version: unknown }).version;
|
||||
},
|
||||
async markIndexing() {}
|
||||
};
|
||||
const service = new IngestService(embeddingProvider(), vectorStore(), catalog as never, {
|
||||
enabled: true,
|
||||
artifactRoot: path.join(directory, "artifacts")
|
||||
});
|
||||
const source = { sourceId: "src:scan", sourceType: "file" as const, sourceRef: "scanned.pdf", readPath: filePath, activate: true, expectedActiveVersionId: null };
|
||||
|
||||
const accepted = await service.ingest(source);
|
||||
const duplicate = await service.ingest(source);
|
||||
|
||||
assert.deepEqual(accepted, {
|
||||
accepted: true,
|
||||
sourceId: "src:scan",
|
||||
versionId: accepted.versionId,
|
||||
versionNumber: 4,
|
||||
state: "indexing",
|
||||
phase: "ocr_queued",
|
||||
statusUrl: `/ingestions/${accepted.versionId}`,
|
||||
reviewUrl: null,
|
||||
activated: false
|
||||
});
|
||||
assert.equal(duplicate.versionId, accepted.versionId);
|
||||
assert.equal((createdInput?.sourceContentHash), null);
|
||||
assert.deepEqual((createdInput?.ocrJobs as Array<{ requestedPages: number[] }>)[0]?.requestedPages, [1]);
|
||||
const originalPath = path.join(directory, "artifacts", String(accepted.versionId), "documents");
|
||||
assert.equal((await readFile(path.join(directory, "artifacts", String(accepted.versionId), "manifest.json"), "utf8")).includes("scanned.pdf"), true);
|
||||
assert.equal(path.isAbsolute(originalPath), true);
|
||||
});
|
||||
|
||||
test("OCR acceptance retains every supported native document and reloads a concurrent winner", async (context) => {
|
||||
const previousLifecycle = env.knowledgeLifecycleEnforced;
|
||||
setEnvFlag("knowledgeLifecycleEnforced", true);
|
||||
context.after(() => setEnvFlag("knowledgeLifecycleEnforced", previousLifecycle));
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-multi-"));
|
||||
context.after(() => rm(directory, { recursive: true, force: true }));
|
||||
await writeFile(path.join(directory, "scanned.pdf"), buildPdf(""));
|
||||
await writeFile(path.join(directory, "notes.txt"), "Native companion document\n");
|
||||
const winner = { version: { versionId: "22222222-2222-4222-8222-222222222222", versionNumber: 7, state: "indexing" }, jobs: [{ jobId: "winner-job" }] };
|
||||
let lookupCount = 0;
|
||||
let candidateInput: Record<string, unknown> | undefined;
|
||||
const catalog = {
|
||||
async beginAttempt() { return "attempt-multi"; },
|
||||
async updateAttempt() {},
|
||||
async withSourceLock(_sourceId: string, handler: () => Promise<unknown>) { return handler(); },
|
||||
async findPendingOcrVersion() { lookupCount += 1; return lookupCount === 1 ? undefined : winner; },
|
||||
async createPendingVersion(input: Record<string, unknown>) {
|
||||
candidateInput = input;
|
||||
throw Object.assign(new Error("duplicate pending identity"), { code: "23505" });
|
||||
}
|
||||
};
|
||||
const service = new IngestService(embeddingProvider(), vectorStore(), catalog as never, { enabled: true, artifactRoot: path.join(directory, "artifacts") });
|
||||
|
||||
const accepted = await service.ingest({ sourceId: "src:multi", sourceType: "folder", sourceRef: "bundle", readPath: directory, activate: true, expectedActiveVersionId: null });
|
||||
|
||||
assert.equal(accepted.versionId, winner.version.versionId);
|
||||
assert.equal(candidateInput?.expectedDocumentCount, 2);
|
||||
assert.deepEqual((candidateInput?.documents as Array<{ documentKey: string }>).map(({ documentKey }) => documentKey), ["notes.txt", "scanned.pdf"]);
|
||||
assert.equal((candidateInput?.ocrJobs as unknown[]).length, 1);
|
||||
});
|
||||
|
||||
test("application wiring dispatches a newly accepted OCR job from durable artifacts", async (context) => {
|
||||
const previous = {
|
||||
lifecycle: env.knowledgeLifecycleEnforced,
|
||||
enabled: (env as unknown as Record<string, unknown>).ocrIngestEnabled,
|
||||
root: (env as unknown as Record<string, unknown>).ocrArtifactRoot
|
||||
};
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-ocr-runtime-"));
|
||||
context.after(() => rm(directory, { recursive: true, force: true }));
|
||||
setEnvFlag("knowledgeLifecycleEnforced", true);
|
||||
setEnvFlag("ocrIngestEnabled" as keyof typeof env, true);
|
||||
setEnvFlag("ocrArtifactRoot" as keyof typeof env, path.join(directory, "artifacts"));
|
||||
context.after(() => {
|
||||
setEnvFlag("knowledgeLifecycleEnforced", previous.lifecycle);
|
||||
setEnvFlag("ocrIngestEnabled" as keyof typeof env, previous.enabled);
|
||||
setEnvFlag("ocrArtifactRoot" as keyof typeof env, previous.root);
|
||||
});
|
||||
const pdf = buildPdf("");
|
||||
let queuedJob: Record<string, unknown> | undefined;
|
||||
let persistedKey: unknown;
|
||||
const dispatchCalls: string[] = [];
|
||||
const catalog = {
|
||||
async beginAttempt() { return "runtime-attempt"; }, async updateAttempt() {},
|
||||
async withSourceLock(_sourceId: string, handler: () => Promise<unknown>) { return handler(); },
|
||||
async findPendingOcrVersion() { return undefined; },
|
||||
async createPendingVersion(input: Record<string, unknown>) {
|
||||
const job = (input.ocrJobs as Array<Record<string, unknown>>)[0]!;
|
||||
persistedKey = job.remoteIdempotencyKey;
|
||||
queuedJob = { ...job, jobId: "runtime-job", versionId: input.versionId, remoteJobId: null, state: "queued", completedPages: 0, attemptCount: 0, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null };
|
||||
return { versionId: input.versionId, versionNumber: 3 };
|
||||
},
|
||||
async markIndexing() {},
|
||||
async claimNextOcrJob() { const job = queuedJob; queuedJob = undefined; return job ? { ...job, state: "running", attemptCount: 1 } : undefined; },
|
||||
async setOcrRemoteJob() {},
|
||||
async requeueOcrJob() { dispatchCalls.push("requeued"); },
|
||||
async completeOcrJob() { return false; }, async failOcrJob() {}, async markReviewRequired() {}, async markFailed() {}
|
||||
};
|
||||
const client = {
|
||||
async submit(bytes: Buffer, _expected: unknown, key: string) { assert.equal(key, persistedKey); dispatchCalls.push(`${bytes.length}:${key}`); return { jobId: "runtime-remote" }; },
|
||||
async getStatus() { return { jobId: "runtime-remote", status: "queued", completedPages: 0, totalPages: 1, error: null }; }
|
||||
};
|
||||
const server = createApp({ catalog: catalog as never, ocrClient: client as never, startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
|
||||
const form = new FormData();
|
||||
form.set("file", new Blob([new Uint8Array(pdf)], { type: "application/pdf" }), "runtime.pdf");
|
||||
form.set("sourceId", "src:runtime");
|
||||
form.set("activate", "true");
|
||||
form.set("expectedActiveVersionId", "null");
|
||||
const response = await fetch(`http://127.0.0.1:${address.port}/ingest/upload`, { method: "POST", body: form });
|
||||
assert.equal(response.status, 202);
|
||||
assert.equal((await response.json() as { uploadedResource: string }).uploadedResource, "runtime.pdf");
|
||||
for (let attempt = 0; attempt < 20 && !dispatchCalls.includes("requeued"); attempt += 1) await new Promise((resolve) => setTimeout(resolve, 5));
|
||||
assert.equal(dispatchCalls.length, 2);
|
||||
assert.match(dispatchCalls[0]!, /^\d+:.+:ocr-v1:.+$/u);
|
||||
assert.equal(dispatchCalls[1], "requeued");
|
||||
});
|
||||
|
||||
test("dispatcher recovers only expired work, reuses its remote job, and completes it once", async () => {
|
||||
const calls: string[] = [];
|
||||
const job = {
|
||||
jobId: "job-expired",
|
||||
versionId: "version-1",
|
||||
documentId: "document-1",
|
||||
remoteJobId: "remote-1",
|
||||
remoteIdempotencyKey: "persisted-key",
|
||||
state: "queued" as const,
|
||||
requestedPages: [1],
|
||||
completedPages: 0,
|
||||
configVersion: "ocr-v1",
|
||||
attemptCount: 1,
|
||||
heartbeatAt: null,
|
||||
leaseExpiresAt: null,
|
||||
nextAttemptAt: null,
|
||||
errorCode: null,
|
||||
errorDetail: null
|
||||
};
|
||||
let available = true;
|
||||
const repository = {
|
||||
async recoverExpiredOcrLeases() { calls.push("recover"); return [job]; },
|
||||
async claimNextOcrJob() { throw new Error("recovery must not claim unrelated queued work"); },
|
||||
async claimOcrJob(jobId: string) { assert.equal(jobId, job.jobId); if (!available) return undefined; available = false; calls.push("claim-exact"); return { ...job, state: "running" as const }; },
|
||||
async setOcrRemoteJob() { calls.push("submit-persist"); },
|
||||
async requeueOcrJob() { calls.push("requeue"); },
|
||||
async completeOcrJob() { calls.push("complete"); return true; },
|
||||
async failOcrJob() { calls.push("job-failed"); },
|
||||
async markReviewRequired() { calls.push("review-required"); },
|
||||
async markFailed() { calls.push("version-failed"); }
|
||||
};
|
||||
const client = {
|
||||
async submit() { calls.push("submit"); throw new Error("existing remote work must not be duplicated"); },
|
||||
async getStatus() { calls.push("status"); return { jobId: "remote-1", status: "succeeded", completedPages: 1, totalPages: 1, error: null }; },
|
||||
async getResult() { calls.push("result"); return { pages: [{ page: 1 }] }; }
|
||||
};
|
||||
const dispatcher = new OcrDispatcher(repository, client as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }));
|
||||
|
||||
assert.equal(await dispatcher.recoverExpiredLeases(), 1);
|
||||
assert.deepEqual(calls, ["recover", "claim-exact", "status", "result", "complete", "review-required"]);
|
||||
});
|
||||
|
||||
test("OCR exhaustion fails the leased candidate without activation or engine substitution", async () => {
|
||||
const calls: string[] = [];
|
||||
const repository = {
|
||||
async claimNextOcrJob() {
|
||||
return { jobId: "job-1", versionId: "version-1", documentId: "document-1", remoteJobId: null, remoteIdempotencyKey: "same-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null };
|
||||
},
|
||||
async setOcrRemoteJob() { calls.push("remote"); },
|
||||
async requeueOcrJob() { calls.push("requeue"); },
|
||||
async completeOcrJob() { calls.push("complete"); return false; },
|
||||
async failOcrJob(_jobId: string, code: string) { calls.push(`job:${code}`); },
|
||||
async markReviewRequired() { calls.push("review"); },
|
||||
async markFailed(_versionId: string, code: string) { calls.push(`version:${code}`); }
|
||||
};
|
||||
const client = { async submit() { throw new Error("OCR unavailable after retries"); } };
|
||||
const dispatcher = new OcrDispatcher(repository as never, client as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }));
|
||||
|
||||
assert.equal(await dispatcher.runOnce(), "failed");
|
||||
assert.deepEqual(calls, ["job:OCR_DISPATCH_FAILED", "version:OCR_DISPATCH_FAILED"]);
|
||||
});
|
||||
|
||||
test("reconciler makes expired OCR leases dispatchable while leaving live leases to PostgreSQL", async () => {
|
||||
const calls: string[] = [];
|
||||
const catalog = {
|
||||
async withGlobalTryLock(_name: string, handler: () => Promise<unknown>) { return handler(); },
|
||||
async resolveActiveVersions() { return []; },
|
||||
async validateActiveInvariant() { return []; },
|
||||
async listOrphanedIndexingCandidates() { return []; }
|
||||
};
|
||||
const dispatcher = {
|
||||
async recoverExpiredLeases() { calls.push("ocr-recovery"); return 1; },
|
||||
async dispatchAvailable() { calls.push("ocr-dispatch"); return 2; }
|
||||
};
|
||||
const reconciler = new KnowledgeLifecycleReconciler(catalog as never, vectorStore(), dispatcher as never);
|
||||
|
||||
const status = await reconciler.runOnce();
|
||||
|
||||
assert.equal(status.ocrLeasesRecovered, 1);
|
||||
assert.deepEqual(calls, ["ocr-recovery", "ocr-dispatch"]);
|
||||
});
|
||||
|
||||
test("runtime HTTP routing returns native 201, OCR 202/status, and catalog-down 503", async () => {
|
||||
const previous = { knowledgeLifecycleEnforced: env.knowledgeLifecycleEnforced, lifecycleAdminToken: env.lifecycleAdminToken, ocrIngestEnabled: env.ocrIngestEnabled };
|
||||
setEnvFlag("knowledgeLifecycleEnforced", true);
|
||||
setEnvFlag("lifecycleAdminToken", "unit-8-token");
|
||||
setEnvFlag("ocrIngestEnabled", true);
|
||||
const versionId = "11111111-1111-4111-8111-111111111111";
|
||||
const ingestService = {
|
||||
async ingest(input: Record<string, unknown>) {
|
||||
return input.sourceRef === "native.pdf"
|
||||
? { accepted: true, sourceId: "src:native", versionId: "native-1", versionNumber: 1, state: "active", filesDiscovered: 1, documentsProcessed: 1, chunksStored: 1, activated: true, noOp: false, collectionName: "rag" }
|
||||
: { accepted: true, sourceId: "src:scan", versionId, versionNumber: 2, state: "indexing", phase: "ocr_queued", statusUrl: `/ingestions/${versionId}`, reviewUrl: null, activated: false };
|
||||
},
|
||||
async cleanup() { return { deleted: 0 }; }
|
||||
};
|
||||
const catalog = {
|
||||
async getIngestionStatus(id: string) {
|
||||
assert.equal(id, versionId);
|
||||
return { sourceId: "src:scan", versionId, state: "indexing", phase: "ocr_running", activated: false, documents: [{ documentId: "doc:scan", state: "ocr_running", completedPages: 0, totalPages: 1, pages: [{ page: 1, method: "ocr", state: "ocr_running" }] }], error: null, statusUrl: `/ingestions/${versionId}`, reviewUrl: null };
|
||||
}
|
||||
};
|
||||
const app = createApp({ ingestService: ingestService as never, catalog: catalog as never, startReconciler: false });
|
||||
const server = app.listen(0);
|
||||
try {
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const baseUrl = `http://127.0.0.1:${address.port}`;
|
||||
const request = (sourceRef: string) => fetch(`${baseUrl}/ingest`, { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ sourceType: "file", sourceRef }) });
|
||||
|
||||
assert.equal((await request("native.pdf")).status, 201);
|
||||
const accepted = await request("scan.pdf");
|
||||
assert.equal(accepted.status, 202);
|
||||
assert.equal((await accepted.json() as { statusUrl: string }).statusUrl, `/ingestions/${versionId}`);
|
||||
assert.equal((await fetch(`${baseUrl}/ingestions/${versionId}`)).status, 401);
|
||||
const status = await fetch(`${baseUrl}/ingestions/${versionId}`, { headers: { authorization: "Bearer unit-8-token" } });
|
||||
assert.equal(status.status, 200);
|
||||
assert.equal((await status.json() as { phase: string }).phase, "ocr_running");
|
||||
|
||||
const unavailable = createApp({ ingestService: ingestService as never, catalog: undefined, startReconciler: false }).listen(0);
|
||||
try {
|
||||
const unavailableAddress = unavailable.address();
|
||||
assert.ok(unavailableAddress && typeof unavailableAddress === "object");
|
||||
const response = await fetch(`http://127.0.0.1:${unavailableAddress.port}/ingestions/${versionId}`, { headers: { authorization: "Bearer unit-8-token" } });
|
||||
assert.equal(response.status, 503);
|
||||
assert.equal((await response.json() as { code: string }).code, "CATALOG_UNAVAILABLE");
|
||||
const retrieval = await fetch(`http://127.0.0.1:${unavailableAddress.port}/retrieve`, {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({ query: "previous active content", mode: "documental", intent: "specific" })
|
||||
});
|
||||
assert.equal(retrieval.status, 503);
|
||||
assert.equal((await retrieval.json() as { code: string }).code, "CATALOG_UNAVAILABLE");
|
||||
} finally {
|
||||
unavailable.close();
|
||||
}
|
||||
} finally {
|
||||
server.close();
|
||||
setEnvFlag("knowledgeLifecycleEnforced", previous.knowledgeLifecycleEnforced);
|
||||
setEnvFlag("lifecycleAdminToken", previous.lifecycleAdminToken);
|
||||
setEnvFlag("ocrIngestEnabled", previous.ocrIngestEnabled);
|
||||
}
|
||||
});
|
||||
|
||||
test("status reports all native and OCR documents and fails the version when one document fails", async () => {
|
||||
let query = 0;
|
||||
const pool = { async query() {
|
||||
query += 1;
|
||||
if (query === 1) return { rowCount: 1, rows: [{ source_id: "src:mixed", state: "failed", error_code: "OCR_QUALITY_BLOCKED", error_detail: "Visible ink failed OCR quality" }] };
|
||||
if (query === 2) return { rowCount: 2, rows: [
|
||||
{ document_id: "doc:native", index_state: "indexing", job_state: null, completed_pages: null, requested_pages: null },
|
||||
{ document_id: "doc:ocr", index_state: "failed", job_state: "failed", completed_pages: 0, requested_pages: [2] }
|
||||
] };
|
||||
return { rowCount: 3, rows: [
|
||||
{ document_id: "doc:native", page_number: 1, extraction_method: "native", blocked_reason: null },
|
||||
{ document_id: "doc:ocr", page_number: 1, extraction_method: "native", blocked_reason: null },
|
||||
{ document_id: "doc:ocr", page_number: 2, extraction_method: "ocr", blocked_reason: "OCR_QUALITY_BLOCKED" }
|
||||
] };
|
||||
} };
|
||||
const status = await new CatalogRepository(pool as never).getIngestionStatus("version-mixed") as {
|
||||
phase: string; documents: Array<{ documentId: string; completedPages: number; totalPages: number }>; error: { code: string }
|
||||
};
|
||||
|
||||
assert.equal(status.phase, "failed");
|
||||
assert.deepEqual(status.documents.map(({ documentId, completedPages, totalPages }) => [documentId, completedPages, totalPages]), [
|
||||
["doc:native", 1, 1], ["doc:ocr", 1, 2]
|
||||
]);
|
||||
assert.equal(status.error.code, "OCR_QUALITY_BLOCKED");
|
||||
});
|
||||
183
tests/ocr/e2e.test.ts
Normal file
183
tests/ocr/e2e.test.ts
Normal file
|
|
@ -0,0 +1,183 @@
|
|||
import assert from "node:assert/strict";
|
||||
import test from "node:test";
|
||||
import { createApp } from "../../src/app.js";
|
||||
import { env } from "../../src/config/env.js";
|
||||
import { OcrDispatcher } from "../../src/modules/ocr/dispatcher.js";
|
||||
import { OcrIndexingService } from "../../src/modules/ocr/indexing.js";
|
||||
import { OcrReviewService, type OcrReviewCandidate } from "../../src/modules/ocr/review.js";
|
||||
import { sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
const token = "local-e2e-token";
|
||||
const scanVersion = "11111111-1111-4111-8111-111111111111";
|
||||
const mixedVersion = "22222222-2222-4222-8222-222222222222";
|
||||
|
||||
function headers(json = false): Record<string, string> {
|
||||
return { authorization: `Bearer ${token}`, ...(json ? { "content-type": "application/json" } : {}) };
|
||||
}
|
||||
|
||||
function candidate(): OcrReviewCandidate {
|
||||
const text = "CBGO4a";
|
||||
return {
|
||||
versionId: scanVersion,
|
||||
sourceId: "src:scan",
|
||||
state: "review_required",
|
||||
candidateSha256: "candidate-hash",
|
||||
baseActiveVersionId: "previous-active",
|
||||
currentActiveVersionId: "previous-active",
|
||||
activateRequested: true,
|
||||
processingFingerprint: "fingerprint",
|
||||
metadataHash: "metadata",
|
||||
documents: [{
|
||||
documentId: "doc:scan",
|
||||
pages: [{
|
||||
page: 1,
|
||||
imageUrl: `/ingestions/${scanVersion}/documents/doc:scan/pages/1/image`,
|
||||
nativeText: "",
|
||||
ocr: { text, lines: [{ lineId: "line-1", text, confidence: 0.72, bbox: [1, 2, 30, 12], lineSha256: sha256Hex(text) }] },
|
||||
candidateText: text,
|
||||
differences: ["native text is empty"],
|
||||
risks: [text]
|
||||
}]
|
||||
}]
|
||||
};
|
||||
}
|
||||
|
||||
test("local HTTP flow keeps native ingestion synchronous and activates only reviewed OCR content", async (context) => {
|
||||
const previous = { lifecycle: env.knowledgeLifecycleEnforced, enabled: env.ocrIngestEnabled, token: env.lifecycleAdminToken };
|
||||
Object.assign(env, { knowledgeLifecycleEnforced: true, ocrIngestEnabled: true, lifecycleAdminToken: token });
|
||||
context.after(() => Object.assign(env, {
|
||||
knowledgeLifecycleEnforced: previous.lifecycle,
|
||||
ocrIngestEnabled: previous.enabled,
|
||||
lifecycleAdminToken: previous.token
|
||||
}));
|
||||
let reviewCandidate = candidate();
|
||||
let reviewedText = "";
|
||||
const states = new Map([[scanVersion, "review_required"], [mixedVersion, "review_required"]]);
|
||||
const ingestService = {
|
||||
async ingest(input: { sourceRef: string }) {
|
||||
if (input.sourceRef === "native.pdf") {
|
||||
return { accepted: true, sourceId: "src:native", versionId: "native-active", versionNumber: 1, state: "active", filesDiscovered: 1, documentsProcessed: 1, chunksStored: 1, activated: true, noOp: false, collectionName: "rag" };
|
||||
}
|
||||
const versionId = input.sourceRef === "mixed.pdf" ? mixedVersion : scanVersion;
|
||||
return { accepted: true, sourceId: `src:${input.sourceRef}`, versionId, versionNumber: 2, state: "indexing", phase: "ocr_queued", statusUrl: `/ingestions/${versionId}`, reviewUrl: null, activated: false };
|
||||
},
|
||||
async cleanup() { return { deleted: 0 }; }
|
||||
};
|
||||
const catalog = {
|
||||
async claimNextOcrJob() { return undefined; },
|
||||
async getIngestionStatus(versionId: string) {
|
||||
const state = states.get(versionId);
|
||||
if (!state) return undefined;
|
||||
const mixed = versionId === mixedVersion;
|
||||
return {
|
||||
sourceId: mixed ? "src:mixed.pdf" : "src:scan.pdf", versionId, state, phase: state,
|
||||
activated: state === "active",
|
||||
documents: mixed
|
||||
? [{ documentId: "doc:mixed", state: "ocr_complete", completedPages: 2, totalPages: 2, pages: [
|
||||
{ page: 1, method: "native", state: "native_complete" }, { page: 2, method: "ocr", state: "ocr_complete" }
|
||||
] }]
|
||||
: [{ documentId: "doc:scan", state: "ocr_complete", completedPages: 1, totalPages: 1, pages: [{ page: 1, method: "ocr", state: "ocr_complete" }] }],
|
||||
error: null, statusUrl: `/ingestions/${versionId}`, reviewUrl: state === "review_required" ? `/ingestions/${versionId}/review` : null
|
||||
};
|
||||
}
|
||||
};
|
||||
const review = new OcrReviewService({
|
||||
async loadCandidate(versionId) { return versionId === scanVersion ? reviewCandidate : undefined; },
|
||||
async commitApproval(input) {
|
||||
reviewedText = input.reviewedText;
|
||||
reviewCandidate = { ...reviewCandidate, state: "indexing" };
|
||||
states.set(scanVersion, "indexing");
|
||||
},
|
||||
async commitRejection() { throw new Error("rejection is outside this flow"); }
|
||||
});
|
||||
const indexing = new OcrIndexingService({
|
||||
async findReusableVersion() { return undefined; },
|
||||
async indexReviewed(value) { assert.equal(value.reviewedText, "CBG04a"); return 1; },
|
||||
async markReady(versionId) { states.set(versionId, "ready"); },
|
||||
async settleReusable() { throw new Error("candidate is not reusable"); },
|
||||
async activateVersion(_sourceId, versionId, expected) {
|
||||
assert.equal(expected, "previous-active");
|
||||
states.set(versionId, "active");
|
||||
return versionId;
|
||||
}
|
||||
});
|
||||
const server = createApp({ ingestService: ingestService as never, catalog: catalog as never, ocrClient: {} as never, reviewService: review, indexingService: indexing, startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const base = `http://127.0.0.1:${address.port}`;
|
||||
const ingest = (sourceRef: string) => fetch(`${base}/ingest`, { method: "POST", headers: headers(true), body: JSON.stringify({ sourceType: "file", sourceRef }) });
|
||||
|
||||
assert.equal((await ingest("native.pdf")).status, 201);
|
||||
const accepted = await ingest("scanned.pdf");
|
||||
assert.equal(accepted.status, 202);
|
||||
assert.deepEqual(await accepted.json(), {
|
||||
accepted: true, sourceId: "src:scanned.pdf", versionId: scanVersion, versionNumber: 2,
|
||||
state: "indexing", phase: "ocr_queued", statusUrl: `/ingestions/${scanVersion}`, reviewUrl: null, activated: false
|
||||
});
|
||||
const reviewResponse = await fetch(`${base}/ingestions/${scanVersion}/review`, { headers: headers() });
|
||||
assert.equal(reviewResponse.status, 200);
|
||||
assert.deepEqual((await reviewResponse.json() as OcrReviewCandidate).documents[0]?.pages[0]?.risks, ["CBGO4a"]);
|
||||
const approval = await fetch(`${base}/ingestions/${scanVersion}/approve`, {
|
||||
method: "POST", headers: headers(true), body: JSON.stringify({
|
||||
candidateSha256: "candidate-hash", expectedActiveVersionId: "previous-active", reviewedBy: "local-reviewer",
|
||||
corrections: [{ documentId: "doc:scan", page: 1, lineId: "line-1", expectedLineSha256: sha256Hex("CBGO4a"), replacementText: "CBG04a" }]
|
||||
})
|
||||
});
|
||||
assert.equal(approval.status, 200);
|
||||
assert.deepEqual(await approval.json(), { versionId: scanVersion, state: "active", activated: true, activatedVersionId: scanVersion });
|
||||
assert.equal(reviewedText, "CBG04a");
|
||||
assert.equal((await fetch(`${base}/ingestions/${scanVersion}`, { headers: headers() }).then((response) => response.json()) as { state: string }).state, "active");
|
||||
|
||||
assert.equal((await ingest("mixed.pdf")).status, 202);
|
||||
const mixed = await fetch(`${base}/ingestions/${mixedVersion}`, { headers: headers() }).then((response) => response.json()) as { documents: Array<{ pages: Array<{ method: string }> }> };
|
||||
assert.deepEqual(mixed.documents[0]?.pages.map(({ method }) => method), ["native", "ocr"]);
|
||||
});
|
||||
|
||||
test("resends reuse identity, OCR exhaustion fails closed, and catalog absence returns 503", async (context) => {
|
||||
const previous = { lifecycle: env.knowledgeLifecycleEnforced, enabled: env.ocrIngestEnabled, token: env.lifecycleAdminToken };
|
||||
Object.assign(env, { knowledgeLifecycleEnforced: true, ocrIngestEnabled: true, lifecycleAdminToken: token });
|
||||
context.after(() => Object.assign(env, {
|
||||
knowledgeLifecycleEnforced: previous.lifecycle,
|
||||
ocrIngestEnabled: previous.enabled,
|
||||
lifecycleAdminToken: previous.token
|
||||
}));
|
||||
const identities = new Map<string, string>();
|
||||
let creations = 0;
|
||||
const ingestService = {
|
||||
async ingest(input: { sourceRef: string }) {
|
||||
let versionId = identities.get(input.sourceRef);
|
||||
if (!versionId) { versionId = input.sourceRef === "scan.pdf" ? scanVersion : mixedVersion; identities.set(input.sourceRef, versionId); creations += 1; }
|
||||
return { accepted: true, sourceId: `src:${input.sourceRef}`, versionId, versionNumber: 1, state: "indexing", phase: "ocr_queued", statusUrl: `/ingestions/${versionId}`, reviewUrl: null, activated: false };
|
||||
},
|
||||
async cleanup() { return { deleted: 0 }; }
|
||||
};
|
||||
const server = createApp({ ingestService: ingestService as never, catalog: undefined, startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const base = `http://127.0.0.1:${address.port}`;
|
||||
const resend = () => fetch(`${base}/ingest`, { method: "POST", headers: headers(true), body: JSON.stringify({ sourceType: "file", sourceRef: "scan.pdf" }) }).then((response) => response.json()) as Promise<{ versionId: string }>;
|
||||
assert.deepEqual([(await resend()).versionId, (await resend()).versionId], [scanVersion, scanVersion]);
|
||||
assert.equal(creations, 1);
|
||||
|
||||
const calls: string[] = [];
|
||||
let candidateState = "indexing";
|
||||
const dispatcher = new OcrDispatcher({
|
||||
async claimNextOcrJob() { return { jobId: "job", versionId: scanVersion, documentId: "doc", remoteJobId: null, remoteIdempotencyKey: "stable-key", state: "running", requestedPages: [1], completedPages: 0, configVersion: "ocr-v1", attemptCount: 1, heartbeatAt: null, leaseExpiresAt: null, nextAttemptAt: null, errorCode: null, errorDetail: null }; },
|
||||
async claimOcrJob() { return undefined; }, async recoverExpiredOcrLeases() { return []; }, async setOcrRemoteJob() { calls.push("remote"); },
|
||||
async requeueOcrJob() { calls.push("requeue"); }, async completeOcrJob() { calls.push("complete"); return false; },
|
||||
async failOcrJob() { calls.push("job-failed"); }, async markReviewRequired() { calls.push("review"); },
|
||||
async markFailed() { candidateState = "failed"; calls.push("version-failed"); }
|
||||
}, { async submit() { throw new Error("OCR connection unavailable after bounded retries"); } } as never, async () => ({ bytes: Buffer.from("pdf"), documentSha256: sha256Hex("pdf") }));
|
||||
assert.equal(await dispatcher.runOnce(), "failed");
|
||||
assert.deepEqual(calls, ["job-failed", "version-failed"]);
|
||||
assert.equal(candidateState, "failed");
|
||||
|
||||
const unavailableStatus = await fetch(`${base}/ingestions/${scanVersion}`, { headers: headers() });
|
||||
assert.equal(unavailableStatus.status, 503);
|
||||
assert.equal((await unavailableStatus.json() as { code: string }).code, "CATALOG_UNAVAILABLE");
|
||||
const unavailableRetrieval = await fetch(`${base}/retrieve`, { method: "POST", headers: headers(true), body: JSON.stringify({ query: "prior active", mode: "documental", intent: "specific" }) });
|
||||
assert.equal(unavailableRetrieval.status, 503);
|
||||
assert.equal((await unavailableRetrieval.json() as { code: string }).code, "CATALOG_UNAVAILABLE");
|
||||
});
|
||||
134
tests/ocr/retention.test.ts
Normal file
134
tests/ocr/retention.test.ts
Normal file
|
|
@ -0,0 +1,134 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { access, mkdir, mkdtemp, rm } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import { createApp } from "../../src/app.js";
|
||||
import { env } from "../../src/config/env.js";
|
||||
import { KnowledgeLifecycleReconciler } from "../../src/modules/catalog/reconciler.js";
|
||||
import { CatalogRepository } from "../../src/modules/catalog/repository.js";
|
||||
import { OcrRetentionService, type OcrRetentionCandidate, type OcrRetentionStore } from "../../src/modules/ocr/retention.js";
|
||||
|
||||
const activeId = "11111111-1111-4111-8111-111111111111";
|
||||
const failedId = "22222222-2222-4222-8222-222222222222";
|
||||
const deletingId = "33333333-3333-4333-8333-333333333333";
|
||||
|
||||
function candidate(versionId: string, state: OcrRetentionCandidate["state"], artifactState: OcrRetentionCandidate["artifactState"] = "present"): OcrRetentionCandidate {
|
||||
return { versionId, sourceId: `source:${versionId}`, state, artifactState };
|
||||
}
|
||||
|
||||
test("repository retention selection applies exact TTLs and CAS-protects active versions", async () => {
|
||||
const queries: Array<{ sql: string; params?: unknown[] }> = [];
|
||||
const pool = { async query(sql: string, params?: unknown[]) {
|
||||
queries.push({ sql, params });
|
||||
if (sql.includes("SELECT v.version_id")) return { rowCount: 1, rows: [{ version_id: failedId, source_id: "source:failed", state: "failed", artifact_state: "present" }] };
|
||||
return { rowCount: 1, rows: [] };
|
||||
} };
|
||||
const repository = new CatalogRepository(pool as never);
|
||||
const now = new Date("2026-09-15T12:00:00.000Z");
|
||||
|
||||
assert.deepEqual(await repository.listOcrRetentionCandidates(now), [{ ...candidate(failedId, "failed"), sourceId: "source:failed" }]);
|
||||
assert.equal(await repository.expireOcrReview("source:review", activeId, now), true);
|
||||
assert.equal(await repository.claimOcrRetentionDeletion("source:failed", failedId, "failed", "present"), true);
|
||||
assert.equal(await repository.completeOcrRetentionDeletion("source:failed", failedId, "failed"), true);
|
||||
|
||||
assert.match(queries[0]!.sql, /review_required.*30 days.*failed.*rejected.*7 days.*superseded.*30 days/s);
|
||||
assert.match(queries[0]!.sql, /active_version_id IS DISTINCT FROM v\.version_id/);
|
||||
assert.deepEqual(queries[0]!.params, [now]);
|
||||
assert.match(queries[1]!.sql, /state = 'rejected'.*REVIEW_EXPIRED.*retention_due_at/s);
|
||||
assert.match(queries[2]!.sql, /SET artifact_state = 'retention_deleting'.*state = \$3.*artifact_state = \$4/s);
|
||||
assert.match(queries[2]!.sql, /active_version_id IS DISTINCT FROM v\.version_id/);
|
||||
assert.match(queries[3]!.sql, /artifact_state = 'retention_deleted'.*artifact_state = 'retention_deleting'/s);
|
||||
});
|
||||
|
||||
test("activation cannot race an artifact deletion already claimed by retention", async () => {
|
||||
const pool = {
|
||||
async query(sql: string) {
|
||||
if (sql.includes("maintenance")) return { rowCount: 1, rows: [{ maintenance: false }] };
|
||||
if (sql.includes("FROM rag_sources")) return { rowCount: 1, rows: [{ active_version_id: null }] };
|
||||
return { rowCount: 1, rows: [{ version_id: failedId, source_id: "source:failed", version_number: 2, previous_version_id: null, state: "ready", tags: [], source_content_hash: "hash", processing_fingerprint: "fingerprint", metadata_hash: "metadata", embedding_provider: "test", embedding_model: "test", embedding_dimensions: 3, expected_document_count: 1, expected_point_count: 1, verified_point_count: 1, qdrant_collection: "rag", artifact_state: "retention_deleting" }] };
|
||||
},
|
||||
async connect() { return { query: this.query, release() {} }; }
|
||||
};
|
||||
const repository = new CatalogRepository(pool as never);
|
||||
|
||||
await assert.rejects(repository.activateVersion("source:failed", failedId, null), (error) => error instanceof Error && (error as { code?: string }).code === "VERSION_ARTIFACTS_UNAVAILABLE");
|
||||
});
|
||||
|
||||
test("retention expires review before removal and preserves active artifacts", async (context) => {
|
||||
const root = await mkdtemp(path.join(os.tmpdir(), "rag-retention-"));
|
||||
context.after(() => rm(root, { recursive: true, force: true }));
|
||||
await Promise.all([activeId, failedId].map((id) => mkdir(path.join(root, id))));
|
||||
const calls: string[] = [];
|
||||
const store: OcrRetentionStore = {
|
||||
async listOcrRetentionCandidates() { return [candidate(activeId, "active"), candidate(failedId, "failed"), candidate(deletingId, "review_required")]; },
|
||||
async withVersionTryLock(versionId, handler) { calls.push(`lock:${versionId}`); return handler(); },
|
||||
async expireOcrReview(_sourceId, versionId) { calls.push(`expire:${versionId}`); return true; },
|
||||
async claimOcrRetentionDeletion(_sourceId, versionId) { calls.push(`claim:${versionId}`); return true; },
|
||||
async completeOcrRetentionDeletion(_sourceId, versionId) { calls.push(`complete:${versionId}`); return true; }
|
||||
};
|
||||
|
||||
assert.deepEqual(await new OcrRetentionService(store, root).runOnce(), { expired: 1, deleted: 1, resumed: 0, skipped: 1 });
|
||||
await access(path.join(root, activeId));
|
||||
await assert.rejects(access(path.join(root, failedId)));
|
||||
assert.equal(calls.some((call) => call.includes(activeId)), false);
|
||||
assert.equal(calls.includes(`expire:${deletingId}`), true);
|
||||
});
|
||||
|
||||
test("interrupted deletion resumes idempotently without affecting another version", async (context) => {
|
||||
const root = await mkdtemp(path.join(os.tmpdir(), "rag-retention-resume-"));
|
||||
context.after(() => rm(root, { recursive: true, force: true }));
|
||||
await Promise.all([deletingId, activeId].map((id) => mkdir(path.join(root, id))));
|
||||
let completed = false;
|
||||
let failCompletion = true;
|
||||
const store: OcrRetentionStore = {
|
||||
async listOcrRetentionCandidates() { return completed ? [] : [candidate(deletingId, "rejected", "retention_deleting")]; },
|
||||
async withVersionTryLock(_versionId, handler) { return handler(); }, async expireOcrReview() { return false; },
|
||||
async claimOcrRetentionDeletion() { return true; },
|
||||
async completeOcrRetentionDeletion() { if (failCompletion) { failCompletion = false; throw new Error("simulated restart"); } completed = true; return true; }
|
||||
};
|
||||
const retention = new OcrRetentionService(store, root);
|
||||
|
||||
await assert.rejects(retention.runOnce(), /simulated restart/);
|
||||
await assert.rejects(access(path.join(root, deletingId)));
|
||||
assert.deepEqual(await retention.runOnce(), { expired: 0, deleted: 0, resumed: 1, skipped: 0 });
|
||||
assert.deepEqual(await retention.runOnce(), { expired: 0, deleted: 0, resumed: 0, skipped: 0 });
|
||||
await access(path.join(root, activeId));
|
||||
});
|
||||
|
||||
test("OCR flag off keeps native HTTP synchronous and hides candidate surfaces", async (context) => {
|
||||
const previous = { enabled: env.ocrIngestEnabled, lifecycle: env.knowledgeLifecycleEnforced, token: env.lifecycleAdminToken };
|
||||
Object.assign(env, { ocrIngestEnabled: false, knowledgeLifecycleEnforced: true, lifecycleAdminToken: "retention-token" });
|
||||
context.after(() => Object.assign(env, previous));
|
||||
let candidateReads = 0;
|
||||
const native = { accepted: true, sourceId: "source:native", versionId: "native-1", state: "active" };
|
||||
const app = createApp({
|
||||
ingestService: { async ingest() { return native; }, async cleanup() { return { deleted: 0 }; } } as never,
|
||||
catalog: { async getIngestionStatus() { candidateReads += 1; return {}; } } as never,
|
||||
reviewService: { async view() { candidateReads += 1; return {}; }, async approve() { candidateReads += 1; return {} as never; }, async reject() { candidateReads += 1; return {} as never; } },
|
||||
indexingService: { async index() { candidateReads += 1; return {} as never; } }, startReconciler: false
|
||||
});
|
||||
const server = app.listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const base = `http://127.0.0.1:${address.port}`;
|
||||
|
||||
assert.equal((await fetch(`${base}/ingest`, { method: "POST", headers: { "content-type": "application/json" }, body: "{}" })).status, 201);
|
||||
for (const [suffix, method] of [["", "GET"], ["/review", "GET"], ["/approve", "POST"], ["/reject", "POST"]] as const) {
|
||||
assert.equal((await fetch(`${base}/ingestions/${failedId}${suffix}`, { method, headers: { authorization: "Bearer retention-token", "content-type": "application/json" }, body: method === "POST" ? "{}" : undefined })).status, 404);
|
||||
}
|
||||
assert.equal(candidateReads, 0);
|
||||
});
|
||||
|
||||
test("reconciler runs retention inside its exclusive global lock", async () => {
|
||||
let locked = false;
|
||||
const catalog = {
|
||||
async withGlobalTryLock(_name: string, handler: () => Promise<unknown>) { locked = true; try { return await handler(); } finally { locked = false; } },
|
||||
async resolveActiveVersions() { return []; }, async validateActiveInvariant() { return []; }, async listOrphanedIndexingCandidates() { return []; }
|
||||
};
|
||||
const retention = { async runOnce() { assert.equal(locked, true); return { expired: 0, deleted: 0, resumed: 0, skipped: 0 }; } };
|
||||
const reconciler = new KnowledgeLifecycleReconciler(catalog as never, {} as never, undefined, retention);
|
||||
|
||||
assert.equal((await reconciler.runOnce()).ocrRetentionDeleted, 0);
|
||||
});
|
||||
185
tests/ocr/review.test.ts
Normal file
185
tests/ocr/review.test.ts
Normal file
|
|
@ -0,0 +1,185 @@
|
|||
import assert from "node:assert/strict";
|
||||
import test from "node:test";
|
||||
import { createApp } from "../../src/app.js";
|
||||
import { env } from "../../src/config/env.js";
|
||||
import { CatalogError } from "../../src/modules/catalog/errors.js";
|
||||
import { OcrIndexingService, type ApprovedOcrCandidate, type OcrIndexingStore } from "../../src/modules/ocr/indexing.js";
|
||||
import { OcrReviewService, type OcrReviewCandidate } from "../../src/modules/ocr/review.js";
|
||||
import { sha256Hex } from "../../src/shared/utils/ids.js";
|
||||
|
||||
function candidate(state: OcrReviewCandidate["state"] = "review_required"): OcrReviewCandidate {
|
||||
const lines = ["CBGO4a", "FATo7"].map((text, index) => ({
|
||||
lineId: `line-${index + 1}`, text, confidence: 0.7, bbox: [0, index * 10, 50, index * 10 + 8] as [number, number, number, number], lineSha256: sha256Hex(text)
|
||||
}));
|
||||
return {
|
||||
versionId: "version-2", sourceId: "source-1", state, candidateSha256: "candidate-hash",
|
||||
baseActiveVersionId: "version-1", currentActiveVersionId: "version-1", activateRequested: true,
|
||||
processingFingerprint: "fingerprint", metadataHash: "metadata", documents: [{ documentId: "document-1", pages: [{
|
||||
page: 1, imageUrl: "/private/page-1.png", nativeText: "", ocr: { text: "CBGO4a\nFATo7", lines },
|
||||
candidateText: "CBGO4a\nFATo7", differences: ["native text is empty"], risks: ["CBGO4a", "FATo7"]
|
||||
}] }]
|
||||
};
|
||||
}
|
||||
|
||||
test("review HTTP rejects unauthorized and premature access without exposing artifacts", async (context) => {
|
||||
const previous = { token: env.lifecycleAdminToken, enabled: env.ocrIngestEnabled };
|
||||
Object.assign(env, { lifecycleAdminToken: "review-token", ocrIngestEnabled: true });
|
||||
context.after(() => Object.assign(env, { lifecycleAdminToken: previous.token, ocrIngestEnabled: previous.enabled }));
|
||||
let value = candidate();
|
||||
let reads = 0;
|
||||
const review = new OcrReviewService({ async loadCandidate() { reads += 1; return value; }, async commitApproval() { throw new Error("not called"); } });
|
||||
const server = createApp({ reviewService: review, startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const url = `http://127.0.0.1:${address.port}/ingestions/version-2/review`;
|
||||
|
||||
const unauthorized = await fetch(url);
|
||||
assert.equal(unauthorized.status, 401);
|
||||
assert.equal((await fetch(url.replace("/review", "/approve"), { method: "POST" })).status, 401);
|
||||
assert.equal(reads, 0);
|
||||
value = candidate("indexing");
|
||||
const premature = await fetch(url, { headers: { authorization: "Bearer review-token" } });
|
||||
assert.equal(premature.status, 409);
|
||||
assert.deepEqual(await premature.json(), { ok: false, error: "Version is not awaiting OCR review", code: "INVALID_VERSION_STATE" });
|
||||
});
|
||||
|
||||
test("review exposes audit detail and commits current corrections as one immutable set", async () => {
|
||||
const commits: unknown[] = [];
|
||||
const service = new OcrReviewService({ async loadCandidate() { return candidate(); }, async commitApproval(value) { commits.push(value); } });
|
||||
const review = await service.view("version-2");
|
||||
assert.deepEqual(review.documents[0]?.pages[0]?.ocr.lines.map(({ text, confidence, bbox }) => [text, confidence, bbox]), [
|
||||
["CBGO4a", 0.7, [0, 0, 50, 8]], ["FATo7", 0.7, [0, 10, 50, 18]]
|
||||
]);
|
||||
assert.deepEqual(review.documents[0]?.pages[0]?.risks, ["CBGO4a", "FATo7"]);
|
||||
|
||||
const approved = await service.approve("version-2", {
|
||||
candidateSha256: "candidate-hash", expectedActiveVersionId: "version-1", reviewedBy: "admin",
|
||||
corrections: [
|
||||
{ documentId: "document-1", page: 1, lineId: "line-1", expectedLineSha256: sha256Hex("CBGO4a"), replacementText: "CBG04a" },
|
||||
{ documentId: "document-1", page: 1, lineId: "line-2", expectedLineSha256: sha256Hex("FATo7"), replacementText: "FAT07" }
|
||||
]
|
||||
});
|
||||
assert.equal(commits.length, 1);
|
||||
assert.deepEqual((commits[0] as { corrections: Array<{ replacementText: string }> }).corrections.map(({ replacementText }) => replacementText), ["CBG04a", "FAT07"]);
|
||||
assert.equal(approved.reviewedText, "CBG04a\nFAT07");
|
||||
assert.equal(approved.reviewedTextSha256, sha256Hex("CBG04a\nFAT07"));
|
||||
});
|
||||
|
||||
test("stale or duplicate corrections conflict before any correction or transition", async () => {
|
||||
let commits = 0;
|
||||
const service = new OcrReviewService({ async loadCandidate() { return candidate(); }, async commitApproval() { commits += 1; } });
|
||||
const base = { candidateSha256: "stale", expectedActiveVersionId: "version-1", reviewedBy: "admin", corrections: [] };
|
||||
await assert.rejects(service.approve("version-2", base), (error) => error instanceof CatalogError && error.statusCode === 409);
|
||||
const duplicate = { ...base, candidateSha256: "candidate-hash", corrections: Array(2).fill({
|
||||
documentId: "document-1", page: 1, lineId: "line-1", expectedLineSha256: sha256Hex("CBGO4a"), replacementText: "CBG04a"
|
||||
}) };
|
||||
await assert.rejects(service.approve("version-2", duplicate), (error) => error instanceof CatalogError && error.code === "CORRECTION_CONFLICT");
|
||||
await assert.rejects(service.approve("version-2", { ...base, candidateSha256: "candidate-hash", expectedActiveVersionId: "version-old" }),
|
||||
(error) => error instanceof CatalogError && error.code === "ACTIVE_VERSION_CHANGED");
|
||||
await assert.rejects(service.approve("version-2", { ...base, candidateSha256: "candidate-hash", corrections: [{
|
||||
documentId: "document-1", page: 1, lineId: "line-1", expectedLineSha256: sha256Hex("stale"), replacementText: "CBG04a"
|
||||
}] }), (error) => error instanceof CatalogError && error.code === "CORRECTION_CONFLICT");
|
||||
assert.equal(commits, 0);
|
||||
});
|
||||
|
||||
function approved(activateRequested: boolean, state: ApprovedOcrCandidate["state"] = "indexing"): ApprovedOcrCandidate {
|
||||
return { versionId: "version-2", sourceId: "source-1", state, activateRequested, expectedActiveVersionId: "version-1", reviewedText: "reviewed", reviewedTextSha256: sha256Hex("reviewed"), processingFingerprint: "fingerprint", metadataHash: "metadata" };
|
||||
}
|
||||
|
||||
test("indexing activates only when requested and leaves a new candidate ready on an activation race", async () => {
|
||||
const calls: string[] = [];
|
||||
const store: OcrIndexingStore = {
|
||||
async findReusableVersion() { return undefined; }, async indexReviewed() { calls.push("embed"); return 1; },
|
||||
async markReady() { calls.push("ready"); }, async settleReusable() { throw new Error("not reusable"); },
|
||||
async activateVersion() { calls.push("activate"); return "version-2"; }
|
||||
};
|
||||
const service = new OcrIndexingService(store);
|
||||
assert.deepEqual(await service.index(approved(true)), { versionId: "version-2", state: "active", activated: true, activatedVersionId: "version-2" });
|
||||
assert.deepEqual(calls.splice(0), ["embed", "ready", "activate"]);
|
||||
assert.deepEqual(await service.index(approved(false)), { versionId: "version-2", state: "ready", activated: false });
|
||||
assert.deepEqual(calls.splice(0), ["embed", "ready"]);
|
||||
store.activateVersion = async () => { throw new CatalogError("race", 409, "ACTIVE_VERSION_PRECONDITION_FAILED"); };
|
||||
await assert.rejects(service.index(approved(true)), (error) => error instanceof CatalogError && error.code === "ACTIVE_VERSION_CHANGED");
|
||||
assert.deepEqual(calls, ["embed", "ready"]);
|
||||
});
|
||||
|
||||
test("reusable and rejected candidates create no embeddings and reusable activation obeys CAS intent", async () => {
|
||||
const calls: string[] = [];
|
||||
const store: OcrIndexingStore = {
|
||||
async findReusableVersion() { return { versionId: "version-existing" }; }, async indexReviewed() { calls.push("embed"); return 1; },
|
||||
async markReady() { calls.push("ready"); }, async settleReusable(_candidate, _reusable, activate) { calls.push(`settle:${activate}`); return activate; },
|
||||
async activateVersion() { throw new Error("new activation must not run"); }
|
||||
};
|
||||
const service = new OcrIndexingService(store);
|
||||
assert.deepEqual(await service.index(approved(true)), { versionId: "version-2", state: "rejected", activated: true, activatedVersionId: "version-existing", errorCode: "DUPLICATE_REUSABLE_VERSION" });
|
||||
assert.deepEqual(await service.index(approved(false)), { versionId: "version-2", state: "rejected", activated: false, errorCode: "DUPLICATE_REUSABLE_VERSION" });
|
||||
store.settleReusable = async () => { throw new CatalogError("race", 409, "ACTIVE_VERSION_PRECONDITION_FAILED"); };
|
||||
await assert.rejects(service.index(approved(true)), (error) => error instanceof CatalogError && error.code === "ACTIVE_VERSION_CHANGED");
|
||||
await assert.rejects(service.index(approved(true, "rejected")), (error) => error instanceof CatalogError && error.code === "INVALID_VERSION_STATE");
|
||||
assert.deepEqual(calls, ["settle:true", "settle:false"]);
|
||||
});
|
||||
|
||||
test("authenticated rejection is durable, conflict-safe, and never indexes or activates", async (context) => {
|
||||
const previous = { token: env.lifecycleAdminToken, enabled: env.ocrIngestEnabled };
|
||||
Object.assign(env, { lifecycleAdminToken: "review-token", ocrIngestEnabled: true });
|
||||
context.after(() => Object.assign(env, { lifecycleAdminToken: previous.token, ocrIngestEnabled: previous.enabled }));
|
||||
let value = candidate();
|
||||
let reads = 0;
|
||||
const audits: Array<{ reviewedBy: string; reason: string }> = [];
|
||||
const review = new OcrReviewService({
|
||||
async loadCandidate() { reads += 1; return value; },
|
||||
async commitApproval() { throw new Error("approval must not run"); },
|
||||
async commitRejection({ reviewedBy, reason }) {
|
||||
audits.push({ reviewedBy, reason });
|
||||
value = { ...value, state: "rejected" };
|
||||
}
|
||||
});
|
||||
let indexingCalls = 0;
|
||||
const catalog = { async getIngestionStatus() {
|
||||
return { versionId: value.versionId, state: value.state, phase: value.state, activated: false, error: value.state === "rejected" ? { code: "OCR_REJECTED", message: audits[0]?.reason, retryable: false } : null };
|
||||
} };
|
||||
const server = createApp({ catalog: catalog as never, reviewService: review, indexingService: { async index() { indexingCalls += 1; throw new Error("indexing must not run"); } }, startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const url = `http://127.0.0.1:${address.port}/ingestions/version-2`;
|
||||
|
||||
const unauthorized = await fetch(`${url}/reject`, { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ candidateSha256: "candidate-hash", reviewedBy: "admin", reason: "Unreadable code" }) });
|
||||
assert.equal(unauthorized.status, 401);
|
||||
assert.equal(reads, 0);
|
||||
const invalid = await fetch(`${url}/reject`, { method: "POST", headers: { authorization: "Bearer review-token", "content-type": "application/json" }, body: "{}" });
|
||||
assert.equal(invalid.status, 400);
|
||||
assert.deepEqual(await invalid.json(), { ok: false, error: "Candidate hash, reviewer, and rejection reason are required", code: "INVALID_REJECTION" });
|
||||
assert.equal(reads, 0);
|
||||
const stale = await fetch(`${url}/reject`, { method: "POST", headers: { authorization: "Bearer review-token", "content-type": "application/json" }, body: JSON.stringify({ candidateSha256: "stale", reviewedBy: "admin", reason: "Unreadable code" }) });
|
||||
assert.equal(stale.status, 409);
|
||||
assert.equal(audits.length, 0);
|
||||
const rejected = await fetch(`${url}/reject`, { method: "POST", headers: { authorization: "Bearer review-token", "content-type": "application/json" }, body: JSON.stringify({ candidateSha256: "candidate-hash", reviewedBy: "admin", reason: "Unreadable code" }) });
|
||||
assert.equal(rejected.status, 200);
|
||||
assert.deepEqual(await rejected.json(), { versionId: "version-2", state: "rejected", activated: false });
|
||||
assert.deepEqual(audits, [{ reviewedBy: "admin", reason: "Unreadable code" }]);
|
||||
const status = await fetch(url, { headers: { authorization: "Bearer review-token" } });
|
||||
assert.equal(status.status, 200);
|
||||
assert.deepEqual(await status.json(), { versionId: "version-2", state: "rejected", phase: "rejected", activated: false, error: { code: "OCR_REJECTED", message: "Unreadable code", retryable: false } });
|
||||
const repeated = await fetch(`${url}/reject`, { method: "POST", headers: { authorization: "Bearer review-token", "content-type": "application/json" }, body: JSON.stringify({ candidateSha256: "candidate-hash", reviewedBy: "admin", reason: "Again" }) });
|
||||
assert.equal(repeated.status, 409);
|
||||
assert.equal(indexingCalls, 0);
|
||||
assert.equal(audits.length, 1);
|
||||
});
|
||||
|
||||
test("playground serves the authenticated OCR review controls and audit fields", async (context) => {
|
||||
const server = createApp({ startReconciler: false }).listen(0);
|
||||
context.after(() => server.close());
|
||||
const address = server.address();
|
||||
assert.ok(address && typeof address === "object");
|
||||
const base = `http://127.0.0.1:${address.port}`;
|
||||
const [html, script, styles] = await Promise.all([
|
||||
fetch(`${base}/playground`).then((response) => response.text()),
|
||||
fetch(`${base}/playground/app.js`).then((response) => response.text()),
|
||||
fetch(`${base}/playground/styles.css`).then((response) => response.text())
|
||||
]);
|
||||
for (const marker of ["data-tab=\"review\"", "reviewVersionId", "reviewToken", "loadReviewButton", "approveReviewButton", "rejectReviewButton", "reviewCandidate"]) assert.match(html, new RegExp(marker));
|
||||
for (const marker of ["Authorization", "/review", "/approve", "/reject", "candidateSha256", "expectedLineSha256", "imageUrl", "nativeText", "confidence", "bbox", "differences", "risks"]) assert.match(script, new RegExp(marker));
|
||||
assert.match(styles, /\.review-page/);
|
||||
});
|
||||
77
tests/parsers/pdf-pages.test.ts
Normal file
77
tests/parsers/pdf-pages.test.ts
Normal file
|
|
@ -0,0 +1,77 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { access, mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import {
|
||||
isSupportedDocument,
|
||||
parseDocument,
|
||||
parsePdfPages
|
||||
} from "../../src/modules/parsers/parser-registry.js";
|
||||
|
||||
const fixturePath = new URL("../fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||
|
||||
test("parsePdfPages extracts ordered one-based pages with stable native text hashes", async () => {
|
||||
const pages = await parsePdfPages(fixturePath);
|
||||
|
||||
assert.deepEqual(pages, [
|
||||
{
|
||||
page: 1,
|
||||
text: "Native page one.",
|
||||
textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4"
|
||||
},
|
||||
{
|
||||
page: 2,
|
||||
text: "",
|
||||
textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
|
||||
},
|
||||
{
|
||||
page: 3,
|
||||
text: "Native page three.",
|
||||
textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e"
|
||||
}
|
||||
]);
|
||||
});
|
||||
|
||||
test("parseDocument preserves the synchronous native PDF content contract", async () => {
|
||||
const parsed = await parseDocument(fixturePath.pathname);
|
||||
|
||||
assert.equal(parsed.title, "native-three-pages.pdf");
|
||||
assert.equal(parsed.content, "Native page one.\n\nNative page three.");
|
||||
assert.equal(parsed.mimeType, "application/pdf");
|
||||
assert.equal(parsed.chunkMode, "documental");
|
||||
});
|
||||
|
||||
test("page extraction rejects non-PDF inputs before parsing bytes", async () => {
|
||||
await assert.rejects(
|
||||
parsePdfPages(new URL("../fixtures/ocr/not-a-pdf.txt", import.meta.url)),
|
||||
/only PDF documents support page extraction/i
|
||||
);
|
||||
});
|
||||
|
||||
test("supported text documents are read as data and never executed", async () => {
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-parser-threat-"));
|
||||
const markerPath = path.join(directory, "executed");
|
||||
const payload = `$(touch ${markerPath})`;
|
||||
|
||||
try {
|
||||
for (const fileName of ["requirements.txt", "executable.md"]) {
|
||||
const filePath = path.join(directory, fileName);
|
||||
await writeFile(filePath, payload, "utf8");
|
||||
const parsed = await parseDocument(filePath);
|
||||
|
||||
assert.equal(isSupportedDocument(filePath), true);
|
||||
assert.equal(parsed.content, payload);
|
||||
}
|
||||
await assert.rejects(access(markerPath), /ENOENT/);
|
||||
} finally {
|
||||
await rm(directory, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("build and executable lookalike formats remain unsupported", () => {
|
||||
assert.deepEqual(
|
||||
["CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument),
|
||||
[false, false, false]
|
||||
);
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue