From b0fdef001f277c33ed71cf02a0f816fbe3b8ef9e Mon Sep 17 00:00:00 2001 From: Paco POR-CORREO Date: Mon, 14 Sep 2026 22:35:11 +0200 Subject: [PATCH] feat(ocr): add PaddleOCR rendering --- docs/HISTORIAL_SESIONES.md | 31 +++- ocr-service/Dockerfile | 32 ++++ ocr-service/Dockerfile.dockerignore | 5 + ocr-service/README.md | 14 +- ocr-service/app/engine.py | 48 ++++++ ocr-service/app/main.py | 6 +- ocr-service/app/models.py | 17 ++ ocr-service/app/render.py | 123 +++++++++++++++ ocr-service/requirements.txt | 5 + ocr-service/tests/test_render.py | 147 ++++++++++++++++++ .../ocr-ingest-integration/apply-progress.md | 39 ++++- .../changes/ocr-ingest-integration/tasks.md | 4 +- 12 files changed, 459 insertions(+), 12 deletions(-) create mode 100644 ocr-service/Dockerfile create mode 100644 ocr-service/Dockerfile.dockerignore create mode 100644 ocr-service/app/engine.py create mode 100644 ocr-service/app/models.py create mode 100644 ocr-service/app/render.py create mode 100644 ocr-service/tests/test_render.py diff --git a/docs/HISTORIAL_SESIONES.md b/docs/HISTORIAL_SESIONES.md index 813774a..c0f7379 100644 --- a/docs/HISTORIAL_SESIONES.md +++ b/docs/HISTORIAL_SESIONES.md @@ -3,7 +3,7 @@ **Proyecto:** Workspace de tools IA para empresas **Modulo:** RAG **Ultima actualizacion:** 2026-09-14 -**Ultima modificacion por:** Subagente Correccion OCR Unit 5 +**Ultima modificacion por:** Subagente OCR Unit 6 Render **Estado:** Activo --- @@ -607,3 +607,32 @@ Continuidad operativa y evolutiva del modulo RAG. - `npm test` 45/45, `npm run check`, compileall y espacios correctos; servidor y temporales eliminados, `.venv` retenido; tareas 3.1/3.2 completadas. Revision: `sha256:0d8b2c57bbab967473eaf99a7ca900168554e64254fa80197b3626b4570b58d0`. **Archivos modificados:** `.gitignore`, `ocr-service/{README.md,requirements.txt,app/,tests/}`, `openspec/changes/ocr-ingest-integration/apply-progress.md`, `docs/HISTORIAL_SESIONES.md`. + +--- + +### 2026-09-14 - Subagente OCR Unit 6 Render - Renderizado y runtime PaddleOCR + +**Agente:** **Subagente OCR Unit 6 Render** +**Rol/responsabilidad:** Implementar exclusivamente Unit 6, tareas 3.3 y 3.4, mediante TDD estricto, sin iniciar Unit 7 ni realizar acciones de entrega Git. +**Modelo:** openai/gpt-5.6-sol +**Session ID OpenCode:** `ses_f5e8c95a3ffeugBz7od2TtJr4n` +**Directorio:** `/home/pancho/Documentos/Empresa/Desarrollo/IA/RAG` + +**Trabajo realizado:** +- Implementados renderizado PDF determinista a 200 DPI, limite previo de 25 megapixeles, adaptador PaddleOCR, IDs de linea, metricas y esquema de resultado contractual. +- Añadidas pruebas RED-first con motor falso determinista y cobertura de paginas seleccionadas, limites, payload Paddle y contrato de imagen. +- Creada imagen CPU no privilegiada con PaddleOCR 3.4.0/PaddlePaddle 3.2.2, modelos baked, un worker, volumen privado y limites operativos documentados. +- Restringido el contexto Docker con una lista de inclusion minima para no enviar rutas sensibles o ajenas al servicio. + +**Validacion y estado final:** +- RED valido por ausencia de `app.engine`; GREEN focalizado 5/5 y suite OCR completa 13/13. +- Build Docker corregido tras detectar `libGL.so.1` ausente; harness offline limitado a 3 CPU/5 GiB cargo modelos baked y genero PNG 1700x2200 con esquema `1`. +- Un rebuild opcional posterior agoto el almacenamiento Docker; se revirtio esa unica optimizacion no validada, se restauro exactamente el Dockerfile ya probado y se elimino la imagen local. +- `npm test` 45/45, `npm run check`, `compileall` y espacios correctos; 397 lineas nativas, dentro del limite de 400. Tareas 3.3/3.4 completadas; Unit 7 no iniciada. + +**Archivos modificados:** +- `ocr-service/{Dockerfile,Dockerfile.dockerignore,README.md,requirements.txt}` +- `ocr-service/app/{engine.py,main.py,models.py,render.py}` +- `ocr-service/tests/test_render.py` +- `openspec/changes/ocr-ingest-integration/{tasks.md,apply-progress.md}` +- `docs/HISTORIAL_SESIONES.md` diff --git a/ocr-service/Dockerfile b/ocr-service/Dockerfile new file mode 100644 index 0000000..336d3f5 --- /dev/null +++ b/ocr-service/Dockerfile @@ -0,0 +1,32 @@ +FROM python:3.11.13-slim-bookworm + +LABEL resource.cpu.max="3" \ + resource.memory.max="5GiB" + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + OCR_JOBS_DB="/data/jobs/jobs.db" \ + OCR_LOAD_ENGINE="1" \ + HOME="/opt/ocr-home" + +RUN apt-get update \ + && apt-get install --yes --no-install-recommends libgl1 libglib2.0-0 libgomp1 \ + && rm -rf /var/lib/apt/lists/* \ + && groupadd --system ocr && useradd --system --gid ocr --home-dir /opt/ocr-home ocr \ + && mkdir -p /opt/ocr-home /data/jobs \ + && chown -R ocr:ocr /opt/ocr-home /data/jobs + +WORKDIR /srv/ocr +COPY ocr-service/requirements.txt ./requirements.txt +RUN python -m pip install --no-cache-dir --requirement requirements.txt + +COPY ocr-service/app ./app +USER ocr +RUN python -m app.models + +VOLUME ["/data/jobs"] +EXPOSE 8000 +HEALTHCHECK --interval=30s --timeout=5s --start-period=90s --retries=3 \ + CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health/ready', timeout=4)" + +CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"] diff --git a/ocr-service/Dockerfile.dockerignore b/ocr-service/Dockerfile.dockerignore new file mode 100644 index 0000000..fc11f50 --- /dev/null +++ b/ocr-service/Dockerfile.dockerignore @@ -0,0 +1,5 @@ +** +!ocr-service/ +!ocr-service/app/ +!ocr-service/app/** +!ocr-service/requirements.txt diff --git a/ocr-service/README.md b/ocr-service/README.md index 9e27407..d9acbb2 100644 --- a/ocr-service/README.md +++ b/ocr-service/README.md @@ -2,13 +2,11 @@ Unit 5 uses the repository-local virtual environment `ocr-service/.venv`. It is intentionally retained between runs and ignored by Git. Do not install these dependencies globally. -## Create or recreate +## Install From the repository root: ```bash -rm -rf ocr-service/.venv -python3 -m venv ocr-service/.venv ocr-service/.venv/bin/python -m pip install -r ocr-service/requirements.txt ``` @@ -17,7 +15,13 @@ Activate it with `source ocr-service/.venv/bin/activate`. ## Test ```bash -ocr-service/.venv/bin/python -m pytest ocr-service/tests -k auth +ocr-service/.venv/bin/python -m pytest ocr-service/tests ``` -PaddleOCR, rendering, container images, and production model readiness belong to Unit 6 and are not installed by this manifest. +The container build downloads the pinned PaddleOCR models. Runtime startup loads those baked models before readiness can report healthy. Deploy one replica with a maximum of 3 CPU and 5 GiB RAM; Docker image labels document these platform-enforced limits. + +Build from the repository root so the Dockerfile can use the service-scoped paths: + +```bash +docker build -f ocr-service/Dockerfile -t rag-ocr-service . +``` diff --git a/ocr-service/app/engine.py b/ocr-service/app/engine.py new file mode 100644 index 0000000..9b72080 --- /dev/null +++ b/ocr-service/app/engine.py @@ -0,0 +1,48 @@ +from dataclasses import dataclass +from typing import Any, Protocol + + +@dataclass(frozen=True) +class EngineLine: + text: str + confidence: float + bbox: tuple[int, int, int, int] + + +class OcrEngine(Protocol): + def recognize(self, image: object) -> list[EngineLine]: ... + + +class PaddleOcrEngine: + def __init__(self, pipeline: Any | None = None) -> None: + self._injected = pipeline is not None + if pipeline is None: + from paddleocr import PaddleOCR + + pipeline = PaddleOCR( + text_detection_model_name="PP-OCRv5_mobile_det", + text_recognition_model_name="latin_PP-OCRv5_mobile_rec", + use_doc_orientation_classify=False, + use_doc_unwarping=False, + use_textline_orientation=False, + device="cpu", + ) + self.pipeline = pipeline + + def recognize(self, image: object) -> list[EngineLine]: + if not self._injected: + import numpy + + image = numpy.asarray(image) + lines: list[EngineLine] = [] + for prediction in self.pipeline.predict(image): + payload = prediction.json() if callable(prediction.json) else prediction.json + result = payload.get("res", payload) + texts = result.get("rec_texts", []) + scores = result.get("rec_scores", []) + boxes = result.get("rec_boxes", []) + if hasattr(boxes, "tolist"): + boxes = boxes.tolist() + for text, score, box in zip(texts, scores, boxes, strict=True): + lines.append(EngineLine(str(text), float(score), tuple(int(value) for value in box))) + return lines diff --git a/ocr-service/app/main.py b/ocr-service/app/main.py index d610a28..08500f5 100644 --- a/ocr-service/app/main.py +++ b/ocr-service/app/main.py @@ -11,6 +11,8 @@ from typing import Annotated, Any from fastapi import Depends, FastAPI, File, Form, Header, HTTPException, Response, UploadFile +from .models import load_runtime_engine + MAX_UPLOAD_BYTES = 50 * 1024 * 1024 MAX_PAGES = 100 @@ -168,8 +170,10 @@ def create_app( return application +runtime_engine = load_runtime_engine() + app = create_app( os.getenv("OCR_INTERNAL_TOKEN", ""), os.getenv("OCR_JOBS_DB", ":memory:"), - engine_ready=os.getenv("OCR_ENGINE_READY") == "1", + engine_ready=runtime_engine is not None, ) diff --git a/ocr-service/app/models.py b/ocr-service/app/models.py new file mode 100644 index 0000000..c021dd1 --- /dev/null +++ b/ocr-service/app/models.py @@ -0,0 +1,17 @@ +import os + +from .engine import PaddleOcrEngine + + +def load_runtime_engine() -> PaddleOcrEngine | None: + if os.getenv("OCR_LOAD_ENGINE") != "1": + return None + try: + return PaddleOcrEngine() + except Exception: + return None + + +if __name__ == "__main__": + PaddleOcrEngine() + print("PaddleOCR models loaded") diff --git a/ocr-service/app/render.py b/ocr-service/app/render.py new file mode 100644 index 0000000..4848f52 --- /dev/null +++ b/ocr-service/app/render.py @@ -0,0 +1,123 @@ +import io +import math +import statistics +import time +from dataclasses import dataclass +from typing import Callable + +from .engine import EngineLine, OcrEngine + + +RENDER_DPI = 200 +MAX_RENDER_PIXELS = 25_000_000 + + +class PdfRenderError(ValueError): + pass + + +@dataclass(frozen=True) +class RenderedPage: + page: int + width: int + height: int + png: bytes + image: object + + +def render_pdf_pages(pdf: bytes, pages: list[int], max_pixels: int = MAX_RENDER_PIXELS) -> list[RenderedPage]: + import pypdfium2 + + if not pages or pages != sorted(set(pages)) or any(type(page) is not int or page < 1 for page in pages): + raise PdfRenderError("PDF pages must be unique, ordered, one-based integers") + try: + document = pypdfium2.PdfDocument(pdf) + except Exception as error: + raise PdfRenderError("PDF cannot be opened for deterministic rendering") from error + + rendered: list[RenderedPage] = [] + try: + for page_number in pages: + if page_number > len(document): + raise PdfRenderError(f"PDF page {page_number} does not exist") + page = document[page_number - 1] + try: + page_width, page_height = page.get_size() + width = math.ceil(page_width * RENDER_DPI / 72) + height = math.ceil(page_height * RENDER_DPI / 72) + if width * height > max_pixels: + raise PdfRenderError("Rendered page exceeds the 25 megapixels limit") + bitmap = page.render(scale=RENDER_DPI / 72) + try: + image = bitmap.to_pil() + output = io.BytesIO() + image.save(output, format="PNG") + rendered.append(RenderedPage(page_number, image.width, image.height, output.getvalue(), image)) + finally: + bitmap.close() + finally: + page.close() + finally: + document.close() + return rendered + + +def _metrics(lines: list[EngineLine], text: str) -> dict[str, int | float]: + confidences = sorted(line.confidence for line in lines) + return { + "lineCount": len(lines), + "nonWhitespaceCharacters": sum(not character.isspace() for character in text), + "medianConfidence": statistics.median(confidences) if confidences else 0.0, + "p10Confidence": confidences[math.floor((len(confidences) - 1) * 0.1)] if confidences else 0.0, + "lowConfidenceLineRatio": sum(value < 0.5 for value in confidences) / len(confidences) if confidences else 0.0, + } + + +def process_pdf( + job_id: str, + document_sha256: str, + pdf: bytes, + requested_pages: list[int], + engine: OcrEngine, + processing_ms: Callable[[int], int] | None = None, +) -> dict: + results = [] + for rendered in render_pdf_pages(pdf, requested_pages): + started = time.perf_counter_ns() + indexed = list(enumerate(engine.recognize(rendered.image), start=1)) + indexed.sort(key=lambda item: (item[1].bbox[1], item[1].bbox[0], item[0])) + lines = [line for _, line in indexed] + text = "\n".join(line.text for line in lines) + serialized_lines = [ + { + "lineId": f"p{rendered.page}-l{index:02d}-{'-'.join(map(str, line.bbox))}", + "text": line.text, + "confidence": line.confidence, + "bbox": list(line.bbox), + } + for index, line in enumerate(lines, start=1) + ] + elapsed = math.ceil((time.perf_counter_ns() - started) / 1_000_000) + results.append({ + "page": rendered.page, + "width": rendered.width, + "height": rendered.height, + "processingMs": processing_ms(rendered.page) if processing_ms else elapsed, + "text": text, + "metrics": _metrics(lines, text), + "lines": serialized_lines, + }) + return { + "schemaVersion": "1", + "jobId": job_id, + "documentSha256": document_sha256, + "engine": { + "name": "paddleocr", + "version": "3.4.0", + "runtime": "paddlepaddle-3.2.2", + "device": "cpu", + "configVersion": "ocr-v1", + "dpi": RENDER_DPI, + }, + "pages": results, + } diff --git a/ocr-service/requirements.txt b/ocr-service/requirements.txt index 2b08222..77747b9 100644 --- a/ocr-service/requirements.txt +++ b/ocr-service/requirements.txt @@ -1,5 +1,10 @@ fastapi==0.116.1 httpx==0.28.1 +numpy==2.2.6 +paddleocr==3.4.0 +paddlepaddle==3.2.2 +Pillow==11.3.0 pytest==8.4.1 +pypdfium2==4.30.0 python-multipart==0.0.20 uvicorn==0.35.0 diff --git a/ocr-service/tests/test_render.py b/ocr-service/tests/test_render.py new file mode 100644 index 0000000..628e622 --- /dev/null +++ b/ocr-service/tests/test_render.py @@ -0,0 +1,147 @@ +import hashlib +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).parents[1])) + +from app.engine import EngineLine, PaddleOcrEngine +from app.render import PdfRenderError, process_pdf, render_pdf_pages + + +FIXTURE = Path(__file__).parents[2] / "tests" / "fixtures" / "ocr" / "native-three-pages.pdf" +DOCUMENT_SHA256 = hashlib.sha256(FIXTURE.read_bytes()).hexdigest() + + +class DeterministicEngine: + def __init__(self) -> None: + self.calls = 0 + + def recognize(self, _image: object) -> list[EngineLine]: + self.calls += 1 + if self.calls % 2 == 1: + return [ + EngineLine("FATo7", 0.98, (120, 340, 245, 372)), + EngineLine("second line", 0.74, (80, 410, 300, 450)), + ] + return [EngineLine("page three", 0.91, (50, 60, 250, 100))] + + +def test_render_produces_selected_200_dpi_png_pages() -> None: + pages = render_pdf_pages(FIXTURE.read_bytes(), [1, 3]) + + assert [(page.page, page.width, page.height) for page in pages] == [ + (1, 1700, 2200), + (3, 1700, 2200), + ] + assert all(page.png.startswith(b"\x89PNG\r\n\x1a\n") for page in pages) + + +def test_render_rejects_invalid_pages_and_pixel_limit_before_image_creation() -> None: + with pytest.raises(PdfRenderError, match="one-based"): + render_pdf_pages(FIXTURE.read_bytes(), [0]) + with pytest.raises(PdfRenderError, match="25 megapixels"): + render_pdf_pages(FIXTURE.read_bytes(), [1], max_pixels=1_000_000) + + +def test_render_builds_repeatable_result_schema_with_deterministic_engine() -> None: + def run() -> dict: + return process_pdf( + job_id="ocr_repeatable", + document_sha256=DOCUMENT_SHA256, + pdf=FIXTURE.read_bytes(), + requested_pages=[1, 3], + engine=DeterministicEngine(), + processing_ms=lambda _page: 17, + ) + + first = run() + assert first == run() + assert first["schemaVersion"] == "1" + assert first["documentSha256"] == DOCUMENT_SHA256 + assert first["engine"] == { + "name": "paddleocr", + "version": "3.4.0", + "runtime": "paddlepaddle-3.2.2", + "device": "cpu", + "configVersion": "ocr-v1", + "dpi": 200, + } + assert first["pages"][0] == { + "page": 1, + "width": 1700, + "height": 2200, + "processingMs": 17, + "text": "FATo7\nsecond line", + "metrics": { + "lineCount": 2, + "nonWhitespaceCharacters": 15, + "medianConfidence": 0.86, + "p10Confidence": 0.74, + "lowConfidenceLineRatio": 0.0, + }, + "lines": [ + { + "lineId": "p1-l01-120-340-245-372", + "text": "FATo7", + "confidence": 0.98, + "bbox": [120, 340, 245, 372], + }, + { + "lineId": "p1-l02-80-410-300-450", + "text": "second line", + "confidence": 0.74, + "bbox": [80, 410, 300, 450], + }, + ], + } + assert [page["page"] for page in first["pages"]] == [1, 3] + + +def test_render_paddle_adapter_preserves_text_confidence_and_boxes() -> None: + class Prediction: + json = { + "res": { + "rec_texts": ["CBGO4a", "NSAvo6"], + "rec_scores": [0.97, 0.83], + "rec_boxes": [[10, 20, 110, 50], [15, 70, 130, 100]], + } + } + + class Pipeline: + def predict(self, _image: object) -> list[Prediction]: + return [Prediction()] + + lines = PaddleOcrEngine(pipeline=Pipeline()).recognize(object()) + + assert lines == [ + EngineLine("CBGO4a", 0.97, (10, 20, 110, 50)), + EngineLine("NSAvo6", 0.83, (15, 70, 130, 100)), + ] + + +def test_render_container_pins_cpu_runtime_models_and_single_worker() -> None: + service_root = Path(__file__).parents[1] + requirements = (service_root / "requirements.txt").read_text() + dockerfile = (service_root / "Dockerfile").read_text() + dockerignore = (service_root / "Dockerfile.dockerignore").read_text().splitlines() + model_loader = (service_root / "app" / "models.py").read_text() + + assert "paddleocr==3.4.0" in requirements + assert "paddlepaddle==3.2.2" in requirements + assert "pypdfium2==4.30.0" in requirements + assert "RUN python -m app.models" in dockerfile + assert 'OCR_JOBS_DB="/data/jobs/jobs.db"' in dockerfile + assert '"--workers", "1"' in dockerfile + assert "USER ocr" in dockerfile + assert 'resource.cpu.max="3"' in dockerfile + assert 'resource.memory.max="5GiB"' in dockerfile + assert "PaddleOcrEngine" in model_loader + assert dockerignore == [ + "**", + "!ocr-service/", + "!ocr-service/app/", + "!ocr-service/app/**", + "!ocr-service/requirements.txt", + ] diff --git a/openspec/changes/ocr-ingest-integration/apply-progress.md b/openspec/changes/ocr-ingest-integration/apply-progress.md index e861dae..cd57eda 100644 --- a/openspec/changes/ocr-ingest-integration/apply-progress.md +++ b/openspec/changes/ocr-ingest-integration/apply-progress.md @@ -3,9 +3,9 @@ ## Current State - **Mode:** Strict TDD -- **Delivery:** Feature-branch-chain, Units 1–5 complete; maintainer-approved `size:exception` for Unit 2 only -- **Completed tasks:** 1.1, 1.2, 1.3, 1.4, 2.1, 2.2, 2.3, 2.4, 2.5, 3.1, 3.2 -- **Overall task progress:** 11/30 complete +- **Delivery:** Feature-branch-chain, Units 1–6 complete; maintainer-approved `size:exception` for Unit 2 only +- **Completed tasks:** 1.1, 1.2, 1.3, 1.4, 2.1, 2.2, 2.3, 2.4, 2.5, 3.1, 3.2, 3.3, 3.4 +- **Overall task progress:** 13/30 complete ## Unit 1: Migration @@ -191,3 +191,36 @@ None — the migration follows the proposal, specifications, design, and closed - Cleanup: Uvicorn PID 268425 terminated and absent; temporary PDF, SQLite DB, and log removed; `.venv` retained. - Rollback: remove `.gitignore` Unit 5 entries and `ocr-service/`, then revert tasks 3.1/3.2 and this Unit 5 progress/history block; Units 1–4 remain intact. - Fresh evidence revision: `sha256:0d8b2c57bbab967473eaf99a7ca900168554e64254fa80197b3626b4570b58d0`; corrected final authored change count: 397 additions plus deletions, within 400. + +## Unit 6: OCR Rendering and Image + +### Implementation Summary + +- Added deterministic 200 DPI PDF page rendering with a pre-render 25-megapixel guard, a PaddleOCR adapter, stable line identities, quality metrics, and the exact result schema. +- Added a deterministic fake-driven test seam and preserved ambiguous OCR tokens without correction. +- Added the private CPU image with PaddleOCR `3.4.0`, PaddlePaddle CPU `3.2.2`, baked Latin/mobile models, one Uvicorn worker, a non-root user, private job storage, healthcheck, and documented 3 CPU/5 GiB deployment limits. +- Added a Dockerfile-specific deny-by-default build context so restricted and unrelated repository paths are never sent to the Docker daemon. + +### TDD Cycle Evidence + +| Task | Test File | Layer | Safety Net | RED | GREEN | TRIANGULATE | REFACTOR | +|---|---|---|---|---|---|---|---| +| 3.3 | `ocr-service/tests/test_render.py` | Unit/integration fixture | Existing OCR API suite passed 8/8 before production edits. | Focused collection failed because `app.engine` did not exist. | Focused render suite passed 5/5; complete OCR suite passed 13/13. | Selected pages 1/3, invalid page and pixel-limit paths, repeatable fake results, and Paddle payload normalization exercise distinct behavior. | Named constants and pure metric/result transforms retained; focused suite remained 5/5. | +| 3.4 | `ocr-service/tests/test_render.py` | Static/container runtime | Existing OCR API suite passed 8/8 before production edits. | Docker contract was absent; deny-by-default context triangulation then failed until `Dockerfile.dockerignore` existed. | Image built with baked models; offline constrained container rendered and validated PNG/result output. | Static pin/runtime assertions plus real model build and `--network none` execution cover configuration and runtime paths. | Added missing OpenCV runtime libraries after the first build exposed `libGL.so.1`; exact corrected image passed. | + +### Work Unit Evidence + +| Evidence | Result | +|---|---| +| Focused test command and exact result | `ocr-service/.venv/bin/python -m pytest ocr-service/tests -k render -q` — exit 0; 5 passed, 8 deselected, 1 dependency deprecation warning. | +| Runtime harness command/scenario and exact result | `docker build --file ocr-service/Dockerfile --tag rag-ocr-service:unit6 .` — corrected exact image exited 0 and baked both models. A `docker run --rm --network none --cpus 3 --memory 5g ...` harness loaded cached PaddleOCR 3.4.0/PaddlePaddle 3.2.2 models, rendered a 1700x2200 PNG (23,309 bytes), returned schema `1`, one page, and one OCR line; host Pillow reopened it as PNG 1700x2200. | +| Rollback boundary | Remove `ocr-service/{Dockerfile,Dockerfile.dockerignore,app/engine.py,app/models.py,app/render.py,tests/test_render.py}` and revert Unit 6 changes in `app/main.py`, `requirements.txt`, and `README.md`; Units 1–5 remain intact. | + +### Validation and Boundary + +- RED was followed by GREEN and post-refactor reruns. One expected GREEN iteration corrected a test-side non-whitespace count from 16 to the mathematically correct 15. +- The first image build failed on missing `libGL.so.1`; adding the required slim-image runtime libraries produced a successful build and runtime harness. A later optional connectivity-check refactor rebuild exhausted Docker storage; that unvalidated one-line refactor was reverted, the validated Dockerfile bytes were restored, and the image tag was removed during cleanup. +- `npm test` passed 45/45; `npm run check`, Python `compileall`, tracked/untracked whitespace checks, and the full OCR pytest suite (13/13) passed. +- Start: Unit 5 private queue/API is complete. End: Unit 6 render/engine/schema/container behavior is complete. Unit 7 client work was not started. +- Native Unit 6 implementation/tests/docs are 397 authored additions plus deletions, within the 400-line budget; required SDD progress and workspace history metadata are administrative evidence outside that native slice. +- No specification or design deviation. The Docker image is private by deployment contract; CPU/RAM labels document limits that the platform must enforce. diff --git a/openspec/changes/ocr-ingest-integration/tasks.md b/openspec/changes/ocr-ingest-integration/tasks.md index 9bf49e1..844da40 100644 --- a/openspec/changes/ocr-ingest-integration/tasks.md +++ b/openspec/changes/ocr-ingest-integration/tasks.md @@ -46,8 +46,8 @@ Tracker feature/ocr-ingest-integration is draft/no-merge and sole main target. P - [x] 3.1 RED HTTP: bearer 401; idempotency/conflict 409; allowlist; limits 413/422; pressure 429; no-retry; integrity mismatch - [x] 3.2 GREEN: `ocr-service/` API, queue, allowlist, limits, health -- [ ] 3.3 GREEN: PaddleOCR 3.4.0, 200 DPI, schema, stub -- [ ] 3.4 GREEN: `ocr-service/Dockerfile` pinned wheels/models/resources +- [x] 3.3 GREEN: PaddleOCR 3.4.0, 200 DPI, schema, stub +- [x] 3.4 GREEN: `ocr-service/Dockerfile` pinned wheels/models/resources ## 4 Orchestration (Units 7–8; O1–O4)