feat(ocr): extract ordered PDF pages

This commit is contained in:
Paco POR-CORREO 2026-09-14 14:20:52 +02:00
parent d2ecf19102
commit 905f8854db
4 changed files with 222 additions and 3 deletions

36
scripts/spike-pdfjs.ts Normal file
View file

@ -0,0 +1,36 @@
import assert from "node:assert/strict";
import { readFile } from "node:fs/promises";
import pdf from "pdf-parse";
interface PdfPageData {
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
items: Array<{ str?: string }>;
}>;
}
const fixtureUrl = new URL("../tests/fixtures/ocr/native-three-pages.pdf", import.meta.url);
const input = process.argv[2] ? new URL(`file://${process.argv[2]}`) : fixtureUrl;
const nodeMajor = Number.parseInt(process.versions.node.split(".")[0] ?? "0", 10);
assert.ok(nodeMajor >= 22, `Node 22 or newer is required; received ${process.versions.node}`);
const pages: Array<{ page: number; text: string }> = [];
const bytes = Uint8Array.from(await readFile(input));
const result = await pdf(bytes as Buffer, {
version: "v2.0.550",
pagerender: async (pageData: PdfPageData) => {
const textContent = await pageData.getTextContent({ normalizeWhitespace: false, disableCombineTextItems: false });
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
pages.push({ page: pages.length + 1, text });
return text;
}
});
assert.equal(result.numpages, pages.length);
assert.deepEqual(pages, [
{ page: 1, text: "Native page one." },
{ page: 2, text: "" },
{ page: 3, text: "Native page three." }
]);
console.log(JSON.stringify({ node: process.versions.node, parser: `pdf-parse/pdfjs-${result.version}`, pages }));

View file

@ -1,7 +1,9 @@
import { readFile } from "node:fs/promises"; import { readFile } from "node:fs/promises";
import { fileURLToPath } from "node:url";
import path from "node:path"; import path from "node:path";
import pdf from "pdf-parse"; import pdf from "pdf-parse";
import type { ChunkingMode } from "../process/chunking.js"; import type { ChunkingMode } from "../process/chunking.js";
import { sha256Hex } from "../../shared/utils/ids.js";
export interface ParsedDocument { export interface ParsedDocument {
title: string; title: string;
@ -10,6 +12,18 @@ export interface ParsedDocument {
chunkMode: ChunkingMode; chunkMode: ChunkingMode;
} }
export interface ParsedPdfPage {
page: number;
text: string;
textSha256: string;
}
interface PdfPageData {
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
items: Array<{ str?: string }>;
}>;
}
const documentalExtensions = [".md", ".txt", ".pdf"] as const; const documentalExtensions = [".md", ".txt", ".pdf"] as const;
const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const; const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const;
const parserExtensions = [...documentalExtensions, ...codeExtensions] as const; const parserExtensions = [...documentalExtensions, ...codeExtensions] as const;
@ -19,6 +33,9 @@ export function supportedParserExtensions(): string[] {
} }
export function isSupportedDocument(filePath: string): boolean { export function isSupportedDocument(filePath: string): boolean {
if (path.basename(filePath).toLowerCase() === "cmakelists.txt") {
return false;
}
return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]); return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]);
} }
@ -43,16 +60,42 @@ function inferMimeType(extension: string, chunkMode: ChunkingMode): string {
return "text/plain"; return "text/plain";
} }
export async function parsePdfPages(filePath: string | URL): Promise<ParsedPdfPage[]> {
const resolvedPath = filePath instanceof URL ? fileURLToPath(filePath) : filePath;
if (path.extname(resolvedPath).toLowerCase() !== ".pdf") {
throw new Error("Only PDF documents support page extraction");
}
const pages: ParsedPdfPage[] = [];
const bytes = Uint8Array.from(await readFile(filePath));
const result = await pdf(bytes as Buffer, {
version: "v2.0.550",
pagerender: async (pageData: PdfPageData) => {
const textContent = await pageData.getTextContent({
normalizeWhitespace: false,
disableCombineTextItems: false
});
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) });
return text;
}
});
if (result.numrender !== result.numpages || pages.length !== result.numpages) {
throw new Error("PDF page extraction did not render every page");
}
return pages;
}
export async function parseDocument(filePath: string): Promise<ParsedDocument> { export async function parseDocument(filePath: string): Promise<ParsedDocument> {
const extension = path.extname(filePath).toLowerCase(); const extension = path.extname(filePath).toLowerCase();
const chunkMode = inferChunkMode(filePath); const chunkMode = inferChunkMode(filePath);
if (extension === ".pdf") { if (extension === ".pdf") {
const buffer = await readFile(filePath); const pages = await parsePdfPages(filePath);
const result = await pdf(buffer);
return { return {
title: path.basename(filePath), title: path.basename(filePath),
content: result.text.trim(), content: pages.map((page) => page.text).filter(Boolean).join("\n\n"),
mimeType: inferMimeType(extension, chunkMode), mimeType: inferMimeType(extension, chunkMode),
chunkMode chunkMode
}; };

View file

@ -0,0 +1,63 @@
%PDF-1.4
1 0 obj
<< /Type /Catalog /Pages 2 0 R >>
endobj
2 0 obj
<< /Type /Pages /Kids [4 0 R 5 0 R 6 0 R] /Count 3 >>
endobj
3 0 obj
<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>
endobj
4 0 obj
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 7 0 R >>
endobj
5 0 obj
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 8 0 R >>
endobj
6 0 obj
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 9 0 R >>
endobj
7 0 obj
<< /Length 48 >>
stream
BT
/F1 12 Tf
72 720 Td
(Native page one.) Tj
ET
endstream
endobj
8 0 obj
<< /Length 6 >>
stream
BT
ET
endstream
endobj
9 0 obj
<< /Length 50 >>
stream
BT
/F1 12 Tf
72 720 Td
(Native page three.) Tj
ET
endstream
endobj
xref
0 10
0000000000 65535 f
0000000009 00000 n
0000000058 00000 n
0000000127 00000 n
0000000197 00000 n
0000000323 00000 n
0000000449 00000 n
0000000575 00000 n
0000000672 00000 n
0000000726 00000 n
trailer
<< /Size 10 /Root 1 0 R >>
startxref
825
%%EOF

View file

@ -0,0 +1,77 @@
import assert from "node:assert/strict";
import { access, mkdtemp, rm, writeFile } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import test from "node:test";
import {
isSupportedDocument,
parseDocument,
parsePdfPages
} from "../../src/modules/parsers/parser-registry.js";
const fixturePath = new URL("../fixtures/ocr/native-three-pages.pdf", import.meta.url);
test("parsePdfPages extracts ordered one-based pages with stable native text hashes", async () => {
const pages = await parsePdfPages(fixturePath);
assert.deepEqual(pages, [
{
page: 1,
text: "Native page one.",
textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4"
},
{
page: 2,
text: "",
textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
{
page: 3,
text: "Native page three.",
textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e"
}
]);
});
test("parseDocument preserves the synchronous native PDF content contract", async () => {
const parsed = await parseDocument(fixturePath.pathname);
assert.equal(parsed.title, "native-three-pages.pdf");
assert.equal(parsed.content, "Native page one.\n\nNative page three.");
assert.equal(parsed.mimeType, "application/pdf");
assert.equal(parsed.chunkMode, "documental");
});
test("page extraction rejects non-PDF inputs before parsing bytes", async () => {
await assert.rejects(
parsePdfPages(new URL("../fixtures/ocr/not-a-pdf.txt", import.meta.url)),
/only PDF documents support page extraction/i
);
});
test("supported text documents are read as data and never executed", async () => {
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-parser-threat-"));
const markerPath = path.join(directory, "executed");
const payload = `$(touch ${markerPath})`;
try {
for (const fileName of ["requirements.txt", "executable.md"]) {
const filePath = path.join(directory, fileName);
await writeFile(filePath, payload, "utf8");
const parsed = await parseDocument(filePath);
assert.equal(isSupportedDocument(filePath), true);
assert.equal(parsed.content, payload);
}
await assert.rejects(access(markerPath), /ENOENT/);
} finally {
await rm(directory, { recursive: true, force: true });
}
});
test("build and executable lookalike formats remain unsupported", () => {
assert.deepEqual(
["CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument),
[false, false, false]
);
});