feat(ocr): extract ordered PDF pages
This commit is contained in:
parent
d2ecf19102
commit
905f8854db
4 changed files with 222 additions and 3 deletions
36
scripts/spike-pdfjs.ts
Normal file
36
scripts/spike-pdfjs.ts
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { readFile } from "node:fs/promises";
|
||||
import pdf from "pdf-parse";
|
||||
|
||||
interface PdfPageData {
|
||||
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||
items: Array<{ str?: string }>;
|
||||
}>;
|
||||
}
|
||||
|
||||
const fixtureUrl = new URL("../tests/fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||
const input = process.argv[2] ? new URL(`file://${process.argv[2]}`) : fixtureUrl;
|
||||
const nodeMajor = Number.parseInt(process.versions.node.split(".")[0] ?? "0", 10);
|
||||
|
||||
assert.ok(nodeMajor >= 22, `Node 22 or newer is required; received ${process.versions.node}`);
|
||||
|
||||
const pages: Array<{ page: number; text: string }> = [];
|
||||
const bytes = Uint8Array.from(await readFile(input));
|
||||
const result = await pdf(bytes as Buffer, {
|
||||
version: "v2.0.550",
|
||||
pagerender: async (pageData: PdfPageData) => {
|
||||
const textContent = await pageData.getTextContent({ normalizeWhitespace: false, disableCombineTextItems: false });
|
||||
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||
pages.push({ page: pages.length + 1, text });
|
||||
return text;
|
||||
}
|
||||
});
|
||||
|
||||
assert.equal(result.numpages, pages.length);
|
||||
assert.deepEqual(pages, [
|
||||
{ page: 1, text: "Native page one." },
|
||||
{ page: 2, text: "" },
|
||||
{ page: 3, text: "Native page three." }
|
||||
]);
|
||||
|
||||
console.log(JSON.stringify({ node: process.versions.node, parser: `pdf-parse/pdfjs-${result.version}`, pages }));
|
||||
|
|
@ -1,7 +1,9 @@
|
|||
import { readFile } from "node:fs/promises";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import path from "node:path";
|
||||
import pdf from "pdf-parse";
|
||||
import type { ChunkingMode } from "../process/chunking.js";
|
||||
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||
|
||||
export interface ParsedDocument {
|
||||
title: string;
|
||||
|
|
@ -10,6 +12,18 @@ export interface ParsedDocument {
|
|||
chunkMode: ChunkingMode;
|
||||
}
|
||||
|
||||
export interface ParsedPdfPage {
|
||||
page: number;
|
||||
text: string;
|
||||
textSha256: string;
|
||||
}
|
||||
|
||||
interface PdfPageData {
|
||||
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||
items: Array<{ str?: string }>;
|
||||
}>;
|
||||
}
|
||||
|
||||
const documentalExtensions = [".md", ".txt", ".pdf"] as const;
|
||||
const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const;
|
||||
const parserExtensions = [...documentalExtensions, ...codeExtensions] as const;
|
||||
|
|
@ -19,6 +33,9 @@ export function supportedParserExtensions(): string[] {
|
|||
}
|
||||
|
||||
export function isSupportedDocument(filePath: string): boolean {
|
||||
if (path.basename(filePath).toLowerCase() === "cmakelists.txt") {
|
||||
return false;
|
||||
}
|
||||
return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]);
|
||||
}
|
||||
|
||||
|
|
@ -43,16 +60,42 @@ function inferMimeType(extension: string, chunkMode: ChunkingMode): string {
|
|||
return "text/plain";
|
||||
}
|
||||
|
||||
export async function parsePdfPages(filePath: string | URL): Promise<ParsedPdfPage[]> {
|
||||
const resolvedPath = filePath instanceof URL ? fileURLToPath(filePath) : filePath;
|
||||
if (path.extname(resolvedPath).toLowerCase() !== ".pdf") {
|
||||
throw new Error("Only PDF documents support page extraction");
|
||||
}
|
||||
|
||||
const pages: ParsedPdfPage[] = [];
|
||||
const bytes = Uint8Array.from(await readFile(filePath));
|
||||
const result = await pdf(bytes as Buffer, {
|
||||
version: "v2.0.550",
|
||||
pagerender: async (pageData: PdfPageData) => {
|
||||
const textContent = await pageData.getTextContent({
|
||||
normalizeWhitespace: false,
|
||||
disableCombineTextItems: false
|
||||
});
|
||||
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||
pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) });
|
||||
return text;
|
||||
}
|
||||
});
|
||||
|
||||
if (result.numrender !== result.numpages || pages.length !== result.numpages) {
|
||||
throw new Error("PDF page extraction did not render every page");
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
|
||||
export async function parseDocument(filePath: string): Promise<ParsedDocument> {
|
||||
const extension = path.extname(filePath).toLowerCase();
|
||||
const chunkMode = inferChunkMode(filePath);
|
||||
|
||||
if (extension === ".pdf") {
|
||||
const buffer = await readFile(filePath);
|
||||
const result = await pdf(buffer);
|
||||
const pages = await parsePdfPages(filePath);
|
||||
return {
|
||||
title: path.basename(filePath),
|
||||
content: result.text.trim(),
|
||||
content: pages.map((page) => page.text).filter(Boolean).join("\n\n"),
|
||||
mimeType: inferMimeType(extension, chunkMode),
|
||||
chunkMode
|
||||
};
|
||||
|
|
|
|||
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
%PDF-1.4
|
||||
1 0 obj
|
||||
<< /Type /Catalog /Pages 2 0 R >>
|
||||
endobj
|
||||
2 0 obj
|
||||
<< /Type /Pages /Kids [4 0 R 5 0 R 6 0 R] /Count 3 >>
|
||||
endobj
|
||||
3 0 obj
|
||||
<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>
|
||||
endobj
|
||||
4 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 7 0 R >>
|
||||
endobj
|
||||
5 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 8 0 R >>
|
||||
endobj
|
||||
6 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 9 0 R >>
|
||||
endobj
|
||||
7 0 obj
|
||||
<< /Length 48 >>
|
||||
stream
|
||||
BT
|
||||
/F1 12 Tf
|
||||
72 720 Td
|
||||
(Native page one.) Tj
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
8 0 obj
|
||||
<< /Length 6 >>
|
||||
stream
|
||||
BT
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
9 0 obj
|
||||
<< /Length 50 >>
|
||||
stream
|
||||
BT
|
||||
/F1 12 Tf
|
||||
72 720 Td
|
||||
(Native page three.) Tj
|
||||
ET
|
||||
endstream
|
||||
endobj
|
||||
xref
|
||||
0 10
|
||||
0000000000 65535 f
|
||||
0000000009 00000 n
|
||||
0000000058 00000 n
|
||||
0000000127 00000 n
|
||||
0000000197 00000 n
|
||||
0000000323 00000 n
|
||||
0000000449 00000 n
|
||||
0000000575 00000 n
|
||||
0000000672 00000 n
|
||||
0000000726 00000 n
|
||||
trailer
|
||||
<< /Size 10 /Root 1 0 R >>
|
||||
startxref
|
||||
825
|
||||
%%EOF
|
||||
77
tests/parsers/pdf-pages.test.ts
Normal file
77
tests/parsers/pdf-pages.test.ts
Normal file
|
|
@ -0,0 +1,77 @@
|
|||
import assert from "node:assert/strict";
|
||||
import { access, mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import os from "node:os";
|
||||
import path from "node:path";
|
||||
import test from "node:test";
|
||||
import {
|
||||
isSupportedDocument,
|
||||
parseDocument,
|
||||
parsePdfPages
|
||||
} from "../../src/modules/parsers/parser-registry.js";
|
||||
|
||||
const fixturePath = new URL("../fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||
|
||||
test("parsePdfPages extracts ordered one-based pages with stable native text hashes", async () => {
|
||||
const pages = await parsePdfPages(fixturePath);
|
||||
|
||||
assert.deepEqual(pages, [
|
||||
{
|
||||
page: 1,
|
||||
text: "Native page one.",
|
||||
textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4"
|
||||
},
|
||||
{
|
||||
page: 2,
|
||||
text: "",
|
||||
textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
|
||||
},
|
||||
{
|
||||
page: 3,
|
||||
text: "Native page three.",
|
||||
textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e"
|
||||
}
|
||||
]);
|
||||
});
|
||||
|
||||
test("parseDocument preserves the synchronous native PDF content contract", async () => {
|
||||
const parsed = await parseDocument(fixturePath.pathname);
|
||||
|
||||
assert.equal(parsed.title, "native-three-pages.pdf");
|
||||
assert.equal(parsed.content, "Native page one.\n\nNative page three.");
|
||||
assert.equal(parsed.mimeType, "application/pdf");
|
||||
assert.equal(parsed.chunkMode, "documental");
|
||||
});
|
||||
|
||||
test("page extraction rejects non-PDF inputs before parsing bytes", async () => {
|
||||
await assert.rejects(
|
||||
parsePdfPages(new URL("../fixtures/ocr/not-a-pdf.txt", import.meta.url)),
|
||||
/only PDF documents support page extraction/i
|
||||
);
|
||||
});
|
||||
|
||||
test("supported text documents are read as data and never executed", async () => {
|
||||
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-parser-threat-"));
|
||||
const markerPath = path.join(directory, "executed");
|
||||
const payload = `$(touch ${markerPath})`;
|
||||
|
||||
try {
|
||||
for (const fileName of ["requirements.txt", "executable.md"]) {
|
||||
const filePath = path.join(directory, fileName);
|
||||
await writeFile(filePath, payload, "utf8");
|
||||
const parsed = await parseDocument(filePath);
|
||||
|
||||
assert.equal(isSupportedDocument(filePath), true);
|
||||
assert.equal(parsed.content, payload);
|
||||
}
|
||||
await assert.rejects(access(markerPath), /ENOENT/);
|
||||
} finally {
|
||||
await rm(directory, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("build and executable lookalike formats remain unsupported", () => {
|
||||
assert.deepEqual(
|
||||
["CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument),
|
||||
[false, false, false]
|
||||
);
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue