feat(ocr): extract ordered PDF pages
This commit is contained in:
parent
d2ecf19102
commit
905f8854db
4 changed files with 222 additions and 3 deletions
36
scripts/spike-pdfjs.ts
Normal file
36
scripts/spike-pdfjs.ts
Normal file
|
|
@ -0,0 +1,36 @@
|
||||||
|
import assert from "node:assert/strict";
|
||||||
|
import { readFile } from "node:fs/promises";
|
||||||
|
import pdf from "pdf-parse";
|
||||||
|
|
||||||
|
interface PdfPageData {
|
||||||
|
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||||
|
items: Array<{ str?: string }>;
|
||||||
|
}>;
|
||||||
|
}
|
||||||
|
|
||||||
|
const fixtureUrl = new URL("../tests/fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||||
|
const input = process.argv[2] ? new URL(`file://${process.argv[2]}`) : fixtureUrl;
|
||||||
|
const nodeMajor = Number.parseInt(process.versions.node.split(".")[0] ?? "0", 10);
|
||||||
|
|
||||||
|
assert.ok(nodeMajor >= 22, `Node 22 or newer is required; received ${process.versions.node}`);
|
||||||
|
|
||||||
|
const pages: Array<{ page: number; text: string }> = [];
|
||||||
|
const bytes = Uint8Array.from(await readFile(input));
|
||||||
|
const result = await pdf(bytes as Buffer, {
|
||||||
|
version: "v2.0.550",
|
||||||
|
pagerender: async (pageData: PdfPageData) => {
|
||||||
|
const textContent = await pageData.getTextContent({ normalizeWhitespace: false, disableCombineTextItems: false });
|
||||||
|
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||||
|
pages.push({ page: pages.length + 1, text });
|
||||||
|
return text;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(result.numpages, pages.length);
|
||||||
|
assert.deepEqual(pages, [
|
||||||
|
{ page: 1, text: "Native page one." },
|
||||||
|
{ page: 2, text: "" },
|
||||||
|
{ page: 3, text: "Native page three." }
|
||||||
|
]);
|
||||||
|
|
||||||
|
console.log(JSON.stringify({ node: process.versions.node, parser: `pdf-parse/pdfjs-${result.version}`, pages }));
|
||||||
|
|
@ -1,7 +1,9 @@
|
||||||
import { readFile } from "node:fs/promises";
|
import { readFile } from "node:fs/promises";
|
||||||
|
import { fileURLToPath } from "node:url";
|
||||||
import path from "node:path";
|
import path from "node:path";
|
||||||
import pdf from "pdf-parse";
|
import pdf from "pdf-parse";
|
||||||
import type { ChunkingMode } from "../process/chunking.js";
|
import type { ChunkingMode } from "../process/chunking.js";
|
||||||
|
import { sha256Hex } from "../../shared/utils/ids.js";
|
||||||
|
|
||||||
export interface ParsedDocument {
|
export interface ParsedDocument {
|
||||||
title: string;
|
title: string;
|
||||||
|
|
@ -10,6 +12,18 @@ export interface ParsedDocument {
|
||||||
chunkMode: ChunkingMode;
|
chunkMode: ChunkingMode;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export interface ParsedPdfPage {
|
||||||
|
page: number;
|
||||||
|
text: string;
|
||||||
|
textSha256: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
interface PdfPageData {
|
||||||
|
getTextContent(options: { normalizeWhitespace: boolean; disableCombineTextItems: boolean }): Promise<{
|
||||||
|
items: Array<{ str?: string }>;
|
||||||
|
}>;
|
||||||
|
}
|
||||||
|
|
||||||
const documentalExtensions = [".md", ".txt", ".pdf"] as const;
|
const documentalExtensions = [".md", ".txt", ".pdf"] as const;
|
||||||
const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const;
|
const codeExtensions = [".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".py", ".json", ".yml", ".yaml"] as const;
|
||||||
const parserExtensions = [...documentalExtensions, ...codeExtensions] as const;
|
const parserExtensions = [...documentalExtensions, ...codeExtensions] as const;
|
||||||
|
|
@ -19,6 +33,9 @@ export function supportedParserExtensions(): string[] {
|
||||||
}
|
}
|
||||||
|
|
||||||
export function isSupportedDocument(filePath: string): boolean {
|
export function isSupportedDocument(filePath: string): boolean {
|
||||||
|
if (path.basename(filePath).toLowerCase() === "cmakelists.txt") {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]);
|
return parserExtensions.includes(path.extname(filePath).toLowerCase() as (typeof parserExtensions)[number]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -43,16 +60,42 @@ function inferMimeType(extension: string, chunkMode: ChunkingMode): string {
|
||||||
return "text/plain";
|
return "text/plain";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export async function parsePdfPages(filePath: string | URL): Promise<ParsedPdfPage[]> {
|
||||||
|
const resolvedPath = filePath instanceof URL ? fileURLToPath(filePath) : filePath;
|
||||||
|
if (path.extname(resolvedPath).toLowerCase() !== ".pdf") {
|
||||||
|
throw new Error("Only PDF documents support page extraction");
|
||||||
|
}
|
||||||
|
|
||||||
|
const pages: ParsedPdfPage[] = [];
|
||||||
|
const bytes = Uint8Array.from(await readFile(filePath));
|
||||||
|
const result = await pdf(bytes as Buffer, {
|
||||||
|
version: "v2.0.550",
|
||||||
|
pagerender: async (pageData: PdfPageData) => {
|
||||||
|
const textContent = await pageData.getTextContent({
|
||||||
|
normalizeWhitespace: false,
|
||||||
|
disableCombineTextItems: false
|
||||||
|
});
|
||||||
|
const text = textContent.items.flatMap((item) => item.str ?? []).join(" ").trim();
|
||||||
|
pages.push({ page: pages.length + 1, text, textSha256: sha256Hex(text) });
|
||||||
|
return text;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
if (result.numrender !== result.numpages || pages.length !== result.numpages) {
|
||||||
|
throw new Error("PDF page extraction did not render every page");
|
||||||
|
}
|
||||||
|
return pages;
|
||||||
|
}
|
||||||
|
|
||||||
export async function parseDocument(filePath: string): Promise<ParsedDocument> {
|
export async function parseDocument(filePath: string): Promise<ParsedDocument> {
|
||||||
const extension = path.extname(filePath).toLowerCase();
|
const extension = path.extname(filePath).toLowerCase();
|
||||||
const chunkMode = inferChunkMode(filePath);
|
const chunkMode = inferChunkMode(filePath);
|
||||||
|
|
||||||
if (extension === ".pdf") {
|
if (extension === ".pdf") {
|
||||||
const buffer = await readFile(filePath);
|
const pages = await parsePdfPages(filePath);
|
||||||
const result = await pdf(buffer);
|
|
||||||
return {
|
return {
|
||||||
title: path.basename(filePath),
|
title: path.basename(filePath),
|
||||||
content: result.text.trim(),
|
content: pages.map((page) => page.text).filter(Boolean).join("\n\n"),
|
||||||
mimeType: inferMimeType(extension, chunkMode),
|
mimeType: inferMimeType(extension, chunkMode),
|
||||||
chunkMode
|
chunkMode
|
||||||
};
|
};
|
||||||
|
|
|
||||||
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
63
tests/fixtures/ocr/native-three-pages.pdf
vendored
Normal file
|
|
@ -0,0 +1,63 @@
|
||||||
|
%PDF-1.4
|
||||||
|
1 0 obj
|
||||||
|
<< /Type /Catalog /Pages 2 0 R >>
|
||||||
|
endobj
|
||||||
|
2 0 obj
|
||||||
|
<< /Type /Pages /Kids [4 0 R 5 0 R 6 0 R] /Count 3 >>
|
||||||
|
endobj
|
||||||
|
3 0 obj
|
||||||
|
<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>
|
||||||
|
endobj
|
||||||
|
4 0 obj
|
||||||
|
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 7 0 R >>
|
||||||
|
endobj
|
||||||
|
5 0 obj
|
||||||
|
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 8 0 R >>
|
||||||
|
endobj
|
||||||
|
6 0 obj
|
||||||
|
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 3 0 R >> >> /Contents 9 0 R >>
|
||||||
|
endobj
|
||||||
|
7 0 obj
|
||||||
|
<< /Length 48 >>
|
||||||
|
stream
|
||||||
|
BT
|
||||||
|
/F1 12 Tf
|
||||||
|
72 720 Td
|
||||||
|
(Native page one.) Tj
|
||||||
|
ET
|
||||||
|
endstream
|
||||||
|
endobj
|
||||||
|
8 0 obj
|
||||||
|
<< /Length 6 >>
|
||||||
|
stream
|
||||||
|
BT
|
||||||
|
ET
|
||||||
|
endstream
|
||||||
|
endobj
|
||||||
|
9 0 obj
|
||||||
|
<< /Length 50 >>
|
||||||
|
stream
|
||||||
|
BT
|
||||||
|
/F1 12 Tf
|
||||||
|
72 720 Td
|
||||||
|
(Native page three.) Tj
|
||||||
|
ET
|
||||||
|
endstream
|
||||||
|
endobj
|
||||||
|
xref
|
||||||
|
0 10
|
||||||
|
0000000000 65535 f
|
||||||
|
0000000009 00000 n
|
||||||
|
0000000058 00000 n
|
||||||
|
0000000127 00000 n
|
||||||
|
0000000197 00000 n
|
||||||
|
0000000323 00000 n
|
||||||
|
0000000449 00000 n
|
||||||
|
0000000575 00000 n
|
||||||
|
0000000672 00000 n
|
||||||
|
0000000726 00000 n
|
||||||
|
trailer
|
||||||
|
<< /Size 10 /Root 1 0 R >>
|
||||||
|
startxref
|
||||||
|
825
|
||||||
|
%%EOF
|
||||||
77
tests/parsers/pdf-pages.test.ts
Normal file
77
tests/parsers/pdf-pages.test.ts
Normal file
|
|
@ -0,0 +1,77 @@
|
||||||
|
import assert from "node:assert/strict";
|
||||||
|
import { access, mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||||
|
import os from "node:os";
|
||||||
|
import path from "node:path";
|
||||||
|
import test from "node:test";
|
||||||
|
import {
|
||||||
|
isSupportedDocument,
|
||||||
|
parseDocument,
|
||||||
|
parsePdfPages
|
||||||
|
} from "../../src/modules/parsers/parser-registry.js";
|
||||||
|
|
||||||
|
const fixturePath = new URL("../fixtures/ocr/native-three-pages.pdf", import.meta.url);
|
||||||
|
|
||||||
|
test("parsePdfPages extracts ordered one-based pages with stable native text hashes", async () => {
|
||||||
|
const pages = await parsePdfPages(fixturePath);
|
||||||
|
|
||||||
|
assert.deepEqual(pages, [
|
||||||
|
{
|
||||||
|
page: 1,
|
||||||
|
text: "Native page one.",
|
||||||
|
textSha256: "b9efe3745cacd6c87189435d7b83374ee7e5f77fca9d52e8d4e76f4a12a419f4"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
page: 2,
|
||||||
|
text: "",
|
||||||
|
textSha256: "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
page: 3,
|
||||||
|
text: "Native page three.",
|
||||||
|
textSha256: "e50a3249ccbe373afaac6ed06c36692f7190b260f8c55f21b45e4a5066e9352e"
|
||||||
|
}
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("parseDocument preserves the synchronous native PDF content contract", async () => {
|
||||||
|
const parsed = await parseDocument(fixturePath.pathname);
|
||||||
|
|
||||||
|
assert.equal(parsed.title, "native-three-pages.pdf");
|
||||||
|
assert.equal(parsed.content, "Native page one.\n\nNative page three.");
|
||||||
|
assert.equal(parsed.mimeType, "application/pdf");
|
||||||
|
assert.equal(parsed.chunkMode, "documental");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("page extraction rejects non-PDF inputs before parsing bytes", async () => {
|
||||||
|
await assert.rejects(
|
||||||
|
parsePdfPages(new URL("../fixtures/ocr/not-a-pdf.txt", import.meta.url)),
|
||||||
|
/only PDF documents support page extraction/i
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("supported text documents are read as data and never executed", async () => {
|
||||||
|
const directory = await mkdtemp(path.join(os.tmpdir(), "rag-parser-threat-"));
|
||||||
|
const markerPath = path.join(directory, "executed");
|
||||||
|
const payload = `$(touch ${markerPath})`;
|
||||||
|
|
||||||
|
try {
|
||||||
|
for (const fileName of ["requirements.txt", "executable.md"]) {
|
||||||
|
const filePath = path.join(directory, fileName);
|
||||||
|
await writeFile(filePath, payload, "utf8");
|
||||||
|
const parsed = await parseDocument(filePath);
|
||||||
|
|
||||||
|
assert.equal(isSupportedDocument(filePath), true);
|
||||||
|
assert.equal(parsed.content, payload);
|
||||||
|
}
|
||||||
|
await assert.rejects(access(markerPath), /ENOENT/);
|
||||||
|
} finally {
|
||||||
|
await rm(directory, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
test("build and executable lookalike formats remain unsupported", () => {
|
||||||
|
assert.deepEqual(
|
||||||
|
["CMakeLists.txt", "component.mdx", "README.sh"].map(isSupportedDocument),
|
||||||
|
[false, false, false]
|
||||||
|
);
|
||||||
|
});
|
||||||
Loading…
Add table
Reference in a new issue