mirror of
https://github.com/openclaw/openclaw.git
synced 2026-10-03 01:29:56 +00:00
* fix(pdf): surface partial document extraction * refactor(pdf): consolidate truncation reporting * test(pdf): assert extraction metadata in dispatch e2e * fix(openresponses): preserve file whitespace * fix(media): validate extraction metadata * fix(media): validate document page completeness * test(pdf): keep completeness proof in the worker suite * fix(pdf): recover extraction after worker overload * fix(pdf): sort page ranges before bounded selection * test(pdf): align extraction fixtures with page budgets Co-authored-by: vyctorbrzezowski <51521767+vyctorbrzezowski@users.noreply.github.com> * test(pdf): prove empty page selections through the real worker Replace the mocked empty-selection assertion with actual PDFium worker coverage while preserving invalid-page and cleanup assertions. Co-authored-by: vyctorbrzezowski <51521767+vyctorbrzezowski@users.noreply.github.com> * test(media): isolate document context regression coverage Move real attachment truncation and input immutability checks into the focused file-context suite without growing the grandfathered apply test file. Co-authored-by: vyctorbrzezowski <51521767+vyctorbrzezowski@users.noreply.github.com> Co-authored-by: Vyctor H. Brzezowski <51521767+vyctorbrzezowski@users.noreply.github.com>
201 lines
7.1 KiB
TypeScript
201 lines
7.1 KiB
TypeScript
// Document Extract plugin module implements document extractor behavior.
|
|
import type { PdfDocument, PdfEngine, RenderOptions } from "clawpdf";
|
|
import type {
|
|
DocumentExtractedImage,
|
|
DocumentExtractionRequest,
|
|
DocumentExtractionResult,
|
|
} from "openclaw/plugin-sdk/document-extractor";
|
|
import { truncateUtf16Safe } from "openclaw/plugin-sdk/string-coerce-runtime";
|
|
import type { WorkerTaskControl } from "openclaw/plugin-sdk/worker-task-server";
|
|
|
|
const MAX_EXTRACTED_TEXT_CHARS = 200_000;
|
|
const MAX_RENDER_DIMENSION = 10_000;
|
|
|
|
let pdfEnginePromise: Promise<PdfEngine> | null = null;
|
|
|
|
async function loadPdfEngine(): Promise<PdfEngine> {
|
|
if (!pdfEnginePromise) {
|
|
pdfEnginePromise = import("clawpdf")
|
|
.then(({ createEngine }) => createEngine())
|
|
.catch((err: unknown) => {
|
|
pdfEnginePromise = null;
|
|
throw new Error("Dependency clawpdf is required for PDF extraction", {
|
|
cause: err,
|
|
});
|
|
});
|
|
}
|
|
return pdfEnginePromise;
|
|
}
|
|
|
|
function toDocumentImage(bytes: Uint8Array): DocumentExtractedImage {
|
|
const data = Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength).toString("base64");
|
|
return { type: "image", data, mimeType: "image/png" };
|
|
}
|
|
|
|
function pageRenderOptions(
|
|
width: number,
|
|
height: number,
|
|
maxPixels: number,
|
|
): { options: RenderOptions; reduced: boolean } | null {
|
|
if (!Number.isFinite(width) || !Number.isFinite(height) || width <= 0 || height <= 0) {
|
|
return null;
|
|
}
|
|
const defaultWidth = Math.ceil(width * (96 / 72));
|
|
const defaultHeight = Math.ceil(height * (96 / 72));
|
|
if (
|
|
defaultWidth <= MAX_RENDER_DIMENSION &&
|
|
defaultHeight <= MAX_RENDER_DIMENSION &&
|
|
defaultWidth * defaultHeight <= maxPixels
|
|
) {
|
|
return { options: { dpi: 96, forms: true }, reduced: false };
|
|
}
|
|
|
|
const landscape = width >= height;
|
|
const longer = landscape ? width : height;
|
|
const shorter = landscape ? height : width;
|
|
let low = 1;
|
|
let high = Math.min(MAX_RENDER_DIMENSION, Math.ceil(longer * (96 / 72)));
|
|
let size = 1;
|
|
// Search integer output dimensions, including render()'s upward rounding of
|
|
// the other edge. The 10,000-pixel edge cap bounds this to 14 iterations.
|
|
while (low <= high) {
|
|
const candidate = Math.floor((low + high) / 2);
|
|
const other = Math.max(1, Math.ceil(shorter * (candidate / longer)));
|
|
if (candidate * other <= maxPixels) {
|
|
size = candidate;
|
|
low = candidate + 1;
|
|
} else {
|
|
high = candidate - 1;
|
|
}
|
|
}
|
|
return {
|
|
options: landscape ? { width: size, forms: true } : { height: size, forms: true },
|
|
reduced: true,
|
|
};
|
|
}
|
|
|
|
function isPdfPasswordError(err: unknown): boolean {
|
|
return err !== null && typeof err === "object" && "code" in err && err.code === "password";
|
|
}
|
|
|
|
async function openPdfDocument(params: {
|
|
engine: PdfEngine;
|
|
input: Uint8Array;
|
|
password?: string;
|
|
}): Promise<PdfDocument> {
|
|
try {
|
|
return params.password
|
|
? await params.engine.open(params.input, { password: params.password })
|
|
: await params.engine.open(params.input);
|
|
} catch (err) {
|
|
if (isPdfPasswordError(err)) {
|
|
throw new Error("PDF requires a password or password is incorrect.", { cause: err });
|
|
}
|
|
throw err;
|
|
}
|
|
}
|
|
|
|
export async function extractPdfContent(
|
|
request: DocumentExtractionRequest,
|
|
control: WorkerTaskControl,
|
|
): Promise<DocumentExtractionResult> {
|
|
const engine = await loadPdfEngine();
|
|
control.throwIfCancelled();
|
|
const pdf = await openPdfDocument({
|
|
engine,
|
|
input: request.buffer,
|
|
...(request.password ? { password: request.password } : {}),
|
|
});
|
|
try {
|
|
control.throwIfCancelled();
|
|
const pages = request.pageNumbers
|
|
? request.pageNumbers
|
|
.filter((p) => Number.isInteger(p) && p >= 1 && p <= pdf.pageCount)
|
|
.slice(0, request.maxPages)
|
|
: undefined;
|
|
if (request.pageNumbers?.length && pages?.length === 0) {
|
|
throw new Error(`No requested PDF pages exist in this ${pdf.pageCount}-page document.`);
|
|
}
|
|
const selectedPages =
|
|
pages ?? Array.from({ length: Math.min(pdf.pageCount, request.maxPages) }, (_, i) => i + 1);
|
|
const metadata = {
|
|
pages: {
|
|
processed: selectedPages,
|
|
total: pdf.pageCount,
|
|
selection: request.pageNumbers ? ("explicit" as const) : ("automatic" as const),
|
|
truncated: request.pageNumbers
|
|
? request.pageNumbers.length > selectedPages.length
|
|
: pdf.pageCount > request.maxPages,
|
|
},
|
|
textTruncated: false,
|
|
imagesTruncated: false,
|
|
};
|
|
const imagePages: number[] = [];
|
|
let text = "";
|
|
for (const pageNumber of selectedPages) {
|
|
control.throwIfCancelled();
|
|
const pageText = pdf.page(pageNumber).text();
|
|
if (pageText.trim().length < request.minTextChars) {
|
|
imagePages.push(pageNumber);
|
|
}
|
|
const separator = text ? "\n\n" : "";
|
|
const remaining = MAX_EXTRACTED_TEXT_CHARS - text.length - separator.length;
|
|
const prefix = truncateUtf16Safe(pageText, Math.max(0, remaining));
|
|
metadata.textTruncated ||= prefix.length < pageText.length;
|
|
if (prefix) {
|
|
text += separator + prefix;
|
|
}
|
|
}
|
|
control.throwIfCancelled();
|
|
if (imagePages.length === 0) {
|
|
return { text, images: [], metadata };
|
|
}
|
|
|
|
// Share the aggregate pixel budget only across pages needing image fallback.
|
|
try {
|
|
const { encodePng, PdfError } = await import("clawpdf");
|
|
const images: DocumentExtractedImage[] = [];
|
|
let remainingPixels = request.maxPixels;
|
|
for (const [index, pageNumber] of imagePages.entries()) {
|
|
control.throwIfCancelled();
|
|
if (remainingPixels <= 0) {
|
|
metadata.imagesTruncated = true;
|
|
break;
|
|
}
|
|
const pagesRemaining = imagePages.length - index;
|
|
const maxPixelsPerPage = Math.max(1, Math.ceil(remainingPixels / pagesRemaining));
|
|
if (!Number.isFinite(maxPixelsPerPage)) {
|
|
throw new PdfError("budget", "maxPixels must be a finite positive number");
|
|
}
|
|
const page = pdf.page(pageNumber);
|
|
const plan = pageRenderOptions(page.width, page.height, maxPixelsPerPage);
|
|
if (!plan) {
|
|
metadata.imagesTruncated = true;
|
|
continue;
|
|
}
|
|
metadata.imagesTruncated ||= plan.reduced;
|
|
const rendered = page.render(plan.options);
|
|
control.throwIfCancelled();
|
|
// Node cannot safely terminate a worker inside zlib initialization.
|
|
// Fence one PNG encode; PDFium rendering remains immediately cancellable.
|
|
const bytes = await control.runNativeSection(() =>
|
|
encodePng(rendered.rgba, { width: rendered.width, height: rendered.height }),
|
|
);
|
|
control.throwIfCancelled();
|
|
images.push(toDocumentImage(bytes));
|
|
remainingPixels -= rendered.width * rendered.height;
|
|
}
|
|
return { text, images, metadata };
|
|
} catch (err) {
|
|
control.throwIfCancelled();
|
|
metadata.imagesTruncated = true;
|
|
request.onImageExtractionError?.(err);
|
|
if (!text.trim()) {
|
|
throw new Error("PDF image extraction failed with no extractable text.", { cause: err });
|
|
}
|
|
return { text, images: [], metadata };
|
|
}
|
|
} finally {
|
|
pdf.destroy();
|
|
}
|
|
}
|