如何使用pdfjsLib检测PDF文件是否包含图片?
问题
用pdfjsLib处理PDF文件,已成功提取文本,但卡了3天没找到检测PDF是否包含图片的方法,现有代码如下(stackItem为PDF文件URL),求解决思路:
const loadingTask = pdfjsLib.getDocument(stackItem); loadingTask.promise .then(function (doc: { numPages: any; getMetadata: () => Promise<any>; getPage: (arg0: any) => Promise<any>; }) { const numPages = doc.numPages; console.log("# Document Loaded"); console.log("Number of Pages: " + numPages); console.log(); let lastPromise; lastPromise = doc.getMetadata().then(function (data: { info: any; metadata: { getAll: () => any; }; }) { console.log("# Metadata Is Loaded"); console.log("## Info"); console.log(JSON.stringify(data.info, null, 2)); if (data.metadata) { console.log("## Metadata"); console.log(JSON.stringify(data.metadata.getAll()),); } }); const loadPage = function (pageNum: string) { return doc.getPage(pageNum).then(function (page: { getViewport: (arg0: { scale: number; }) => any; getTextContent: () => Promise<any>; cleanup: () => void; }) { console.log("# Page " + pageNum); const viewport = page.getViewport({ scale: 1.0 }); console.log("Size: " + viewport.width + "x" + viewport.height); console.log(); return page .getTextContent() .then(function (content: { items: any[]; }) { const strings = content.items.map(function (item: { str: any; }) { return item.str; }); console.log("## Text Content"); console.log(strings.join(" ")); page.cleanup(); }) .then(function () { console.log(); }); }); }; // Loading of the first page will wait on metadata and subsequent loadings // will wait on the previous pages. for (let i = 1; i <= numPages; i++) { lastPromise = lastPromise.then(loadPage.bind(null, i)); } return lastPromise; }) .then( function () { console.log("# End of Document"); }, function (err: string) { console.error("Error: " + err); } );
解决思路与代码修改
PDF中的图片通常通过两种形式存在:作为XObject(外部对象)通过Do操作符调用,或是直接内嵌的图片通过BI/EI操作符包裹。我们可以通过解析页面的操作指令来检测图片,修改后的代码如下:
const loadingTask = pdfjsLib.getDocument(stackItem); loadingTask.promise .then(function (doc: { numPages: any; getMetadata: () => Promise<any>; getPage: (arg0: any) => Promise<any>; }) { const numPages = doc.numPages; console.log("# Document Loaded"); console.log("Number of Pages: " + numPages); console.log(); let lastPromise; lastPromise = doc.getMetadata().then(function (data: { info: any; metadata: { getAll: () => any; }; }) { console.log("# Metadata Is Loaded"); console.log("## Info"); console.log(JSON.stringify(data.info, null, 2)); if (data.metadata) { console.log("## Metadata"); console.log(JSON.stringify(data.metadata.getAll()),); } }); const loadPage = function (pageNum: string) { return doc.getPage(pageNum).then(function (page: { getViewport: (arg0: { scale: number; }) => any; getTextContent: () => Promise<any>; getOperatorList: () => Promise<any>; cleanup: () => void; getXObject: (arg0: string) => Promise<any>; }) { console.log("# Page " + pageNum); const viewport = page.getViewport({ scale: 1.0 }); console.log("Size: " + viewport.width + "x" + viewport.height); console.log(); // 检测当前页面是否包含图片 const checkForImages = page.getOperatorList().then(async (opList) => { let hasImage = false; // 遍历所有页面操作指令 for (const op of opList.ops) { const operator = op[0]; // 检测内嵌图片操作符 if (operator === "BI" || operator === "EI") { hasImage = true; break; } // 检测XObject调用并验证是否为图片类型 if (operator === "Do") { const xObjectName = op[1]; const xObject = await page.getXObject(xObjectName); if (xObject.type === "XObjectImage") { hasImage = true; break; } } } console.log(`## Page ${pageNum} has image: ${hasImage}`); return hasImage; }); // 并行执行文本提取与图片检测,提升效率 return Promise.all([ page.getTextContent().then(function (content: { items: any[]; }) { const strings = content.items.map(function (item: { str: any; }) { return item.str; }); console.log("## Text Content"); console.log(strings.join(" ")); }), checkForImages ]).finally(() => { page.cleanup(); console.log(); }); }); }; // 按顺序加载页面 for (let i = 1; i <= numPages; i++) { lastPromise = lastPromise.then(loadPage.bind(null, i)); } return lastPromise; }) .then( function () { console.log("# End of Document"); }, function (err: string) { console.error("Error: " + err); } );
关键说明
page.getOperatorList():获取页面的所有渲染操作指令,这是检测图片的核心入口- 检测
BI/EI:对应直接内嵌在页面中的图片资源 - 检测
Do操作符+验证XObject类型:调用的XObject如果是XObjectImage,即为图片资源 - 用
Promise.all并行处理文本提取和图片检测,避免串行等待浪费时间 - 用
finally确保页面资源被及时清理,防止内存泄漏
内容的提问来源于stack exchange,提问作者tetar
相关产品推荐
相关产品推荐

