如何用JavaScript替换PDF文本?解析专业软件生成PDF的难题
问题:JavaScript替换PDF文本时无法处理Word/LibreWriter生成的PDF
我在项目中需要用JavaScript替换PDF里的文本,目前已经能处理简单结构的PDF,但Word、LibreWriter这类专业软件生成的PDF没法识别和替换文本。
可正常处理的简单PDF片段:
4 0 obj << /Length 132 >> stream BT /F1 12 Tf 50 150 Td (This is a simple PDF) Tj 0 -20 Td /F2 12 Tf (This is bold text) Tj 0 -20 Td /F3 12 Tf (This is italic text) Tj ET endstream endobj
这类PDF的文本能轻松识别,但Word生成的PDF没有可直接读取的真实文本,我想了解这类PDF的解析方法,求相关技术指导。
我尝试用fs流解析替换固定文本可以实现,但处理用户上传或外部输入的文件就失效了,附上我的基础代码:
基础JavaScript代码
const fs = require('fs'); const path = require('path'); var pdfData = null; let parsedText = null; let fontMap = {}; function readPDFFile(filePath) { try { pdfData = fs.readFileSync(filePath, 'utf8'); return pdfData; } catch (err) { console.error(`Could not read the PDF file: ${err.message}`); return null; } } function parsePDF() { if (!pdfData) { console.error('No PDF data available.'); return null; } const lines = pdfData.split('\n'); const text = []; fontMap = {}; let inStream = false; let currentFont = null; lines.forEach(line => { line = line.trim(); if (line === "stream") { inStream = true; } else if (line === "endstream") { inStream = false; } else if (line.startsWith('/BaseFont')) { const match = /\/BaseFont\s\/(\w+)(?:-Bold|-Oblique)?/.exec(line); if (match) { currentFont = match[1]; fontMap[currentFont] = match[0]; } } else if (inStream) { const match = /\((.*)\) Tj/.exec(line); if (match) { text.push({ text: match[1], font: currentFont }); } } }); parsedText = text; return parsedText; } function processParsedText(style) { if (!parsedText) { console.error('No parsed text available.'); return null; } const styleMap = { Italic: '-Oblique', Bold: '-Bold' }; if (!styleMap[style]) { console.error('Invalid style provided.'); return null; } parsedText = parsedText.map(item => { if (item.font) { item.font = `${item.font}${styleMap[style]}`; } return item; }); return parsedText; } function replace(parsedText, newText) { if (!parsedText) { console.error('No parsed text provided.'); return null; } if (!newText || newText.length !== parsedText.length) { console.error('Invalid new text provided.'); return null; } const newContent = parsedText.map((item, index) => ({ text: newText[index], font: item.font })); let updatedPDF = pdfData; parsedText.forEach((item, index) => { const regex = new RegExp(`\\(${item.text}\\) Tj`, 'g'); updatedPDF = updatedPDF.replace(regex, `(${newContent[index].text}) Tj`); }); pdfData = updatedPDF; return newContent; } function saveToFile(filename) { if (!pdfData || !parsedText) { console.error('No data available.'); return; } const combinedContent = `${pdfData}`; fs.writeFileSync(filename, combinedContent, 'utf8'); } function processPDF(inputFilePath, outputFilePath) { readPDFFile(inputFilePath); parsePDF(); saveToFile(outputFilePath); } module.exports = { readPDFFile, parsePDF, processParsedText, replace, saveToFile, processPDF, getPdfData: () => pdfData, getParsedText: () => parsedText };
解决方案解析
为什么Word/LibreWriter生成的PDF无法被你的代码识别?
这类软件生成的PDF通常会做以下处理,导致文本无法直接通过简单正则匹配提取:
- 文本拆分:把单个单词甚至字符拆分成独立的
Tj指令,而非整句存储 - 字体子集化:只嵌入文档用到的字符,字体名称被重命名(比如
ABCDE+Calibri),且字符编码不是标准ASCII/UTF-8 - 压缩流:PDF内容流被Flate压缩,你直接用
utf8读取的是二进制乱码,不是明文的PDF指令
修正方向与实现思路
使用成熟的PDF处理库
不要自己手动解析PDF结构,JavaScript生态里有专门的库处理这类问题:pdf-lib:支持读取、修改PDF文本,处理压缩流和字体子集化,API友好pdfjs-dist:Mozilla的PDF解析引擎,能提取结构化文本,适合先识别再替换
处理压缩内容流
如果你坚持自己实现,需要先检测流是否被压缩(查看/Filter /FlateDecode标记),用pako库解压后再解析指令。适配字体子集化
解析字体的/ToUnicode映射表,把PDF内部的字符编码转换为真实文本,这部分逻辑复杂,建议用现成库。
基于pdf-lib的简单替换示例
const { PDFDocument } = require('pdf-lib'); const fs = require('fs'); async function replacePDFText(inputPath, outputPath, oldText, newText) { const pdfBytes = fs.readFileSync(inputPath); const pdfDoc = await PDFDocument.load(pdfBytes); const pages = pdfDoc.getPages(); for (const page of pages) { const content = await page.getTextContent(); const { items } = content; // 遍历文本项,匹配并替换 for (const item of items) { if (item.str === oldText) { // 用白色矩形遮挡原文本 page.drawRectangle({ x: item.x - 2, y: item.y - item.size, width: item.width + 4, height: item.size + 2, color: 'white', }); // 绘制新文本 page.drawText(newText, { x: item.x, y: item.y, size: item.size, font: pdfDoc.getFont(item.fontName), }); } } } const modifiedPdfBytes = await pdfDoc.save(); fs.writeFileSync(outputPath, modifiedPdfBytes); } // 调用示例 replacePDFText('input.pdf', 'output.pdf', '旧文本', '新文本');
你的现有代码问题
- 直接用
utf8读取PDF:PDF是二进制格式,包含非文本字节,会导致解析错误 - 正则匹配依赖整句
(xxx) Tj:Word生成的PDF文本是拆分的,比如(He) Tj (llo) Tj,无法匹配 - 未处理压缩流:大部分现代PDF的内容流都是压缩的,你读取到的是乱码
内容的提问来源于stack exchange,提问作者Yakup Cemil KAYABAŞ
相关产品推荐
相关产品推荐

