You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用JavaScript替换PDF文本?解析专业软件生成PDF的难题

问题:JavaScript替换PDF文本时无法处理Word/LibreWriter生成的PDF

我在项目中需要用JavaScript替换PDF里的文本,目前已经能处理简单结构的PDF,但Word、LibreWriter这类专业软件生成的PDF没法识别和替换文本。

可正常处理的简单PDF片段:

4 0 obj
<< /Length 132 >>
stream
BT
/F1 12 Tf
50 150 Td
(This is a simple PDF) Tj
0 -20 Td
/F2 12 Tf
(This is bold text) Tj
0 -20 Td
/F3 12 Tf
(This is italic text) Tj
ET
endstream
endobj

这类PDF的文本能轻松识别,但Word生成的PDF没有可直接读取的真实文本,我想了解这类PDF的解析方法,求相关技术指导。

我尝试用fs流解析替换固定文本可以实现,但处理用户上传或外部输入的文件就失效了,附上我的基础代码:

基础JavaScript代码

const fs = require('fs');
const path = require('path');

var pdfData = null;
let parsedText = null;
let fontMap = {};


function readPDFFile(filePath) {
    try {
        pdfData = fs.readFileSync(filePath, 'utf8');
        return pdfData;
    } catch (err) {
        console.error(`Could not read the PDF file: ${err.message}`);
        return null;
    }
}

function parsePDF() {
    if (!pdfData) {
        console.error('No PDF data available.');
        return null;
    }

    const lines = pdfData.split('\n');
    const text = [];
    fontMap = {};

    let inStream = false;
    let currentFont = null;

    lines.forEach(line => {
        line = line.trim();

        if (line === "stream") {
            inStream = true;
        } else if (line === "endstream") {
            inStream = false;
        } else if (line.startsWith('/BaseFont')) {
            const match = /\/BaseFont\s\/(\w+)(?:-Bold|-Oblique)?/.exec(line);
            if (match) {
                currentFont = match[1];
                fontMap[currentFont] = match[0];
            }
        } else if (inStream) {
            const match = /\((.*)\) Tj/.exec(line);
            if (match) {
                text.push({ text: match[1], font: currentFont });
            }
        }
    });

    parsedText = text;
    return parsedText;
}

function processParsedText(style) {
    if (!parsedText) {
        console.error('No parsed text available.');
        return null;
    }

    const styleMap = {
        Italic: '-Oblique',
        Bold: '-Bold'
    };

    if (!styleMap[style]) {
        console.error('Invalid style provided.');
        return null;
    }

    parsedText = parsedText.map(item => {
        if (item.font) {
            item.font = `${item.font}${styleMap[style]}`;
        }
        return item;
    });

    return parsedText;
}

function replace(parsedText, newText) {
    if (!parsedText) {
        console.error('No parsed text provided.');
        return null;
    }

    if (!newText || newText.length !== parsedText.length) {
        console.error('Invalid new text provided.');
        return null;
    }

    const newContent = parsedText.map((item, index) => ({
        text: newText[index],
        font: item.font
    }));

    let updatedPDF = pdfData;
    parsedText.forEach((item, index) => {
        const regex = new RegExp(`\\(${item.text}\\) Tj`, 'g');
        updatedPDF = updatedPDF.replace(regex, `(${newContent[index].text}) Tj`);
    });

    pdfData = updatedPDF;

    return newContent;
}

function saveToFile(filename) {
    if (!pdfData || !parsedText) {
        console.error('No data available.');
        return;
    }

    const combinedContent = `${pdfData}`;
    fs.writeFileSync(filename, combinedContent, 'utf8');
}

function processPDF(inputFilePath, outputFilePath) {
    readPDFFile(inputFilePath);
    parsePDF();
    saveToFile(outputFilePath);
}

module.exports = {
    readPDFFile,
    parsePDF,
    processParsedText,
    replace,
    saveToFile,
    processPDF,
    getPdfData: () => pdfData,
    getParsedText: () => parsedText
};

解决方案解析

为什么Word/LibreWriter生成的PDF无法被你的代码识别?

这类软件生成的PDF通常会做以下处理,导致文本无法直接通过简单正则匹配提取:

  • 文本拆分:把单个单词甚至字符拆分成独立的Tj指令,而非整句存储
  • 字体子集化:只嵌入文档用到的字符,字体名称被重命名(比如ABCDE+Calibri),且字符编码不是标准ASCII/UTF-8
  • 压缩流:PDF内容流被Flate压缩,你直接用utf8读取的是二进制乱码,不是明文的PDF指令

修正方向与实现思路

  1. 使用成熟的PDF处理库
    不要自己手动解析PDF结构,JavaScript生态里有专门的库处理这类问题:

    • pdf-lib:支持读取、修改PDF文本,处理压缩流和字体子集化,API友好
    • pdfjs-dist:Mozilla的PDF解析引擎,能提取结构化文本,适合先识别再替换
  2. 处理压缩内容流
    如果你坚持自己实现,需要先检测流是否被压缩(查看/Filter /FlateDecode标记),用pako库解压后再解析指令。

  3. 适配字体子集化
    解析字体的/ToUnicode映射表,把PDF内部的字符编码转换为真实文本,这部分逻辑复杂,建议用现成库。

基于pdf-lib的简单替换示例

const { PDFDocument } = require('pdf-lib');
const fs = require('fs');

async function replacePDFText(inputPath, outputPath, oldText, newText) {
  const pdfBytes = fs.readFileSync(inputPath);
  const pdfDoc = await PDFDocument.load(pdfBytes);
  const pages = pdfDoc.getPages();

  for (const page of pages) {
    const content = await page.getTextContent();
    const { items } = content;

    // 遍历文本项,匹配并替换
    for (const item of items) {
      if (item.str === oldText) {
        // 用白色矩形遮挡原文本
        page.drawRectangle({
          x: item.x - 2,
          y: item.y - item.size,
          width: item.width + 4,
          height: item.size + 2,
          color: 'white',
        });
        // 绘制新文本
        page.drawText(newText, {
          x: item.x,
          y: item.y,
          size: item.size,
          font: pdfDoc.getFont(item.fontName),
        });
      }
    }
  }

  const modifiedPdfBytes = await pdfDoc.save();
  fs.writeFileSync(outputPath, modifiedPdfBytes);
}

// 调用示例
replacePDFText('input.pdf', 'output.pdf', '旧文本', '新文本');

你的现有代码问题

  • 直接用utf8读取PDF:PDF是二进制格式,包含非文本字节,会导致解析错误
  • 正则匹配依赖整句(xxx) Tj:Word生成的PDF文本是拆分的,比如(He) Tj (llo) Tj,无法匹配
  • 未处理压缩流:大部分现代PDF的内容流都是压缩的,你读取到的是乱码

内容的提问来源于stack exchange,提问作者Yakup Cemil KAYABAŞ

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.19 00:20:11