You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何使用Node.js从PDF中提取并清理数字签名信息?

提取并清理PDF数字签名信息的Node.js方案改进

我正尝试使用Node.js提取PDF文件中的所有数字签名信息,目前已获取到包含相关信息的数据,但内容杂乱。以下是我当前的代码:

const forge = require("node-forge");
const fs = require("fs");
const { Buffer } = require("buffer");


function GetSignaturesByteRange(FileBuffer) {

    const ByteRangeList = [];

    let ByteRangeStart = 0;

    while ((ByteRangeStart = FileBuffer.indexOf("/ByteRange [", ByteRangeStart)) !== -1) {

        const ByteRangeEnd = FileBuffer.indexOf("]", ByteRangeStart);
        const ByteRange = FileBuffer.slice(ByteRangeStart, ByteRangeEnd + 1).toString();

        const ByteRangeString = /(\d+) +(\d+) +(\d+) +(\d+)/.exec(ByteRange);
        const ByteRangeArray = ByteRangeString.slice(1, 5).map(Number); // Convert to integers

        // Add the ByteRange to the list
        ByteRangeList.push(ByteRangeArray);

        // Move past the current ByteRange
        ByteRangeStart = ByteRangeEnd + 1;

    }

    return ByteRangeList;
}

function GetSignatureData(FileBuffer, ByteRange) {

    // Extract the specified range from the buffer
    let ByteRangeBuffer = FileBuffer.slice(ByteRange[1] + 1, ByteRange[2] - 1)

    // Remove the zeroes from the end of the buffer
    let EndIndex = ByteRangeBuffer.length;
    while (EndIndex > 0 && ByteRangeBuffer[EndIndex - 1] === 0x30) {
        EndIndex--;
    }

    ByteRangeBuffer = ByteRangeBuffer.slice(0, EndIndex);

    return ByteRangeBuffer.toString("binary");
}

const InputFile = fs.readFileSync("./signed.pdf");
const AllByteRanges = GetSignaturesByteRange(InputFile);

let SignatureBuffer = GetSignatureData(InputFile, AllByteRanges[0]);


console.log(SignatureBuffer);

fs.writeFileSync("Sigature.bin", Buffer.from(SignatureBuffer, "hex"));

生成的Signature.bin文件包含我签名时的姓名、城市等信息,但被无效数据包裹。请问如何正确提取并清理PDF中的数字签名信息?希望得到代码改进建议。

现有代码的核心问题

  1. 签名范围提取错误:PDF的ByteRange格式为[offset1, length1, offset2, length2],签名数据实际存储在offset2开始的length2长度区间,原代码错误截取了ByteRange[1]+1到ByteRange[2]-1的范围。
  2. 错误过滤合法字节:原代码删除末尾的0x30(ASCII字符'0'),但0x30是ASN.1编码中序列的标准标签,属于签名的合法组成部分,随意删除会破坏签名结构。
  3. 编码转换混乱:将二进制Buffer转成binary字符串再转hex写入文件,会导致编码失真,破坏原始签名数据。

改进后的代码

const forge = require("node-forge");
const fs = require("fs");
const { Buffer } = require("buffer");

function GetSignaturesByteRange(FileBuffer) {
    const ByteRangeList = [];
    let ByteRangeStart = 0;
    const byteRangePrefix = Buffer.from("/ByteRange [");
    
    while ((ByteRangeStart = FileBuffer.indexOf(byteRangePrefix, ByteRangeStart)) !== -1) {
        const ByteRangeEnd = FileBuffer.indexOf("]", ByteRangeStart);
        if (ByteRangeEnd === -1) break;
        
        const ByteRangeStr = FileBuffer.slice(ByteRangeStart, ByteRangeEnd + 1).toString();
        const matches = /(\d+) +(\d+) +(\d+) +(\d+)/.exec(ByteRangeStr);
        if (!matches) {
            ByteRangeStart = ByteRangeEnd + 1;
            continue;
        }
        
        const ByteRangeArray = matches.slice(1, 5).map(Number);
        ByteRangeList.push(ByteRangeArray);
        ByteRangeStart = ByteRangeEnd + 1;
    }
    return ByteRangeList;
}

function ExtractAndParseSignature(FileBuffer, byteRange) {
    // ByteRange结构: [offset1, length1, offset2, length2]
    const [, , sigOffset, sigLength] = byteRange;
    // 提取标准PKCS#7签名Blob
    const sigBuffer = FileBuffer.slice(sigOffset, sigOffset + sigLength);
    
    // 清理PDF签名末尾的0x00填充字节(合法占位填充)
    let cleanSigBuffer = sigBuffer;
    while (cleanSigBuffer.length > 0 && cleanSigBuffer[cleanSigBuffer.length - 1] === 0x00) {
        cleanSigBuffer = cleanSigBuffer.slice(0, -1);
    }
    
    try {
        // 解析PKCS#7签名结构
        const p7 = forge.pkcs7.messageFromAsn1(forge.asn1.fromDer(cleanSigBuffer.toString("binary")));
        
        // 获取签名者信息
        const signerInfo = p7.signers[0];
        if (!signerInfo) {
            throw new Error("未找到签名者信息");
        }
        
        // 从证书中提取结构化的主体信息
        const cert = signerInfo.certificate;
        const subject = cert.subject.attributes;
        
        const signatureInfo = {
            签名者姓名: subject.find(attr => attr.name === "commonName").value,
            组织: subject.find(attr => attr.name === "organizationName")?.value || "未填写",
            城市: subject.find(attr => attr.name === "localityName")?.value || "未填写",
            省份/州: subject.find(attr => attr.name === "stateOrProvinceName")?.value || "未填写",
            国家: subject.find(attr => attr.name === "countryName")?.value || "未填写",
            签名时间: signerInfo.signedAttributes.find(attr => attr.type === forge.pki.oids.signingTime)?.value || "未知"
        };
        
        return {
            原始签名数据: cleanSigBuffer,
            解析后的签名信息: signatureInfo
        };
    } catch (err) {
        console.error("解析签名失败:", err);
        return null;
    }
}

// 主执行流程
const InputFile = fs.readFileSync("./signed.pdf");
const AllByteRanges = GetSignaturesByteRange(InputFile);

if (AllByteRanges.length === 0) {
    console.log("PDF中未找到数字签名");
    process.exit(1);
}

// 处理第一个签名(可循环遍历AllByteRanges处理所有签名)
const signatureResult = ExtractAndParseSignature(InputFile, AllByteRanges[0]);

if (signatureResult) {
    console.log("解析后的签名信息:");
    console.log(signatureResult.解析后的签名信息);
    
    // 保存清理后的原始签名数据
    fs.writeFileSync("CleanSignature.bin", signatureResult.原始签名数据);
    console.log("清理后的签名数据已保存到CleanSignature.bin");
}

关键改进说明

  • 修正签名范围提取:严格遵循PDF规范,从ByteRange的第三个值(offset2)开始提取对应长度的签名数据,确保拿到完整的PKCS#7签名Blob。
  • 正确清理填充:仅删除末尾的0x00填充字节(PDF签名允许用0填充占位),保留签名的合法ASN.1编码结构。
  • 结构化解析签名:利用node-forge解析PKCS#7签名的ASN.1结构,直接从证书中提取签名者姓名、城市等结构化信息,无需手动处理杂乱的原始数据。
  • 避免编码失真:直接操作二进制Buffer,减少不必要的字符串编码转换,确保签名数据的完整性。

内容的提问来源于stack exchange,提问作者Nairel Prandini

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 13:34:59