如何使用Node.js从PDF中提取并清理数字签名信息?
提取并清理PDF数字签名信息的Node.js方案改进
我正尝试使用Node.js提取PDF文件中的所有数字签名信息,目前已获取到包含相关信息的数据,但内容杂乱。以下是我当前的代码:
const forge = require("node-forge"); const fs = require("fs"); const { Buffer } = require("buffer"); function GetSignaturesByteRange(FileBuffer) { const ByteRangeList = []; let ByteRangeStart = 0; while ((ByteRangeStart = FileBuffer.indexOf("/ByteRange [", ByteRangeStart)) !== -1) { const ByteRangeEnd = FileBuffer.indexOf("]", ByteRangeStart); const ByteRange = FileBuffer.slice(ByteRangeStart, ByteRangeEnd + 1).toString(); const ByteRangeString = /(\d+) +(\d+) +(\d+) +(\d+)/.exec(ByteRange); const ByteRangeArray = ByteRangeString.slice(1, 5).map(Number); // Convert to integers // Add the ByteRange to the list ByteRangeList.push(ByteRangeArray); // Move past the current ByteRange ByteRangeStart = ByteRangeEnd + 1; } return ByteRangeList; } function GetSignatureData(FileBuffer, ByteRange) { // Extract the specified range from the buffer let ByteRangeBuffer = FileBuffer.slice(ByteRange[1] + 1, ByteRange[2] - 1) // Remove the zeroes from the end of the buffer let EndIndex = ByteRangeBuffer.length; while (EndIndex > 0 && ByteRangeBuffer[EndIndex - 1] === 0x30) { EndIndex--; } ByteRangeBuffer = ByteRangeBuffer.slice(0, EndIndex); return ByteRangeBuffer.toString("binary"); } const InputFile = fs.readFileSync("./signed.pdf"); const AllByteRanges = GetSignaturesByteRange(InputFile); let SignatureBuffer = GetSignatureData(InputFile, AllByteRanges[0]); console.log(SignatureBuffer); fs.writeFileSync("Sigature.bin", Buffer.from(SignatureBuffer, "hex"));生成的Signature.bin文件包含我签名时的姓名、城市等信息,但被无效数据包裹。请问如何正确提取并清理PDF中的数字签名信息?希望得到代码改进建议。
现有代码的核心问题
- 签名范围提取错误:PDF的
ByteRange格式为[offset1, length1, offset2, length2],签名数据实际存储在offset2开始的length2长度区间,原代码错误截取了ByteRange[1]+1到ByteRange[2]-1的范围。 - 错误过滤合法字节:原代码删除末尾的0x30(ASCII字符'0'),但0x30是ASN.1编码中序列的标准标签,属于签名的合法组成部分,随意删除会破坏签名结构。
- 编码转换混乱:将二进制Buffer转成
binary字符串再转hex写入文件,会导致编码失真,破坏原始签名数据。
改进后的代码
const forge = require("node-forge"); const fs = require("fs"); const { Buffer } = require("buffer"); function GetSignaturesByteRange(FileBuffer) { const ByteRangeList = []; let ByteRangeStart = 0; const byteRangePrefix = Buffer.from("/ByteRange ["); while ((ByteRangeStart = FileBuffer.indexOf(byteRangePrefix, ByteRangeStart)) !== -1) { const ByteRangeEnd = FileBuffer.indexOf("]", ByteRangeStart); if (ByteRangeEnd === -1) break; const ByteRangeStr = FileBuffer.slice(ByteRangeStart, ByteRangeEnd + 1).toString(); const matches = /(\d+) +(\d+) +(\d+) +(\d+)/.exec(ByteRangeStr); if (!matches) { ByteRangeStart = ByteRangeEnd + 1; continue; } const ByteRangeArray = matches.slice(1, 5).map(Number); ByteRangeList.push(ByteRangeArray); ByteRangeStart = ByteRangeEnd + 1; } return ByteRangeList; } function ExtractAndParseSignature(FileBuffer, byteRange) { // ByteRange结构: [offset1, length1, offset2, length2] const [, , sigOffset, sigLength] = byteRange; // 提取标准PKCS#7签名Blob const sigBuffer = FileBuffer.slice(sigOffset, sigOffset + sigLength); // 清理PDF签名末尾的0x00填充字节(合法占位填充) let cleanSigBuffer = sigBuffer; while (cleanSigBuffer.length > 0 && cleanSigBuffer[cleanSigBuffer.length - 1] === 0x00) { cleanSigBuffer = cleanSigBuffer.slice(0, -1); } try { // 解析PKCS#7签名结构 const p7 = forge.pkcs7.messageFromAsn1(forge.asn1.fromDer(cleanSigBuffer.toString("binary"))); // 获取签名者信息 const signerInfo = p7.signers[0]; if (!signerInfo) { throw new Error("未找到签名者信息"); } // 从证书中提取结构化的主体信息 const cert = signerInfo.certificate; const subject = cert.subject.attributes; const signatureInfo = { 签名者姓名: subject.find(attr => attr.name === "commonName").value, 组织: subject.find(attr => attr.name === "organizationName")?.value || "未填写", 城市: subject.find(attr => attr.name === "localityName")?.value || "未填写", 省份/州: subject.find(attr => attr.name === "stateOrProvinceName")?.value || "未填写", 国家: subject.find(attr => attr.name === "countryName")?.value || "未填写", 签名时间: signerInfo.signedAttributes.find(attr => attr.type === forge.pki.oids.signingTime)?.value || "未知" }; return { 原始签名数据: cleanSigBuffer, 解析后的签名信息: signatureInfo }; } catch (err) { console.error("解析签名失败:", err); return null; } } // 主执行流程 const InputFile = fs.readFileSync("./signed.pdf"); const AllByteRanges = GetSignaturesByteRange(InputFile); if (AllByteRanges.length === 0) { console.log("PDF中未找到数字签名"); process.exit(1); } // 处理第一个签名(可循环遍历AllByteRanges处理所有签名) const signatureResult = ExtractAndParseSignature(InputFile, AllByteRanges[0]); if (signatureResult) { console.log("解析后的签名信息:"); console.log(signatureResult.解析后的签名信息); // 保存清理后的原始签名数据 fs.writeFileSync("CleanSignature.bin", signatureResult.原始签名数据); console.log("清理后的签名数据已保存到CleanSignature.bin"); }
关键改进说明
- 修正签名范围提取:严格遵循PDF规范,从
ByteRange的第三个值(offset2)开始提取对应长度的签名数据,确保拿到完整的PKCS#7签名Blob。 - 正确清理填充:仅删除末尾的0x00填充字节(PDF签名允许用0填充占位),保留签名的合法ASN.1编码结构。
- 结构化解析签名:利用
node-forge解析PKCS#7签名的ASN.1结构,直接从证书中提取签名者姓名、城市等结构化信息,无需手动处理杂乱的原始数据。 - 避免编码失真:直接操作二进制Buffer,减少不必要的字符串编码转换,确保签名数据的完整性。
内容的提问来源于stack exchange,提问作者Nairel Prandini
相关产品推荐
相关产品推荐

