如何在PDF末尾%<<-- key -->>{}标签中读写数据?附JS/Python代码
PDF末尾自定义标签的读取与写入实现
一、读取标签内的数据
PDF规范允许在文件末尾(%%EOF之前)添加以%开头的注释内容,你遇到的标签格式为%<--Key-->{JSON数据}<--Key-->,可以通过正则表达式匹配提取其中的JSON内容。
Python 实现
import re import json def read_pdf_tag(pdf_path): with open(pdf_path, 'rb') as f: # 读取文件末尾10KB内容,避免加载大文件到内存 f.seek(-1024 * 10, 2) content = f.read().decode('utf-8', errors='ignore') # 匹配标签格式,提取中间的JSON数据 pattern = r'%<--Key-->(.*?)<--Key-->' match = re.search(pattern, content, re.DOTALL) if match: tag_data = match.group(1) try: return json.loads(tag_data) except json.JSONDecodeError: return "标签内数据不是有效JSON格式" else: return "未找到目标标签" # 使用示例 pdf_file = "example.pdf" tag_content = read_pdf_tag(pdf_file) print(tag_content)
JavaScript(Node.js)实现
const fs = require('fs'); function readPdfTag(pdfPath) { const fileSize = fs.statSync(pdfPath).size; const readBuffer = Buffer.alloc(1024 * 10); // 读取末尾10KB内容 const fd = fs.openSync(pdfPath, 'r'); fs.readSync(fd, readBuffer, 0, readBuffer.length, fileSize - readBuffer.length); fs.closeSync(fd); const content = readBuffer.toString('utf8', 'ignore'); const pattern = /%<--Key-->([\s\S]*?)<--Key-->/; const match = content.match(pattern); if (match) { try { return JSON.parse(match[1]); } catch (e) { return "标签内数据不是有效JSON格式"; } } else { return "未找到目标标签"; } } // 使用示例 const pdfFile = "example.pdf"; const tagContent = readPdfTag(pdfFile); console.log(tagContent);
二、向PDF末尾添加自定义标签
添加标签时需注意不能破坏PDF的结构,正确位置是在startxref行之前(或startxref与%%EOF之间)。如果已有相同标签,可选择替换或追加。
Python 实现(添加/替换标签)
import re import json def add_pdf_tag(pdf_path, custom_data): json_data = json.dumps(custom_data) tag = f'%<--Key-->{json_data}<--Key-->\n' with open(pdf_path, 'r+b') as f: # 读取末尾10KB内容定位关键标记 f.seek(-1024 * 10, 2) content = f.read().decode('utf-8', errors='ignore') # 检查是否已有标签,有则替换,无则插入到startxref前 pattern = r'%<--Key-->(.*?)<--Key-->\n?' if re.search(pattern, content): new_content = re.sub(pattern, tag, content) else: startxref_pos = content.find('startxref') if startxref_pos != -1: new_content = content[:startxref_pos] + tag + content[startxref_pos:] else: # 极端情况,直接追加到%%EOF前 eof_pos = content.find('%%EOF') new_content = content[:eof_pos] + tag + content[eof_pos:] # 定位到读取的起始位置,写入新内容 f.seek(-1024 * 10, 2) f.write(new_content.encode('utf-8')) # 使用示例 custom_data = { "H": "自定义哈希值", "N": "Test", "V": "2.1" } add_pdf_tag("example.pdf", custom_data)
JavaScript(Node.js)实现(添加/替换标签)
const fs = require('fs'); function addPdfTag(pdfPath, customData) { const jsonData = JSON.stringify(customData); const tag = `%<--Key-->${jsonData}<--Key-->\n`; const fileSize = fs.statSync(pdfPath).size; const readBuffer = Buffer.alloc(1024 * 10); const fd = fs.openSync(pdfPath, 'r+'); fs.readSync(fd, readBuffer, 0, readBuffer.length, fileSize - readBuffer.length); let content = readBuffer.toString('utf8', 'ignore'); const pattern = /%<--Key-->[\s\S]*?<--Key-->\n?/; if (pattern.test(content)) { content = content.replace(pattern, tag); } else { const startxrefPos = content.indexOf('startxref'); if (startxrefPos !== -1) { content = content.slice(0, startxrefPos) + tag + content.slice(startxrefPos); } else { const eofPos = content.indexOf('%%EOF'); content = content.slice(0, eofPos) + tag + content.slice(eofPos); } } fs.seekSync(fd, fileSize - readBuffer.length, 0); fs.writeSync(fd, content); fs.closeSync(fd); } // 使用示例 const customData = { "H": "自定义哈希值", "N": "Test", "V": "2.1" }; addPdfTag("example.pdf", customData);
内容的提问来源于stack exchange,提问作者Mayukh Pankaj
相关产品推荐
相关产品推荐

