如何用Python的PyMuPDF移除PDF中的®符号适配Camelot表格提取
批量移除PDF中的®符号以解决Camelot提取表格报错问题
我有上千份多页PDF,需提取表格生成CSV,但Camelot将®符号视为非法字符,报错Illegal character in Name Object (b'/DocuSign\xae')。试过PyPDF2及多种代码均失败,想通过PyMuPDF移除该符号。
尝试的PyMuPDF代码(存在问题)
import os import fitz def replace_text(content, replacements=dict()): lines = content.splitlines() result = "" in_text = False for line in lines: if line == "BT": in_text = True elif line == "ET": in_text = False elif in_text: cmd = line.strip() if cmd.lower() in ['tj', 'tj\n']: replaced_line = line for k, v in replacements.items(): replaced_line = replaced_line.replace(k, v) result += replaced_line + "\n" else: result += line + "\n" else: # 语法错误:无对应if语句 result += line + "\n" return result def remove_illegal_character(input_file, output_folder): filename_base = os.path.splitext(os.path.basename(input_file))[0] output_file = os.path.join(output_folder, filename_base + ".cleaned.pdf") replacements = {"\xae": ""} doc = fitz.open(input_file) for page_number in range(len(doc)): page = doc[page_number] blocks = page.getTextBlocks() for b in blocks: if "\xae" in b[4]: # 检测文本是否包含非法字符 new_text = b[4].replace("\xae", "") page.updateText(fitz.Point(b[0], b[1]), new_text) # updateText API使用错误 doc.save(output_file) doc.close() def main(): input_folder = r'C:\path' output_folder = r'C:\path' # 遍历输入文件夹所有文件 for filename in os.listdir(input_folder): if filename.endswith('.pdf'): input_file = os.path.join(input_folder, filename) # 移除PDF中的非法字符 remove_illegal_character(input_file, output_folder) if __name__ == "__main__": main()
补充:Camelot报错及原代码
使用Camelot时持续触发以下错误:Illegal character in Name Object (b'/DocuSign\xae'),当时使用的代码如下:
def decode_name_object(name): try: return NameObject(name.decode('utf-8')) except (UnicodeEncodeError, UnicodeDecodeError) as e: # 名称对象应使用#加十六进制数表示特殊字符 if not pdf.strict: warnings.warn("Illegal character in Name Object", utils.PdfReadWarning) return NameObject(name) else: # raise utils.PdfReadError("Illegal character in Name Object") return NameObject(name) tables = camelot.read_pdf(r'file.pdf')
解决方案:修正后的PyMuPDF代码
你的代码存在语法错误和API使用问题,以下是可批量移除PDF中®符号的正确实现:
import os import fitz # PyMuPDF def remove_registered_symbol(input_file, output_file): doc = fitz.open(input_file) # 遍历每一页 for page in doc: # 获取页面所有精准文本片段及坐标 text_instances = page.get_text("words") # 反向遍历避免替换后位置偏移 for inst in reversed(text_instances): x0, y0, x1, y1, text, _, _, _ = inst if "\xae" in text: # 替换®符号 new_text = text.replace("\xae", "") # 用白色矩形覆盖原文本 page.draw_rect(fitz.Rect(x0, y0, x1, y1), color=(1,1,1), fill=(1,1,1)) # 写入替换后的文本,匹配原字体大小 page.insert_text(fitz.Point(x0, y1), new_text, fontsize=page.get_fontsize()) # 保存处理后的PDF doc.save(output_file) doc.close() def batch_process_pdfs(input_folder, output_folder): # 创建输出文件夹(不存在则自动创建) os.makedirs(output_folder, exist_ok=True) # 遍历输入文件夹所有PDF文件 for filename in os.listdir(input_folder): if filename.lower().endswith(".pdf"): input_path = os.path.join(input_folder, filename) output_path = os.path.join(output_folder, f"{os.path.splitext(filename)[0]}_cleaned.pdf") remove_registered_symbol(input_path, output_path) print(f"处理完成:{filename}") if __name__ == "__main__": INPUT_FOLDER = r"C:\your_input_path" OUTPUT_FOLDER = r"C:\your_output_path" batch_process_pdfs(INPUT_FOLDER, OUTPUT_FOLDER)
代码说明
- 语法修复:修正了原代码中的缩进错误、无对应
if的else语句 - 精准文本替换:
- 使用
page.get_text("words")获取每个文本片段的精确坐标 - 反向遍历文本片段,避免替换后位置偏移导致的覆盖错误
- 通过绘制白色矩形覆盖原文本,再写入替换后的内容,保证格式一致性
- 使用
- 批量处理优化:自动创建输出文件夹,遍历所有PDF并批量处理
验证效果
处理后的PDF再用Camelot提取表格时,不会再触发Illegal character in Name Object报错,可正常生成CSV文件。
内容的提问来源于stack exchange,提问作者J D
相关产品推荐
相关产品推荐

