You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

MacBook M2运行Camelot提取PDF表格遇Ghostscript依赖错误

MacBook M2上Camelot调用Ghostscript失败的解决办法

问题场景

在MacBook M2设备上,因Tabula无法提取PDF全页文本,转而使用Camelot处理PDF表格,运行时触发Ghostscript相关错误。

错误日志

Traceback (most recent call last):
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/ext/ghostscript/_gsprint.py", line 260, in <module>
    libgs = cdll.LoadLibrary("libgs.so")
            ^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/ctypes/__init__.py", line 454, in LoadLibrary
    return self._dlltype(name)
           ^^^^^^^^^^^^^^^^^^^
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/ctypes/__init__.py", line 376, in __init__
    self._handle = _dlopen(self._name, mode)
                   ^^^^^^^^^^^^^^^^^^^^^^^^^
OSError: dlopen(libgs.so, 0x0006): tried: 'libgs.so' (no such file), '/System/Volumes/Preboot/Cryptexes/OSlibgs.so' (no such file), '/usr/lib/libgs.so' (no such file, not in dyld cache), 'libgs.so' (no such file), '/usr/lib/libgs.so' (no such file, not in dyld cache)

During handling of the above exception, another exception occurred:

Traceback (most recent call last):
  File "/Users/xxx/Desktop/xx/test.py", line 117, in <module>
    extracted_data =  extract_pdf_camelot()
                      ^^^^^^^^^^^^^^^^^^^^^
  File "/Users/xxx/Desktop/xx/test.py", line 106, in extract_pdf_camelot
    tables = camelot.read_pdf(pdf_path)
             ^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/io.py", line 113, in read_pdf
    tables = p.parse(
             ^^^^^^^^
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/handlers.py", line 173, in parse
    t = parser.extract_tables(
        ^^^^^^^^^^^^^^^^^^^^^^
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/parsers/lattice.py", line 402, in extract_tables
    self._generate_image()
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/parsers/lattice.py", line 211, in _generate_image
    from ..ext.ghostscript import Ghostscript
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/ext/ghostscript/__init__.py", line 24, in <module>
    from . import _gsprint as gs
  File "/Library/Frameworks/Python.framework/Versions/3.11/lib/python3.11/site-packages/camelot/ext/ghostscript/_gsprint.py", line 267, in <module>
    raise RuntimeError("Please make sure that Ghostscript is installed")
RuntimeError: Please make sure that Ghostscript is installed

运行代码

import os
import sys
import PyPDF2
from openpyxl import Workbook
import os
import PyPDF2
import tabula
import pandas as pd
from openpyxl import load_workbook
import camelot


pdf_paths =[]
BatchList = []
university = "xx"
college = "xx"
Batch= "2022"
# Program / Degree:row[3]
# Roll No,.:row[1]
# Name:row[2]
# Branch:row[4]
file_add="xxx"
univ_res = "https://convocation.ccc.ac.in/index.php/convocation/Information_page/degree_recipients"

shet = "Batch:" + Batchexcel_path = 'ExcelDatabase/Ix.xlsx'
tracker_path = "ExcelDatabase/tracker.xlsx"



def extract_pdf_data_tabula(pdf_path):# Specify the path to the main folder
    main_folder_path = "./EducationDatabase2/xxC"
    df_list = []
    for root, dirs, files in os.walk(main_folder_path):
     if len(dirs) > 0:
        for no, dir in enumerate(dirs, start =1):
         if no==1:continue
         for filename in os.listdir(main_folder_path+'/'+dir):
            if filename.endswith(".pdf"):
                print(filename)
                pdf_path =  main_folder_path+'/'+dir + '/'+ filename
                batch=""
                batch+=dir[len(dir)-1]
                batch+=dir[len(dir)-2]
                batch+=dir[len(dir)-3]
                batch+=dir[len(dir)-4]
                batch  =  batch[::-1]
                BatchList.append(batch)
                with open(pdf_path, "rb") as pdf_file:
                    pdf_paths.append(pdf_path)
                    readpdf = PyPDF2.PdfReader(pdf_file)
                    totalpages = len(readpdf.pages)
                    df_list.extend( tabula.read_pdf(pdf_path, pages='1-32',stream=True, lattice=True, guess=False, multiple_tables=True))
                    # df_list[len(df_list)-1].dropna(subset=['Discipline'], inplace=True)
                    # print(batch)
                    # print(df_list[len(df_list)-1])
                    # print()
                    return df_list
                    sys.exit(1)


def save_data_to_excel(data, excel_path):
    cnt =2
    # workbook = load_workbook(excel_path)
    shet = "Batch_2022"
    # sheet = workbook[shet]
    for no, df in enumerate(data, start = 1):
        print("Dataframe:", no)
        df2=df.dropna(axis=1)
        # df2=df.dropna(axis=1,thresh=0)
        print(df2.head(1))
        print()
    #     for i in range(len(df)):
    #         row =  df.iloc[i].to_list()
    #         sno =  str(cnt+1)
    #         row =  [university, college, Batch, row[3], row[1], row[2], row[4], file_add,univ_res ]
    #         for i in range(len(row)):
    #           sheet.cell(row= cnt+1, column = i+1).value = row[i]
    #         cnt+=1
    # workbook.save(excel_path)



def extract_pdf_camelot():
    main_folder_path = "./EducationDatabase2/Indian Institute of Technology - Guwahati"
    df_list = []
    for root, dirs, files in os.walk(main_folder_path):
     if len(dirs) > 0:
        for no, dir in enumerate(dirs, start =1):
         if no==1:continue
         for filename in os.listdir(main_folder_path+'/'+dir):
            if filename.endswith(".pdf"):
                print(filename)
                pdf_path =  main_folder_path+'/'+dir + '/'+ filename
                batch=""
                batch+=dir[len(dir)-1]
                batch+=dir[len(dir)-2]
                batch+=dir[len(dir)-3]
                batch+=dir[len(dir)-4]
                batch  =  batch[::-1]
                BatchList.append(batch)
                # extract all the tables in the PDF file
                tables = camelot.read_pdf(pdf_path)
                cntdf =0
                for table in tables:
                   print("Df no:", cntdf+1)
                   df_list.append(table.df)
                   print(table.df.head(2))
                return 



# extracted_data = extract_pdf_data_tabula(pdf_path)
extracted_data =  extract_pdf_camelot()
cnt =2
save_data_to_excel(extracted_data, excel_path)

已尝试无效方案

  • 通过pip3 install ghostscript==0.7安装Python的ghostscript库
  • 从Ghostscript官网下载源码包手动编译安装
  • 查阅GitHub及Stack Overflow相关问题未找到适配M2的有效方案

适配MacBook M2的解决步骤

  1. 清理现有安装

    • 卸载Python的ghostscript库:pip3 uninstall -y ghostscript
    • 删除源码安装的Ghostscript文件(若有),默认安装路径为/usr/local,可直接删除相关文件夹
  2. 用Homebrew安装Ghostscript

    • 确保已安装Homebrew,未安装则先完成Homebrew安装
    • 执行安装命令:brew install ghostscript
  3. 验证安装

    • 终端执行gs --version,能正常输出版本号则安装成功
  4. 创建软链接让Camelot找到库文件
    MacBook M2为ARM架构,Homebrew安装的libgs路径为/opt/homebrew/lib/libgs.dylib,而Camelot默认查找Linux风格的libgs.so,因此创建软链接:

    sudo ln -s /opt/homebrew/lib/libgs.dylib /usr/local/lib/libgs.so
    

    (Intel Mac的libgs路径为/usr/local/lib/libgs.dylib,对应软链接命令为sudo ln -s /usr/local/lib/libgs.dylib /usr/local/lib/libgs.so)

  5. 测试代码
    重新运行Camelot处理代码,错误即可消除,可正常提取PDF表格。

内容的提问来源于stack exchange,提问作者MAYANK GARG

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 22:17:02