如何用Python通过Google Drive API V3将云端PNG转为带指定OCR语言的Google Doc
解决方案:从Google Drive云端PNG图片提取文本(指定OCR语言)
前置准备
- 启用Google Drive API,并下载服务账号密钥文件(
credentials.json) - 安装依赖包:
pip install google-api-python-client google-auth-httplib2 google-auth-oauthlib
代码实现
以下代码会遍历指定Drive文件夹下的所有PNG图片,将其转换为带指定OCR语言的Google Doc,再导出为TXT文件,最后可选择删除临时生成的Google Doc:
from googleapiclient.discovery import build from googleapiclient.http import MediaIoBaseDownload from google.oauth2.service_account import Credentials import io import os # 配置参数 FOLDER_ID = "你的Drive文件夹ID" # 替换为目标文件夹ID OCR_LANGUAGE = "zh-CN" # 替换为需要的OCR语言代码(如en、ja等) OUTPUT_DIR = "./ocr_output" # 本地输出文本的文件夹 DELETE_TEMP_DOC = True # 是否删除转换后的临时Google Doc # 初始化Drive API客户端 def get_drive_service(): creds = Credentials.from_service_account_file( "credentials.json", scopes=["https://www.googleapis.com/auth/drive"] ) return build("drive", "v3", credentials=creds) # 导出Google Doc为TXT文件 def export_doc_to_txt(service, doc_id, filename): request = service.files().export_media(fileId=doc_id, mimeType="text/plain") fh = io.FileIO(filename, "wb") downloader = MediaIoBaseDownload(fh, request) done = False while done is False: status, done = downloader.next_chunk() fh.close() # 处理单个PNG图片 def process_image(service, file): # 复制图片为Google Doc,启用OCR并指定语言 copy_body = { "name": f"{file['name']}_temp_doc", "mimeType": "application/vnd.google-apps.document" } copied_file = service.files().copy( fileId=file["id"], body=copy_body, ocrLanguage=OCR_LANGUAGE ).execute() # 导出为TXT txt_filename = f"{OUTPUT_DIR}/{file['name'].replace('.png', '.txt')}" export_doc_to_txt(service, copied_file["id"], txt_filename) print(f"已导出文本:{txt_filename}") # 删除临时Doc(可选) if DELETE_TEMP_DOC: service.files().delete(fileId=copied_file["id"]).execute() # 主函数:遍历文件夹下的PNG文件 def main(): os.makedirs(OUTPUT_DIR, exist_ok=True) service = get_drive_service() # 查询文件夹下的所有PNG文件 query = f"'{FOLDER_ID}' in parents and mimeType='image/png'" results = service.files().list(q=query, fields="files(id, name)").execute() files = results.get("files", []) if not files: print("未找到PNG图片") return for file in files: print(f"处理图片:{file['name']}") process_image(service, file) if __name__ == "__main__": main()
关键说明
ocrLanguage参数:需使用ISO 639-1语言代码(如zh-CN代表简体中文,en代表英文),具体支持的语言可参考Google官方OCR支持列表- 文件夹ID获取:打开目标Drive文件夹,URL中
folders/后的字符串即为文件夹ID - 权限注意:确保服务账号拥有目标文件夹的访问权限(可将服务账号邮箱添加为文件夹协作者)
内容的提问来源于stack exchange,提问作者user1940163
相关产品推荐
相关产品推荐

