PyDrive2下载共享Google Drive文件遇错求助:实现全量下载
问题描述
我使用PyDrive2批量下载Google Drive共享文件夹中的文件,部分文件可正常下载,但处理不同类型文件时出现报错。需求是优先将文件转为纯文本格式下载,无法转换则按原格式下载,已完成认证流程并处理了部分已知错误,但仍存在两类核心错误:
Use Export with Docs Editors filesThe requested conversion is not supported.
代码片段1
# Paginate file lists by specifying number of max results for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}): print('Received %s files from Files.list()' % len(file_list)) # <= 10 for i, file1 in enumerate(file_list): print('\ntitle: %s, id: %s' % (file1['title'], file1['id'])) print('Downloading {} from GDrive ({}/{})'.format(file1['title'], i, len(file_list))) if file1['title'] in ['SOPs']: continue #file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True) try: #file1.GetContentFile(file1['title']) file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True) except errors.HttpError as err: print(err) continue except ApiRequestError as err: print(err) continue except: print(f'Some other error') continue
代码片段2
from pydrive2.auth import GoogleAuth from pydrive2.drive import GoogleDrive from googleapiclient.errors import HttpError GoogleAuth.DEFAULT_SETTINGS['client_config_file'] = '/Users/kk/acs/client_secrets.json' def main(): gauth = GoogleAuth() gauth.LocalWebserverAuth() drive = GoogleDrive(gauth) for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}): print('Received %s files from Files.list()' % len(file_list)) # <= 10 for i, file1 in enumerate(file_list): print('\ntitle: %s, id: %s' % (file1['title'], file1['id'])) print('Downloading {} from GDrive ({}/{})'.format(file1['title'], i, len(file_list))) #file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True) file1.GetContentFile(file1['title']) # errors: # pydrive2.files.ApiRequestError: <HttpError 400 when requesting https://www.googleapis.com/drive/v2/files/1k*********6UEZb7/export?mimeType=text%2Fplain&alt=media returned "Export only supports Docs Editors files.". Details: "[{'message': 'Export only supports Docs Editors files.', 'domain': 'global', 'reason': 'badRequest'}]"> # googleapiclient.errors.HttpError: <HttpError 403 when requesting https://www.googleapis.com/drive/v2/files/1kd*********UEZb7?acknowledgeAbuse=false&alt=media returned "Only files with binary content can be downloaded. Use Export with Docs Editors files.". Details: "[{'message': 'Only files with binary content can be downloaded. Use Export with Docs Editors files.', 'domain': 'global', 'reason': 'fileNotDownloadable', 'location': 'alt', 'locationType': 'parameter'}]"> # pydrive2.files.ApiRequestError: <HttpError 400 when requesting https://www.googleapis.com/drive/v2/files/1s-*********2mOw/export?mimeType=text%2Fplain&alt=media returned "The requested conversion is not supported.". Details: "[{'message': 'The requested conversion is not supported.', 'domain': 'global', 'reason': 'badRequest', 'location': 'convertTo', 'locationType': 'parameter'}]"> if __name__ == '__main__': main()
解决方案
核心逻辑
Google Drive文件分为两类,需分别处理:
- Docs Editors文件:如Google Docs/Sheets/Slides,无法直接下载原文件,必须通过
Export接口转换格式 - 普通二进制文件:如PDF、图片、压缩包等,可直接调用下载接口获取原文件
通过判断文件的mimeType区分类型,先尝试转纯文本,失败则回退到原格式下载。
优化后代码
from pydrive2.auth import GoogleAuth from pydrive2.drive import GoogleDrive from googleapiclient.errors import HttpError from pydrive2.files import ApiRequestError GoogleAuth.DEFAULT_SETTINGS['client_config_file'] = '/Users/kk/acs/client_secrets.json' # 定义Google Docs编辑器类文件的MIME类型 DOCS_EDITORS_MIMETYPES = [ 'application/vnd.google-apps.document', 'application/vnd.google-apps.spreadsheet', 'application/vnd.google-apps.presentation', 'application/vnd.google-apps.form', 'application/vnd.google-apps.drawing' ] def download_file(file1): print(f'\ntitle: {file1["title"]}, id: {file1["id"]}') print(f'Downloading {file1["title"]} from GDrive') # 跳过指定文件夹 if file1['title'] == 'SOPs': return file_mime = file1['mimeType'] success = False # 第一步:尝试导出为纯文本 try: if file_mime in DOCS_EDITORS_MIMETYPES: # Docs类文件调用Export接口转纯文本 file1.GetContentFile(f'{file1["title"]}.txt', mimetype='text/plain', remove_bom=True) else: # 普通文件尝试转纯文本(如PDF等支持转换的格式) file1.GetContentFile(f'{file1["title"]}_text.txt', mimetype='text/plain', remove_bom=True) print(f'✅ 成功转为纯文本下载:{file1["title"]}') success = True except (HttpError, ApiRequestError) as err: error_msg = str(err) # 针对两类核心错误,回退到原格式下载 if "Export only supports Docs Editors files" in error_msg or "Only files with binary content can be downloaded" in error_msg: try: file1.GetContentFile(file1['title']) print(f'🔄 转为纯文本失败,已下载原格式:{file1["title"]}') success = True except Exception as e: print(f'❌ 原格式下载失败:{file1["title"]} - {str(e)}') elif "The requested conversion is not supported" in error_msg: try: file1.GetContentFile(file1['title']) print(f'⚠️ 不支持转为纯文本,已下载原格式:{file1["title"]}') success = True except Exception as e: print(f'❌ 原格式下载失败:{file1["title"]} - {str(e)}') else: print(f'❌ 未知错误:{file1["title"]} - {str(err)}') except Exception as e: print(f'❌ 其他错误:{file1["title"]} - {str(e)}') if not success: print(f'❌ 最终下载失败:{file1["title"]}') def main(): gauth = GoogleAuth() gauth.LocalWebserverAuth() drive = GoogleDrive(gauth) for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}): print(f'\n📥 收到 {len(file_list)} 个文件') for file1 in file_list: download_file(file1) if __name__ == '__main__': main()
关键说明
- 类型区分:通过
mimeType判断是否为Docs Editors文件,避免对普通文件调用Export接口 - 错误针对性处理:捕获两类核心错误后直接回退到原格式下载,减少无效重试
- 文件名区分:导出纯文本时添加
_text.txt后缀,避免与原文件重名(可根据需求调整) - 异常细化:不再用通配符捕获所有异常,便于定位问题
额外建议
- Google Sheets转纯文本可能丢失格式,可优先尝试导出为CSV(
text/csv)再转纯文本 - 若需遍历子文件夹,可判断
mimeType是否为application/vnd.google-apps.folder,然后递归调用下载逻辑 - 增加日志记录功能,方便后续排查下载失败的文件
内容的提问来源于stack exchange,提问作者user1717931
相关产品推荐
相关产品推荐

