You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyDrive2下载共享Google Drive文件遇错求助:实现全量下载

问题描述

我使用PyDrive2批量下载Google Drive共享文件夹中的文件,部分文件可正常下载,但处理不同类型文件时出现报错。需求是优先将文件转为纯文本格式下载,无法转换则按原格式下载,已完成认证流程并处理了部分已知错误,但仍存在两类核心错误:

  1. Use Export with Docs Editors files
  2. The requested conversion is not supported.

代码片段1

# Paginate file lists by specifying number of max results
for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}):
    print('Received %s files from Files.list()' % len(file_list))  # <= 10
    for i, file1 in enumerate(file_list):
        print('\ntitle: %s, id: %s' % (file1['title'], file1['id']))
        print('Downloading {} from GDrive ({}/{})'.format(file1['title'], i, len(file_list)))
        if file1['title'] in ['SOPs']:
            continue
        #file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True)
        try:
            #file1.GetContentFile(file1['title'])
            file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True)
        except errors.HttpError as err:
            print(err)
            continue
        except ApiRequestError as err:
            print(err)
            continue
        except:
            print(f'Some other error')
            continue

代码片段2

from pydrive2.auth import GoogleAuth
from pydrive2.drive import GoogleDrive
from googleapiclient.errors import HttpError

GoogleAuth.DEFAULT_SETTINGS['client_config_file'] = '/Users/kk/acs/client_secrets.json'


def main():
    gauth = GoogleAuth()
    gauth.LocalWebserverAuth()
    drive = GoogleDrive(gauth)

    for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}):
        print('Received %s files from Files.list()' % len(file_list))  # <= 10
        for i, file1 in enumerate(file_list):
            print('\ntitle: %s, id: %s' % (file1['title'], file1['id']))
            print('Downloading {} from GDrive ({}/{})'.format(file1['title'], i, len(file_list)))

            #file1.GetContentFile(file1['title'], mimetype="text/plain", remove_bom=True)
            file1.GetContentFile(file1['title'])



    # errors:
    # pydrive2.files.ApiRequestError: <HttpError 400 when requesting https://www.googleapis.com/drive/v2/files/1k*********6UEZb7/export?mimeType=text%2Fplain&alt=media returned "Export only supports Docs Editors files.". Details: "[{'message': 'Export only supports Docs Editors files.', 'domain': 'global', 'reason': 'badRequest'}]">

    # googleapiclient.errors.HttpError: <HttpError 403 when requesting https://www.googleapis.com/drive/v2/files/1kd*********UEZb7?acknowledgeAbuse=false&alt=media returned "Only files with binary content can be downloaded. Use Export with Docs Editors files.". Details: "[{'message': 'Only files with binary content can be downloaded. Use Export with Docs Editors files.', 'domain': 'global', 'reason': 'fileNotDownloadable', 'location': 'alt', 'locationType': 'parameter'}]">

    # pydrive2.files.ApiRequestError: <HttpError 400 when requesting https://www.googleapis.com/drive/v2/files/1s-*********2mOw/export?mimeType=text%2Fplain&alt=media returned "The requested conversion is not supported.". Details: "[{'message': 'The requested conversion is not supported.', 'domain': 'global', 'reason': 'badRequest', 'location': 'convertTo', 'locationType': 'parameter'}]">


if __name__ == '__main__':
    main()

解决方案

核心逻辑

Google Drive文件分为两类,需分别处理:

  • Docs Editors文件:如Google Docs/Sheets/Slides,无法直接下载原文件,必须通过Export接口转换格式
  • 普通二进制文件:如PDF、图片、压缩包等,可直接调用下载接口获取原文件

通过判断文件的mimeType区分类型,先尝试转纯文本,失败则回退到原格式下载。

优化后代码

from pydrive2.auth import GoogleAuth
from pydrive2.drive import GoogleDrive
from googleapiclient.errors import HttpError
from pydrive2.files import ApiRequestError

GoogleAuth.DEFAULT_SETTINGS['client_config_file'] = '/Users/kk/acs/client_secrets.json'

# 定义Google Docs编辑器类文件的MIME类型
DOCS_EDITORS_MIMETYPES = [
    'application/vnd.google-apps.document',
    'application/vnd.google-apps.spreadsheet',
    'application/vnd.google-apps.presentation',
    'application/vnd.google-apps.form',
    'application/vnd.google-apps.drawing'
]

def download_file(file1):
    print(f'\ntitle: {file1["title"]}, id: {file1["id"]}')
    print(f'Downloading {file1["title"]} from GDrive')
    
    # 跳过指定文件夹
    if file1['title'] == 'SOPs':
        return
    
    file_mime = file1['mimeType']
    success = False

    # 第一步:尝试导出为纯文本
    try:
        if file_mime in DOCS_EDITORS_MIMETYPES:
            # Docs类文件调用Export接口转纯文本
            file1.GetContentFile(f'{file1["title"]}.txt', mimetype='text/plain', remove_bom=True)
        else:
            # 普通文件尝试转纯文本(如PDF等支持转换的格式)
            file1.GetContentFile(f'{file1["title"]}_text.txt', mimetype='text/plain', remove_bom=True)
        print(f'✅ 成功转为纯文本下载:{file1["title"]}')
        success = True
    except (HttpError, ApiRequestError) as err:
        error_msg = str(err)
        # 针对两类核心错误,回退到原格式下载
        if "Export only supports Docs Editors files" in error_msg or "Only files with binary content can be downloaded" in error_msg:
            try:
                file1.GetContentFile(file1['title'])
                print(f'🔄 转为纯文本失败,已下载原格式:{file1["title"]}')
                success = True
            except Exception as e:
                print(f'❌ 原格式下载失败:{file1["title"]} - {str(e)}')
        elif "The requested conversion is not supported" in error_msg:
            try:
                file1.GetContentFile(file1['title'])
                print(f'⚠️ 不支持转为纯文本,已下载原格式:{file1["title"]}')
                success = True
            except Exception as e:
                print(f'❌ 原格式下载失败:{file1["title"]} - {str(e)}')
        else:
            print(f'❌ 未知错误:{file1["title"]} - {str(err)}')
    except Exception as e:
        print(f'❌ 其他错误:{file1["title"]} - {str(e)}')
    
    if not success:
        print(f'❌ 最终下载失败:{file1["title"]}')

def main():
    gauth = GoogleAuth()
    gauth.LocalWebserverAuth()
    drive = GoogleDrive(gauth)

    for file_list in drive.ListFile({'q': 'trashed=false', 'maxResults': 500}):
        print(f'\n📥 收到 {len(file_list)} 个文件')
        for file1 in file_list:
            download_file(file1)

if __name__ == '__main__':
    main()

关键说明

  1. 类型区分:通过mimeType判断是否为Docs Editors文件,避免对普通文件调用Export接口
  2. 错误针对性处理:捕获两类核心错误后直接回退到原格式下载,减少无效重试
  3. 文件名区分:导出纯文本时添加_text.txt后缀,避免与原文件重名(可根据需求调整)
  4. 异常细化:不再用通配符捕获所有异常,便于定位问题

额外建议

  • Google Sheets转纯文本可能丢失格式,可优先尝试导出为CSV(text/csv)再转纯文本
  • 若需遍历子文件夹,可判断mimeType是否为application/vnd.google-apps.folder,然后递归调用下载逻辑
  • 增加日志记录功能,方便后续排查下载失败的文件

内容的提问来源于stack exchange,提问作者user1717931

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.24 17:34:57