You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python调用Google Drive API批量上传文件过慢的优化问询

How to Speed Up Google Drive Bulk Uploads (2000+ Files)

Hey there! Great job getting the basic directory sync working with Python—totally get why you're frustrated with the slow single-file-per-request approach when you're looking at 2000+ files. Let's break down the best ways to speed this up significantly.

Google Drive API doesn't support true "one-request-multiple-files" uploads, but we can use two key techniques to cut down upload time drastically: batch requests to reduce network overhead, and concurrent uploads to process multiple files at once. Let's dive into each, plus a bonus trick for lots of tiny files.


1. Use Google Drive API Batch Requests

Batch requests let you pack multiple API calls (like creating files/folders) into a single HTTP request. This cuts down on TCP handshake time and network round-trips, which is a huge win when you have hundreds of small operations.

Here's how to modify your existing Directory class to use batches:

from googleapiclient.http import BatchHttpRequest
import os
from os import listdir
from googleapiclient.http import MediaFileUpload

class Directory():
    def __init__(self, directory_path):
        self.directory_path = directory_path

    def create_google_drive_tree(
        self, google_drive_folder="", google_service=False, parent_dir_id=''
    ):
        files_and_dirs = [fd for fd in listdir(self.directory_path)]
        files_and_dirs = ff.sort_files_and_dirs(self.directory_path, files_and_dirs)

        # Track folder IDs for recursive subdirectory creation
        folder_ids = {}

        # First batch: Create all top-level folders in current directory
        folder_batch = google_service.new_batch_http_request()

        def folder_callback(request_id, response, exception, folder_name=""):
            if not exception:
                folder_ids[folder_name] = response.get('id')
                print(f"Created folder: {folder_name}")
            else:
                print(f"Failed to create folder {folder_name}: {exception}")

        for fd in files_and_dirs:
            abs_path = ff.abs_path_from_local_dir(self.directory_path, fd)
            if ff.check_file_or_dir(abs_path) == "dir":
                file_metadata = {
                    'name': fd,
                    'mimeType': 'application/vnd.google-apps.folder',
                    'parents': [parent_dir_id]
                }
                folder_batch.add(
                    google_service.files().create(body=file_metadata, fields='id'),
                    callback=lambda req_id, res, exc, name=fd: folder_callback(req_id, res, exc, name)
                )

        # Execute folder creation batch
        folder_batch.execute()

        # Second batch: Upload all top-level files
        file_batch = google_service.new_batch_http_request()

        def file_callback(request_id, response, exception, file_name=""):
            if exception:
                print(f"Failed to upload file {file_name}: {exception}")
            else:
                print(f"Uploaded file: {file_name}")

        for fd in files_and_dirs:
            abs_path = ff.abs_path_from_local_dir(self.directory_path, fd)
            if ff.check_file_or_dir(abs_path) == "file":
                file_metadata = {
                    'name': fd,
                    'parents': [parent_dir_id]
                }
                media = MediaFileUpload(abs_path)
                file_batch.add(
                    google_service.files().create(body=file_metadata, media_body=media, fields='id'),
                    callback=lambda req_id, res, exc, name=fd: file_callback(req_id, res, exc, name)
                )
            else:
                # Recursively build subdirectory tree using the new folder ID
                sub_dir = Directory(abs_path)
                sub_dir.create_google_drive_tree(fd, google_service, folder_ids[fd])

        # Execute file upload batch
        file_batch.execute()

Note: Batches have a soft limit of 1000 requests per batch—if you have more than that in a single directory, split them into smaller chunks to avoid errors.


2. Add Concurrent Uploads (Multithreading)

Even with batches, single-threaded uploads can be slow because you're waiting for each batch to finish. Using multithreading lets you upload multiple files/batches at the same time. We'll use Python's concurrent.futures.ThreadPoolExecutor for this.

First, add a helper function for file uploads, then integrate it into your Directory class:

from concurrent.futures import ThreadPoolExecutor

def upload_single_file(file_path, parent_id, google_service):
    try:
        file_name = os.path.basename(file_path)
        file_metadata = {
            'name': file_name,
            'parents': [parent_id]
        }
        media = MediaFileUpload(file_path)
        google_service.files().create(
            body=file_metadata, media_body=media, fields='id'
        ).execute()
        return f"Success: {file_name}"
    except Exception as e:
        return f"Failed: {file_name} | Error: {str(e)}"

# Update the file processing section in your Directory class:
def create_google_drive_tree(
    self, google_drive_folder="", google_service=False, parent_dir_id=''
):
    # ... [keep existing folder creation logic] ...

    # Collect all file paths to upload concurrently
    file_paths = []
    for fd in files_and_dirs:
        abs_path = ff.abs_path_from_local_dir(self.directory_path, fd)
        if ff.check_file_or_dir(abs_path) == "file":
            file_paths.append(abs_path)
        else:
            # Recursive subdirectory handling
            sub_dir = Directory(abs_path)
            sub_dir.create_google_drive_tree(fd, google_service, folder_ids[fd])

    # Use thread pool to upload files
    with ThreadPoolExecutor(max_workers=8) as executor:
        # Start with 5-10 workers to avoid hitting Google's rate limits
        results = executor.map(
            lambda path: upload_single_file(path, parent_dir_id, google_service),
            file_paths
        )
        for result in results:
            print(result)

Critical Rate Limit Note

Google Drive API enforces a limit of ~1000 requests per 100 seconds per user. Don't crank max_workers too high (stick to 5-10 initially). If you get 429 Too Many Requests errors, add exponential backoff retries (you can use the tenacity library or implement a simple retry loop).


3. Bonus: Zip & Upload (For Tiny File Floods)

If you're dealing with thousands of tiny files (<1MB each), zipping the entire directory first, uploading the zip, then asking Google Drive to unzip it can be way faster than individual uploads. Here's how:

import zipfile

def zip_local_directory(source_dir, zip_path):
    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:
        for root, _, files in os.walk(source_dir):
            for file in files:
                file_full_path = os.path.join(root, file)
                # Preserve directory structure in the zip
                zip_relative_path = os.path.relpath(file_full_path, source_dir)
                zipf.write(file_full_path, zip_relative_path)

# 1. Zip your directory
zip_output = "/home/geoff/HOME-SYNC.zip"
zip_local_directory(start_path, zip_output)

# 2. Upload the zip file
zip_metadata = {
    'name': "HOME-SYNC.zip",
    'parents': ['1YOTDKowprC2Paq95X-MIKSUG_vpuViQw']
}
media = MediaFileUpload(zip_output, mimetype='application/zip')
uploaded_zip = google_service.files().create(
    body=zip_metadata, media_body=media, fields='id'
).execute()

# 3. Ask Google Drive to unzip the file
unzip_request = {
    'name': "HOME-SYNC",
    'mimeType': 'application/vnd.google-apps.folder',
    'parents': ['1YOTDKowprC2Paq95X-MIKSUG_vpuViQw']
}
unzip_result = google_service.files().copy(
    fileId=uploaded_zip.get('id'),
    body=unzip_request,
    fields='id'
).execute()

# Optional: Delete the zip file after unzipping
google_service.files().delete(fileId=uploaded_zip.get('id')).execute()

This works best for one-time full uploads. For incremental syncs (only uploading changed files), stick with batches + concurrency.


Final Pro Tips

  • Incremental Sync: Track file modification times and only upload files that have changed since your last sync—this cuts down on unnecessary requests.
  • Resumable Uploads: For files larger than 100MB, use MediaFileUpload(resumable=True) to resume uploads if your connection drops.
  • Error Handling: Add retry logic for failed uploads (especially 429 and 5xx errors) to make your script more robust.

内容的提问来源于stack exchange,提问作者Geoff L

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.28 10:11:56