You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Google Drive API与Python实现按文件名生成嵌套文件夹结构

问题描述

我在Google Drive的某个文件夹里有一批文件,文件名包含目标文件夹/子文件夹结构信息,格式如下:

[folder~subfolder] MyFile.ext

部分文件格式为:

[folder] MyOtherFile.ext

我希望通过Google Drive API将这些文件整理到对应的嵌套文件夹结构中,若文件夹不存在则自动创建。

我编写了代码,但目前生成的文件夹未正确嵌套。例如输入文件:

[0_diss] good_journals.txt
[0_diss~historic_disses] Zhang (2020) - NAV premia.pdf
[0_diss~literature] A Guide to Writing the Literature Review.pdf

脚本生成的是同级文件夹:

0_diss
0_diss~historic_disses
0_diss~literature

而我期望生成0_diss文件夹,historic_disses和literature作为0_diss的子文件夹。

以下是我编写的代码:

import os
from google.oauth2.credentials import Credentials
from googleapiclient.discovery import build
from googleapiclient.errors import HttpError

# Set up the Drive API credentials
creds = Credentials.from_authorized_user_file(r"CREDENTIALS_PATH", ['https://www.googleapis.com/auth/drive'])

# Define the name of the Google Drive folder
folder_name = 'FOLDER_CONTAINING_FILES'

# Create a Drive API client
service = build('drive', 'v3', credentials=creds)

# Define the function to parse the subdirectories and sub-subdirectories
def parse_directory(filename):
    # Split the filename by '] '
    parts = filename.split('] ')
    subdirs = []
    for part in parts:
        # Check if the part is a subdirectory
        if '[' in part:
            subdir = part[1:]
            subdirs.append(subdir)
    return subdirs


# Get the list of files in the Google Drive folder
try:
    query = "mimeType='application/vnd.google-apps.folder' and trashed = false and name='" + folder_name + "'"
    folder = service.files().list(q=query).execute().get('files')[0]
    folder_id = folder.get('id')
    query = "'" + folder_id + "' in parents and trashed = false and mimeType != 'application/vnd.google-apps.folder'"
    files = service.files().list(q=query).execute().get('files')
except HttpError as error:
    print(f'An error occurred: {error}')
    files = []

# Create dictionaries to store the subdirectories and sub-subdirectories
subdirs_dict = {}
subsubdirs_dict = {}

# Loop through the files and create the subdirectories and sub-subdirectories if necessary
for file in files:
    filename = file.get('name')
    subdirs = parse_directory(filename)
    if len(subdirs) == 0:
        continue
    parent_id = folder_id
    for i in range(len(subdirs)):
        subdir = subdirs[i]
        if i == len(subdirs) - 1:
            subsubdir = ''
        else:
            subsubdir = subdirs[i+1]
        if subdir not in subdirs_dict:
            # Create the subdirectory if it doesn't exist
            metadata = {'name': subdir, 'parents': [parent_id], 'mimeType': 'application/vnd.google-apps.folder'}
            subdir_file = service.files().create(body=metadata, fields='id').execute()
            subdirs_dict[subdir] = subdir_file.get('id')
        parent_id = subdirs_dict[subdir]
        if subsubdir != '':
            if subsubdir not in subsubdirs_dict:
                # Create the sub-subdirectory if it doesn't exist
                metadata = {'name': subsubdir, 'parents': [parent_id], 'mimeType': 'application/vnd.google-apps.folder'}
                subsubdir_file = service.files().create(body=metadata, fields='id').execute()
                subsubdirs_dict[subsubdir] = subsubdir_file.get('id')
            parent_id = subsubdirs_dict[subsubdir]

# Loop through the files and move them to the appropriate subdirectory or sub-subdirectory
for file in files:
    filename = file.get('name')
    subdirs = parse_directory(filename)
    if len(subdirs) == 0:
        continue
    parent_id = folder_id
    for i in range(len(subdirs)):
        subdir = subdirs[i]
        if i == len(subdirs) - 1:
            subsubdir = ''
        else:
            subsubdir = subdirs[i+1]
        if subdir in subdirs_dict:
            parent_id = subdirs_dict[subdir]
        if subsubdir != '' and subsubdir in subsubdirs_dict:
            parent_id = subsubdirs_dict[subsubdir]
        metadata = {'name': filename, 'parents': [parent_id]}
        service.files().update(fileId=file.get('id'), body=metadata).execute()
修正后的代码
import os
from google.oauth2.credentials import Credentials
from googleapiclient.discovery import build
from googleapiclient.errors import HttpError

# 设置Drive API凭据
creds = Credentials.from_authorized_user_file(r"CREDENTIALS_PATH", ['https://www.googleapis.com/auth/drive'])

# 目标根文件夹名称
folder_name = 'FOLDER_CONTAINING_FILES'

# 创建Drive API客户端
service = build('drive', 'v3', credentials=creds)

def parse_directory(filename):
    """解析文件名中的文件夹路径,返回层级列表"""
    if '[' not in filename or ']' not in filename:
        return []
    # 提取[]包裹的路径部分
    dir_part = filename.split('[')[1].split(']')[0]
    # 按~分割成层级文件夹
    return dir_part.split('~')

def get_or_create_folder(parent_id, folder_name):
    """检查父文件夹下是否存在指定名称的子文件夹,不存在则创建,返回文件夹ID"""
    # 查询父文件夹下的同名子文件夹
    query = f"'{parent_id}' in parents and mimeType='application/vnd.google-apps.folder' and name='{folder_name}' and trashed=false"
    response = service.files().list(q=query, fields='files(id)').execute()
    folders = response.get('files', [])
    if folders:
        return folders[0]['id']
    # 不存在则创建新文件夹
    metadata = {
        'name': folder_name,
        'parents': [parent_id],
        'mimeType': 'application/vnd.google-apps.folder'
    }
    folder = service.files().create(body=metadata, fields='id').execute()
    return folder['id']

# 获取根文件夹ID
try:
    query = f"mimeType='application/vnd.google-apps.folder' and trashed=false and name='{folder_name}'"
    root_folder = service.files().list(q=query, fields='files(id)').execute().get('files', [])[0]
    root_folder_id = root_folder['id']
    # 获取根文件夹下的所有非文件夹文件
    query = f"'{root_folder_id}' in parents and trashed=false and mimeType!='application/vnd.google-apps.folder'"
    files = service.files().list(q=query, fields='files(id,name)').execute().get('files', [])
except HttpError as error:
    print(f'发生错误: {error}')
    files = []

# 处理每个文件:创建嵌套文件夹并移动文件
for file in files:
    file_id = file['id']
    filename = file['name']
    folder_path = parse_directory(filename)
    if not folder_path:
        continue
    
    # 逐层创建/获取文件夹,构建完整嵌套路径
    current_parent_id = root_folder_id
    for folder_name in folder_path:
        current_parent_id = get_or_create_folder(current_parent_id, folder_name)
    
    # 移动文件到最终目标文件夹
    try:
        service.files().update(
            fileId=file_id,
            body={'parents': [current_parent_id]}
        ).execute()
        print(f'已移动文件: {filename} 到路径: {" -> ".join(folder_path)}')
    except HttpError as error:
        print(f'移动文件失败 {filename}: {error}')
关键修改说明
  1. 解析函数优化:重新编写parse_directory函数,正确提取[]内的路径部分并按~分割成层级文件夹列表,比如[0_diss~historic_disses]会被解析为['0_diss', 'historic_disses']。
  2. 文件夹管理逻辑重构:新增get_or_create_folder函数,负责在指定父文件夹下检查是否存在目标子文件夹,不存在则创建。这样能确保文件夹正确嵌套,且避免不同父文件夹下同名子文件夹的ID冲突。
  3. 简化文件移动流程:合并文件夹创建和文件移动的循环,逐层处理每个文件的文件夹路径,直接获取最终目标文件夹ID后移动文件,避免原代码中两次循环的冗余和逻辑错误。
  4. 查询语句优化:使用f-string格式化查询语句,更清晰且避免拼接错误;同时限制返回字段为id或id,name,减少API返回数据量。

内容的提问来源于stack exchange,提问作者scrappylyf

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.29 06:12:58