You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将Google Drive文件列表的Google脚本转换为Python脚本?

Google Drive文件信息批量导出(Python替代GAS实现)

问题背景

我有一段Google Apps Script,可递归列出Google Drive指定文件夹内所有文件的名称、完整存储路径及文件URL,并将信息写入Google表格。尝试用PyDrive转成Python实现时,只能获取文件标题和ID,无法正确获取URL,也没能实现递归遍历子文件夹的功能,求技术帮助。

原Google Apps Script代码

// TODO: Set folder ID
var folderId = '*your folder id is here*';
var array=[];
// Main function 2: List all files & folders, & write into the current sheet.
function listAll(){
  getFolderTree(folderId, true);   
};

// Get Folder Tree
function getFolderTree(folderId, listAll) {
  try {
    // Get folder by id
    var parentFolder = DriveApp.getFolderById(folderId);
    //go to 1st sheet
    var spreadsheet = SpreadsheetApp.getActiveSpreadsheet();
    SpreadsheetApp.setActiveSheet(spreadsheet.getSheets()[0]);
    // Initialise the sheet
    var file, data, sheet = SpreadsheetApp.getActiveSheet();
    sheet.clear();
    // Get files and folders
    getChildFolders(parentFolder.getName(), parentFolder, data, sheet, listAll);
  } 
  catch (e) {
    Logger.log(e.toString());
  }
};

// Get the list of files and folders and their metadata in recursive mode
function getChildFolders(parentName, parent, data, sheet, listAll) {
  var childFolders = parent.getFolders();
  // List folders inside the folder
    while (childFolders.hasNext()) {
    var childFolder = childFolders.next();
    
    // List files inside the folder
    var files = childFolder.getFiles();
    while (listAll & files.hasNext()) {
      var childFile = files.next();
      //Logger.log("File Name: " + childFile.getName());
      data = [ 
        parentName + "/" + childFolder.getName() + "/" + childFile.getName(),
        childFile.getName(),
        childFile.getUrl(),
      ];
      // Write
      array.push(data);
    }
    // Recursive call of the subfolder
    //Logger.log(array);
    getChildFolders(parentName + "/" + childFolder.getName(), childFolder, data, sheet, listAll);  
        SpreadsheetApp.getActiveSheet().getRange(1, 1, array.length, 3).setValues(array);
  };
 
};

尝试的Python代码(存在问题)

# Import PyDrive and associated libraries.
# This only needs to be done once per notebook.
from pydrive.auth import GoogleAuth
from pydrive.drive import GoogleDrive
from google.colab import auth
from oauth2client.client import GoogleCredentials

# Authenticate and create the PyDrive client.
# This only needs to be done once per notebook.
auth.authenticate_user()
gauth = GoogleAuth()
gauth.credentials = GoogleCredentials.get_application_default()
drive = GoogleDrive(gauth)
import pandas as pd
df=pd.DataFrame(columns=('title', 'id','createdDate','modifiedDate','downloadUrl'))
# List .txt files in the root.
#
# Search query reference:
# https://developers.google.com/drive/v2/web/search-parameters
listed = drive.ListFile({'q': "'19r2AtADzB_5DN0_DFIzhO27LrbYf2WAk' in parents and trashed=false"}).GetList()
file.FetchMetadata()
for file in listed:
 
  listoffile=pd.DataFrame([[file['title'],file['id'],file['createdDate'],file['modifiedDate'],'https://docs.google.com/uc?export=download&id='+file['id']]],columns=('title', 'id','createdDate','modifiedDate','downloadUrl'))
  df=df.append(listoffile)
  

解决方案

以下是修正后的Python代码,解决了URL获取和递归遍历的问题:

完整Python实现代码

from pydrive.auth import GoogleAuth
from pydrive.drive import GoogleDrive
from google.colab import auth
from oauth2client.client import GoogleCredentials
import pandas as pd

# 认证初始化
auth.authenticate_user()
gauth = GoogleAuth()
gauth.credentials = GoogleCredentials.get_application_default()
drive = GoogleDrive(gauth)

# 存储文件信息的列表
file_data = []

def traverse_folder(folder, current_path):
    """递归遍历文件夹,收集文件信息"""
    # 获取当前文件夹下的所有文件
    file_list = drive.ListFile({'q': f"'{folder['id']}' in parents and trashed=false and mimeType != 'application/vnd.google-apps.folder'"}).GetList()
    for file in file_list:
        # 获取文件的原生URL和下载URL
        file_url = file.get('alternateLink', '')  # 与GAS的getUrl()返回结果一致
        download_url = f"https://drive.google.com/uc?export=download&id={file['id']}"
        # 整理数据
        file_data.append({
            '完整路径': f"{current_path}/{file['title']}",
            '文件名': file['title'],
            '文件URL': file_url,
            '下载URL': download_url,
            '创建时间': file['createdDate'],
            '修改时间': file['modifiedDate'],
            '文件ID': file['id']
        })
    
    # 递归遍历子文件夹
    subfolder_list = drive.ListFile({'q': f"'{folder['id']}' in parents and trashed=false and mimeType = 'application/vnd.google-apps.folder'"}).GetList()
    for subfolder in subfolder_list:
        new_path = f"{current_path}/{subfolder['title']}"
        traverse_folder(subfolder, new_path)

# 替换为你的目标文件夹ID
target_folder_id = "19r2AtADzB_5DN0_DFIzhO27LrbYf2WAk"
# 获取根文件夹对象
root_folder = drive.CreateFile({'id': target_folder_id})
root_folder.FetchMetadata()

# 开始遍历
traverse_folder(root_folder, root_folder['title'])

# 转换为DataFrame并保存为Excel
df = pd.DataFrame(file_data)
df.to_excel('drive_files_info.xlsx', index=False)

print("文件信息已导出完成!")

关键说明

  1. URL获取:

    • 原生文件网页URL通过file['alternateLink']获取,和GAS里的getUrl()返回结果一致
    • 下载URL使用通用格式生成,适配大部分文件类型
  2. 递归遍历:

    • 定义traverse_folder函数,递归遍历每个子文件夹,记录完整路径
    • 通过mimeType区分文件和文件夹,避免重复遍历
  3. 数据处理:

    • 用列表收集所有文件字典数据,最后转成pandas DataFrame,方便导出为Excel或进一步处理

内容的提问来源于stack exchange,提问作者bedy kharisma

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.09 20:15:33