如何用Microsoft Graph API的/lists端点获取SharePoint指定路径的文件和文件夹
修改Python脚本实现/lists端点获取指定路径的文件和文件夹
核心思路
- 先通过你熟悉的
/drives端点获取指定路径文件夹的唯一ID - 在
/lists请求中添加$filter条件,筛选ParentFolderId等于目标文件夹ID的列表项,精准定位指定路径下的内容 - 利用列表项的
FileSystemObjectType字段区分文件(值为0)和文件夹(值为1)
修改后的完整代码
import asyncio import aiohttp async def get_target_folder_id(session, accessToken, siteId, driveId, target_path): """通过/drives端点获取指定路径文件夹的ID""" url = f"https://graph.microsoft.com/v1.0/sites/{siteId}/drives/{driveId}/root:/{target_path}:" headers = { 'Authorization': f'Bearer {accessToken}', 'Accept': 'application/json' } async with session.get(url, headers=headers) as response: if response.status != 200: print(f"获取目标文件夹ID失败: {response.status}") return None folder_data = await response.json() return folder_data.get('id') async def get_all_items_in_target_path(accessToken, siteId, libraryId, driveId, target_path, batch_size=15): """获取指定路径下的文件和文件夹""" async with aiohttp.ClientSession() as session: # 第一步:获取目标文件夹ID folder_id = await get_target_folder_id(session, accessToken, siteId, driveId, target_path) if not folder_id: print("无法获取目标文件夹ID,终止操作") return # 构造带过滤条件的lists请求URL filter_str = f"ParentFolderId eq '{folder_id}'" url = f"https://graph.microsoft.com/v1.0/sites/{siteId}/lists/{libraryId}/items?top={batch_size}&$filter={filter_str}&$expand=fields" headers = { 'Authorization': f'Bearer {accessToken}', 'Accept': 'application/json', 'Prefer': 'HonorNonIndexedQueriesWarningMayFailRandomly' } while url: async with session.get(url, headers=headers) as response: if response.status != 200: print(f"请求失败: {response.status}") # 处理节流信息 retry_after = response.headers.get('Retry-After') throttle_info = { 'Retry-After': retry_after, 'Throttle Limit Percentage': response.headers.get('x-ms-throttle-limit-percentage'), 'Throttle Scope': response.headers.get('x-ms-throttle-scope'), 'Throttle Reason': response.headers.get('x-ms-throttle-reason') } for k, v in throttle_info.items(): if v: print(f"{k}: {v}") break data = await response.json() items = data.get('value', []) if not items: break # 处理当前批次的项:区分文件/文件夹,过滤图片 processed_items = [] for item in items: fields = item.get('fields', {}) item_type = fields.get('FileSystemObjectType') web_url = item.get('webUrl', '') item_info = { 'id': item['id'], 'name': fields.get('FileLeafRef'), 'webUrl': web_url, 'is_folder': item_type == 1, 'is_file': item_type == 0 } # 保留图片文件(如果需要) if item_type == 0 and web_url.lower().endswith(('.jpg', '.jpeg', '.png', '.gif')): processed_items.append(item_info) # 如果需要保留文件夹,取消下面的注释 # elif item_type == 1: # processed_items.append(item_info) if processed_items: yield processed_items await asyncio.sleep(0.1) url = data.get('@odata.nextLink') # 使用示例(取消注释即可运行) # async def main(): # access_token = "你的access token" # site_id = "你的site ID" # library_id = "你的文档库列表ID" # drive_id = "你的文档库drive ID" # target_path = "General/index/images/animals" # async for batch in get_all_items_in_target_path(access_token, site_id, library_id, drive_id, target_path): # print(batch) # if __name__ == "__main__": # asyncio.run(main())
关键修改说明
- 新增文件夹ID获取逻辑:
get_target_folder_id函数复用你熟悉的/drives端点,快速定位目标文件夹的ID - 精准过滤路径:通过
ParentFolderId eq '{folder_id}'的filter条件,确保只获取指定路径下的直接子项 - 字段复用优化:请求时添加
$expand=fields,一次性获取FileSystemObjectType(区分文件/文件夹)、FileLeafRef(文件名/文件夹名)等字段,无需额外调用API获取详情 - 灵活筛选:保留了原有的图片格式过滤逻辑,同时可通过注释开启文件夹的返回
内容的提问来源于stack exchange,提问作者the programmer
相关产品推荐
相关产品推荐

