如何用Python从Azure多容器下载对应每日特定Blob文件
问题描述
我是Python新手,需要从Azure存储中每日创建的日期命名容器(如01092022)里,下载对应名称的xlsx格式Blob文件(如p.01092022.xlsx)。现有代码仅能处理单个容器,请问如何遍历所有目标容器并下载对应文件?
尝试的代码
# download_blobs.py # Python program to bulk download blob files from azure storage # Uses latest python SDK() for Azure blob storage # Requires python 3.6 or above import os from azure.storage.blob import BlobServiceClient, BlobClient from azure.storage.blob import ContentSettings, ContainerClient # IMPORTANT: Replace connection string with your storage account connection string # Usually starts with DefaultEndpointsProtocol=https;... MY_CONNECTION_STRING = "my_conection_string" # Replace with blob container MY_BLOB_CONTAINER = "^092022" # Replace with the local folder where you want files to be downloaded LOCAL_BLOB_PATH = "a_local_path" # Replace with the blob to download BLOB_NAME = "^xlsx'" class AzureBlobFileDownloader: def __init__(self): print("Intializing AzureBlobFileDownloader") # Initialize the connection to Azure storage account self.blob_service_client = BlobServiceClient.from_connection_string(MY_CONNECTION_STRING) self.my_container = self.blob_service_client.get_container_client(MY_BLOB_CONTAINER) def save_blob(self,file_name,file_content): # Get full path to the file download_file_path = os.path.join(LOCAL_BLOB_PATH, file_name) # for nested blobs, create local path as well! os.makedirs(os.path.dirname(download_file_path), exist_ok=True) with open(download_file_path, "wb") as file: file.write(file_content) def download_all_blobs_in_container(self): my_blobs = self.my_container.list_blobs(BLOB_NAME) for blob in my_blobs: print(blob.name) bytes = self.my_container.get_blob_client(blob).download_blob().readall() self.save_blob(blob.name, bytes) # Initialize class and upload files azure_blob_file_downloader = AzureBlobFileDownloader() azure_blob_file_downloader.download_all_blobs_in_container()
解决方案
要实现遍历所有日期命名容器并下载对应文件,需要调整代码逻辑,完成容器筛选、目标Blob定位、批量下载三个核心步骤,修改后的代码如下:
import os import re from azure.storage.blob import BlobServiceClient # 替换为你的存储账户连接字符串 MY_CONNECTION_STRING = "your_connection_string" # 本地文件下载路径 LOCAL_BLOB_PATH = "your_local_path" # 日期容器的正则匹配规则(DDMMYYYY格式) DATE_CONTAINER_PATTERN = r'^\d{8}$' # Blob文件名固定前缀 BLOB_PREFIX = "p." class AzureBlobFileDownloader: def __init__(self): print("初始化AzureBlobFileDownloader") self.blob_service_client = BlobServiceClient.from_connection_string(MY_CONNECTION_STRING) def save_blob(self, container_name, file_name, file_content): # 按容器名创建本地子目录,区分不同日期的文件 download_file_path = os.path.join(LOCAL_BLOB_PATH, container_name, file_name) os.makedirs(os.path.dirname(download_file_path), exist_ok=True) with open(download_file_path, "wb") as file: file.write(file_content) print(f"已下载: {download_file_path}") def download_target_blobs(self): # 遍历存储账户下所有容器 containers = self.blob_service_client.list_containers() for container in containers: container_name = container.name # 筛选出符合DDMMYYYY格式的日期容器 if re.match(DATE_CONTAINER_PATTERN, container_name): print(f"开始处理容器: {container_name}") container_client = self.blob_service_client.get_container_client(container_name) # 构造目标Blob文件名 target_blob_name = f"{BLOB_PREFIX}{container_name}.xlsx" try: blob_client = container_client.get_blob_client(target_blob_name) # 检查Blob是否存在,存在则下载 if blob_client.exists(): file_content = blob_client.download_blob().readall() self.save_blob(container_name, target_blob_name, file_content) else: print(f"容器{container_name}中未找到目标文件: {target_blob_name}") except Exception as e: print(f"处理容器{container_name}出错: {str(e)}") if __name__ == "__main__": downloader = AzureBlobFileDownloader() downloader.download_target_blobs()
关键修改说明
- 容器筛选:用正则表达式
DATE_CONTAINER_PATTERN精准匹配DDMMYYYY格式的容器名,避免处理无关容器 - 批量遍历:新增遍历所有容器的逻辑,对每个符合条件的容器单独处理
- 文件定位:根据容器名构造对应Blob文件名,确保下载的是与容器日期匹配的xlsx文件
- 本地存储优化:按容器名创建本地子目录,避免不同日期的文件互相覆盖
- 异常处理:添加捕获异常的逻辑,避免单个容器处理失败导致整个程序终止
内容的提问来源于stack exchange,提问作者David Valdera
相关产品推荐
相关产品推荐

