无需下载快速遍历S3大型Zip文件条目并计算偏移量
S3大型Zip文件条目遍历与偏移计算方案(C#)及Python代码修复
一、C#高效处理S3上的大型Zip文件
核心思路是利用Zip文件的**中央目录(Central Directory)**结构,仅下载必要的元数据字节(而非整个文件)来遍历条目并获取偏移量,步骤如下:
- 通过S3 API获取文件总大小,避免全量下载
- 读取文件末尾的**中央目录结束记录(EOCD)**及Zip64相关结构(若存在),解析出中央目录的起始位置和大小
- 发起S3 Range请求,仅下载中央目录的字节范围
- 手动解析中央目录的每个条目,提取文件名和本地文件头偏移量(即条目在Zip文件中的起始位置)
C#实现代码
using Amazon.S3; using Amazon.S3.Model; using System.IO; using System.Text; using System.Linq; public async Task<List<ZipEntryInfo>> GetZipEntriesFromS3Async(string bucketName, string key) { var s3Client = new AmazonS3Client(); var metadataResponse = await s3Client.GetObjectMetadataAsync(bucketName, key); long fileSize = metadataResponse.ContentLength; // 读取末尾100字节,覆盖EOCD和Zip64相关结构 var rangeRequest = new GetObjectRequest { BucketName = bucketName, Key = key, ByteRange = new ByteRange(fileSize - 100, fileSize - 1) }; using var response = await s3Client.GetObjectAsync(rangeRequest); using var ms = new MemoryStream(); await response.ResponseStream.CopyToAsync(ms); byte[] eocdBytes = ms.ToArray(); long centralDirStart = 0; long centralDirSize = 0; bool isZip64 = false; // 查找标准EOCD签名:0x06054b50 int eocdOffset = FindSignature(eocdBytes, new byte[] { 0x50, 0x4b, 0x05, 0x06 }); if (eocdOffset == -1) { // 查找Zip64定位器签名:0x07064b50 int zip64LocatorOffset = FindSignature(eocdBytes, new byte[] { 0x50, 0x4b, 0x06, 0x07 }); if (zip64LocatorOffset == -1) throw new InvalidDataException("无效的ZIP文件"); // 解析Zip64定位器,获取Zip64 EOCD的偏移 long zip64EocdOffset = BitConverter.ToInt64(eocdBytes, zip64LocatorOffset + 8); // 读取Zip64 EOCD var zip64EocdRequest = new GetObjectRequest { BucketName = bucketName, Key = key, ByteRange = new ByteRange(zip64EocdOffset, zip64EocdOffset + 55) // Zip64 EOCD最小56字节 }; using var zip64EocdResponse = await s3Client.GetObjectAsync(zip64EocdRequest); using var zip64Ms = new MemoryStream(); await zip64EocdResponse.ResponseStream.CopyToAsync(zip64Ms); byte[] zip64EocdBytes = zip64Ms.ToArray(); // 验证Zip64 EOCD签名 if (!zip64EocdBytes.Take(4).SequenceEqual(new byte[] { 0x50, 0x4b, 0x06, 0x06 })) throw new InvalidDataException("无效的Zip64 EOCD"); centralDirSize = BitConverter.ToInt64(zip64EocdBytes, 48); centralDirStart = BitConverter.ToInt64(zip64EocdBytes, 56); isZip64 = true; } else { // 解析标准EOCD centralDirSize = BitConverter.ToInt32(eocdBytes, eocdOffset + 12); centralDirStart = BitConverter.ToInt32(eocdBytes, eocdOffset + 16); } // 读取中央目录 var centralDirRequest = new GetObjectRequest { BucketName = bucketName, Key = key, ByteRange = new ByteRange(centralDirStart, centralDirStart + centralDirSize - 1) }; using var centralDirResponse = await s3Client.GetObjectAsync(centralDirRequest); using var centralDirStream = centralDirResponse.ResponseStream; var entries = new List<ZipEntryInfo>(); byte[] entryHeader = new byte[46]; // 中央目录条目最小46字节 while (await centralDirStream.ReadAsync(entryHeader, 0, 4) > 0) { // 验证中央目录条目签名:0x02014b50 if (!entryHeader.Take(4).SequenceEqual(new byte[] { 0x50, 0x4b, 0x01, 0x02 })) break; // 读取完整条目头 int fileNameLength = BitConverter.ToInt16(entryHeader, 28); int extraFieldLength = BitConverter.ToInt16(entryHeader, 30); int commentLength = BitConverter.ToInt16(entryHeader, 32); int totalEntrySize = 46 + fileNameLength + extraFieldLength + commentLength; byte[] fullEntry = new byte[totalEntrySize]; Array.Copy(entryHeader, fullEntry, 4); await centralDirStream.ReadAsync(fullEntry, 4, totalEntrySize - 4); // 提取文件名和偏移量 string fileName = Encoding.UTF8.GetString(fullEntry, 46, fileNameLength); long localHeaderOffset = isZip64 ? BitConverter.ToInt64(fullEntry, 42) : BitConverter.ToInt32(fullEntry, 42); entries.Add(new ZipEntryInfo { FileName = fileName, Offset = localHeaderOffset }); } return entries; } // 辅助方法:查找字节数组中的签名 private int FindSignature(byte[] data, byte[] signature) { for (int i = 0; i <= data.Length - signature.Length; i++) { bool match = true; for (int j = 0; j < signature.Length; j++) { if (data[i + j] != signature[j]) { match = false; break; } } if (match) return i; } return -1; } // 存储条目信息的实体类 public class ZipEntryInfo { public string FileName { get; set; } public long Offset { get; set; } }
二、Python代码错误修复
错误原因
你遇到的ValueError: negative seek value -55是因为目标Zip文件是Zip64格式,标准EOCD仅22字节,但Zip64需要额外的EOCD定位器(20字节)和Zip64 EOCD(56字节),固定读取末尾22字节会导致seek位置超出文件起始点。此外,直接将EOCD字节流传给ZipFile也无法解析,因为ZipFile需要完整的中央目录。
修复后的Python代码
import io import struct def get_zip_entries(zip_path): entries = [] with open(zip_path, 'rb') as f: # 获取文件总大小 f.seek(0, io.SEEK_END) file_size = f.tell() # 读取末尾100字节,覆盖EOCD和Zip64相关结构 f.seek(max(0, file_size - 100), io.SEEK_SET) tail_bytes = f.read() eocd_signature = b'PK\x05\x06' eocd_idx = tail_bytes.find(eocd_signature) zip64 = False central_dir_start = 0 central_dir_size = 0 if eocd_idx == -1: # 处理Zip64格式 zip64_locator_signature = b'PK\x06\x07' locator_idx = tail_bytes.find(zip64_locator_signature) if locator_idx == -1: raise ValueError("无效的ZIP文件") # 解析Zip64定位器,获取Zip64 EOCD的偏移 zip64_eocd_offset = struct.unpack('<Q', tail_bytes[locator_idx+8:locator_idx+16])[0] # 读取Zip64 EOCD f.seek(zip64_eocd_offset) zip64_eocd_bytes = f.read(56) if zip64_eocd_bytes[:4] != b'PK\x06\x06': raise ValueError("无效的Zip64 EOCD") central_dir_size = struct.unpack('<Q', zip64_eocd_bytes[48:56])[0] central_dir_start = struct.unpack('<Q', zip64_eocd_bytes[56:64])[0] zip64 = True else: # 处理标准ZIP格式 eocd_data = tail_bytes[eocd_idx:] central_dir_size = struct.unpack('<I', eocd_data[12:16])[0] central_dir_start = struct.unpack('<I', eocd_data[16:20])[0] # 读取中央目录 f.seek(central_dir_start) central_dir_bytes = f.read(central_dir_size) stream = io.BytesIO(central_dir_bytes) # 遍历中央目录条目 while True: sig = stream.read(4) if not sig or sig != b'PK\x01\x02': break header = stream.read(42) if len(header) < 42: break # 解析基础字段 filename_len = struct.unpack('<H', header[24:26])[0] extra_len = struct.unpack('<H', header[26:28])[0] comment_len = struct.unpack('<H', header[28:30])[0] local_header_offset = struct.unpack('<I', header[38:42])[0] # 处理Zip64的偏移量 if zip64: extra_data = stream.read(extra_len) zip64_extra_idx = extra_data.find(b'\x01\x00') if zip64_extra_idx != -1: local_header_offset = struct.unpack('<Q', extra_data[zip64_extra_idx+4:zip64_extra_idx+12])[0] else: stream.seek(extra_len + comment_len, io.SEEK_CUR) # 读取文件名 filename = stream.read(filename_len).decode('utf-8') if not zip64: stream.seek(comment_len, io.SEEK_CUR) entries.append({ 'filename': filename, 'offset': local_header_offset }) return entries # 使用示例 zip_file = "path/to/zip" entries = get_zip_entries(zip_file) for entry in entries: print(f"文件名: {entry['filename']}, 偏移量: {entry['offset']}")
内容的提问来源于stack exchange,提问作者Aviv Cohen
相关产品推荐
相关产品推荐

