Python zipfile.extractAll不区分大小写致同名大小写不同文件覆盖的问题咨询
Python zipfile.extractAll不区分大小写致同名大小写不同文件覆盖的问题咨询
嘿,这个问题我之前也碰到过,来给你捋捋清楚~
问题原因
你遇到的核心问题是操作系统文件系统的大小写不敏感性(比如Windows默认的NTFS、FAT32都是大小写不区分的)。虽然ZIP文件本身会完整保留文件名的大小写信息,但当你用zipfile.extractall()解压时,系统会把T_N_with_client.docx和T_N_with_Client.docx判定为同一个文件名——后解压的文件会直接覆盖先解压的,所以最后你看到的是其中一个文件名,但内容却是后解压的那个文件的内容。
解决方案
要避免这个覆盖问题,不能直接依赖extractall(),得手动遍历ZIP里的每个文件,自定义解压逻辑来处理大小写冲突。我帮你修改了代码里的extract_zip函数,这样就能完整保留所有带大小写差异的文件:
import os import zipfile from pathlib import Path from tkinter import Tk from tkinter.filedialog import askopenfilename def list_zip_contents(zip_file_path): """Lists all files in a zip file without extracting.""" try: with zipfile.ZipFile(zip_file_path, 'r') as zip_ref: contents = zip_ref.namelist() print(f"ZIP file contains {len(contents)} files:") for file in contents: print(f"{file}") return contents except (zipfile.BadZipFile, OSError) as e: print(f"Error reading the zip file: {e}") raise def extract_zip(zip_file_path, extraction_dir): """Extracts a zip file to the specified directory, handling case-sensitive filename conflicts.""" try: with zipfile.ZipFile(zip_file_path, 'r') as zip_ref: for file_info in zip_ref.infolist(): # 构建目标文件路径 target_path = os.path.join(extraction_dir, file_info.filename) target_dir = os.path.dirname(target_path) # 确保目标目录存在 os.makedirs(target_dir, exist_ok=True) # 检查是否存在大小写不同的同名文件 if os.path.exists(target_path): # 获取系统中实际存在的文件名(大小写可能和ZIP内的不同) actual_name = os.path.basename(os.path.realpath(target_path)) if actual_name != file_info.filename: # 生成带序号的新文件名,避免覆盖 base_name, ext = os.path.splitext(file_info.filename) counter = 1 new_file_name = f"{base_name}_{counter}{ext}" new_target_path = os.path.join(target_dir, new_file_name) # 循环直到找到未被占用的文件名 while os.path.exists(new_target_path): counter += 1 new_file_name = f"{base_name}_{counter}{ext}" new_target_path = os.path.join(target_dir, new_file_name) target_path = new_target_path # 读取ZIP内的文件内容并写入目标路径 with open(target_path, 'wb') as f: f.write(zip_ref.read(file_info.filename)) # 保留原文件的修改时间戳 date_time = zipfile.ZipInfo(*file_info.date_time[:6]).date_time os.utime(target_path, (date_time, date_time)) print(f"All files have been extracted to: {extraction_dir}") except (zipfile.BadZipFile, OSError) as e: print(f"Error extracting the zip file: {e}") raise def main(): try: # Open GUI to select the input zip file Tk().withdraw() # Hide the root window zip_file_path = askopenfilename(title="Select the zip file", filetypes=[("Zip files", "*.zip")]) if not zip_file_path: print("No file selected. Exiting.") return if not os.path.isfile(zip_file_path): print("Invalid file path. Please check and try again.") return # List contents of the ZIP file try: zip_contents = list_zip_contents(zip_file_path) except Exception as e: print("Failed to list contents of the zip file. Exiting.") return # Extract to a directory with the same name as the source zip file (without extension) extraction_dir = os.path.splitext(zip_file_path)[0] try: os.makedirs(extraction_dir, exist_ok=True) except OSError as e: print(f"Error creating extraction directory: {e}") return # Extract the ZIP file try: extract_zip(zip_file_path, extraction_dir) except Exception as e: print("Failed to extract zip file. Exiting.") return # Count files in the extracted directory extracted_files = list(Path(extraction_dir).rglob("*")) print(f"Extracted directory contains {len(extracted_files)} files:") for file in extracted_files: print(f" {file}") except OSError as e: print(f"An OS error occurred: {e}") except Exception as e: print(f"An unexpected error occurred: {e}") if __name__ == "__main__": main()
代码说明
- 遍历ZIP文件详情:用
infolist()获取每个文件的完整信息,而不只是文件名,方便后续处理时间戳等属性。 - 检测大小写冲突:如果目标路径已存在,且系统中实际文件名和ZIP内的文件名大小写不一致,就生成带序号的新文件名(比如
T_N_with_client_1.docx),彻底避免覆盖。 - 手动控制文件写入:不用内置的
extract()方法,直接读取ZIP内的文件内容写入目标路径,完全掌控解压流程。 - 保留原文件属性:还原原文件的修改时间戳,保证解压后的文件和ZIP内的文件属性一致。
如果是在Linux或者开启了大小写敏感模式的macOS系统上,直接用extractall()就不会有这个问题,但Windows默认不支持大小写敏感,所以这个自定义逻辑能完美解决你的需求~
备注:内容来源于stack exchange,提问作者W K
相关产品推荐
相关产品推荐

