Python:移除字典内列表重复条目,仅保留首次出现项
解决文件整理程序中重复文件显示问题
问题说明
你的文件整理程序会生成一个字典,键是目标文件夹名称,值是待移动的源文件列表。但同一文件会出现在多个列表中,导致显示该文件将被多次移动,实际却只能移动一次。需要实现:移除所有列表中的重复条目,仅保留该条目首次出现的列表,其余列表中删除该条目。
示例:
原始字典:
dict_books = { 'key_biography': ['All quiet on the western front', 'My life'], 'key_fiction': ['All quiet on the western front', 'Lord of the Rings'], 'key_fairytales':['Lord of the Rings'] }
期望输出:
dict_books = { 'key_biography': ['All quiet on the western front', 'My life'], 'key_fiction': ['Lord of the Rings'], 'key_fairytales':[] }
你的当前程序输出中,lord of the rings.txt同时出现在audiobooks和movies的待移动列表里,需要修正这个显示问题。
解决方案
核心思路是维护一个已处理文件集合,遍历字典的每个列表时,只保留未在集合中出现过的文件,同时将这些文件加入集合,确保后续列表遇到重复文件时直接过滤掉。
下面是修改后的完整代码,重点新增了跨列表去重逻辑:
import os import sys import shutil from datetime import date from datetime import datetime #directories with relative paths dir_source_files = r'source files' dir_destination = r'destination' dir_keyword_files = r'keywords' dir_logfiles = r'logfiles' def create_folders_and_move_files(): #now a new directory is created and has the same name as the .txt file scanned_dir_keyword_files = os.scandir(dir_keyword_files) for file in scanned_dir_keyword_files: #creates new dirs named after .txt files in destination folder print((f".txt file found in dir_keyword_files:\n{file} name of new folder: " + os.path.splitext(file.name)[0])) new_dir_name = str(os.path.splitext(file.name)[0]) path_for_new_dir = os.path.join(dir_destination, new_dir_name) try: os.makedirs(path_for_new_dir, exist_ok = False) print(f"directory creation succesful. Created directory: {new_dir_name}\n") except OSError as error: print(f"directory creation failed. '{new_dir_name}' already exists\n") #creating a clean keywordlist, finding keywords in sourcefiles, moving sourcefiles when matched for txt_file_with_keywords in os.scandir(dir_keyword_files): print(f" TASK 1: iterating through parent txt-keywordfile:\n {txt_file_with_keywords} ") keywordlist_a = [] with open(txt_file_with_keywords) as txt_full_with_hashtags: for keyword_with_hashtags in txt_full_with_hashtags.readlines(): keywords_without_hashtags = keyword_with_hashtags.rstrip().split('#') #automatically creates list and removes the hashtag from every keyword keywords_without_hashtags.remove('') #removes empty entries from list print(f" TASK 2: fill child keywordlist_a with keywords from parent:") for keyword_without_hashtag in keywords_without_hashtags: keywordlist_a.append(keyword_without_hashtag) print('keyword added to keywordlist_a: ' + keyword_without_hashtag) print("keywordlist_a ready:") print(keywordlist_a) #### this is where the fun begins: for scanned_dir_destination in os.scandir(dir_destination): print("This is the destination path " + str(os.path.realpath(scanned_dir_destination))) for keyword_a in keywordlist_a: for scanned_scource_file in os.scandir(dir_source_files): print(f" TASK 3: comparing keyword '{keyword_a}' to '{scanned_scource_file}'") if keyword_a.lower() in str(scanned_scource_file).lower(): print(f"Bingo! '{keyword_a}' found in '{scanned_scource_file}'") try: shutil.move(scanned_scource_file, scanned_dir_destination) print(f" TASK 4: moving {scanned_scource_file} to {scanned_dir_destination}") except: pass print("\n") def extract_files_from_subfolder_in_source_files(): for possible_dir in os.scandir(dir_source_files): if possible_dir.is_dir(): subfolder_path_in_source_files = os.listdir(dir_source_files + "/" + possible_dir.name) print(f"Subfolder found in source folder: {possible_dir}.") for subfolder_file in subfolder_path_in_source_files: shutil.move(os.path.join(possible_dir.path, subfolder_file), dir_source_files) print(f"FILE: {subfolder_file} extracted from {possible_dir} and moved to {dir_source_files}.") print("All files scanned for subfolders") def delete_empty_folder_in_source_files(): for a_possibly_empty_folder in os.scandir(dir_source_files): print(a_possibly_empty_folder) if a_possibly_empty_folder.is_dir(): print(f"{a_possibly_empty_folder} is dir") subfolder_path = os.listdir(dir_source_files + "/" + a_possibly_empty_folder.name) if len(subfolder_path) == 0: print(f"EMPTY FOLDER FOUND \n The folder {a_possibly_empty_folder} is empty and will be deleted.\n") shutil.rmtree(a_possibly_empty_folder) #extract_files_from_subfolder_in_source_files() #create_folders_and_move_files() #delete_empty_folder_in_source_files() def create_logfile(): current_date = datetime.now() dt_string = current_date.strftime("%Y%m%d%H%M%S") print(f"creating logfile: {dt_string}.txt") try: logfile = open(f"{dir_logfiles}\logfile {dt_string}.txt", "x+") except OSError as error: print(f"logfile creation failed. logfile already exists\n") pass def give_dict_destinationFolders_and_listOf_srcFiles(): dict_destinationFolder_and_listOf_srcFiles = {} for txt_file_with_keywords in os.scandir(dir_keyword_files): #creates new dirs named after .txt files in destination folder list_value_listOf_srcFiles=[] new_dir_name = str(os.path.splitext(txt_file_with_keywords.name)[0]) keywordlist_a = [] with open(txt_file_with_keywords) as txt_full_with_hashtags: for keyword_with_hashtags in txt_full_with_hashtags.readlines(): keywords_without_hashtags = keyword_with_hashtags.rstrip().split('#') #automatically creates list and removes the hashtag from every keyword keywords_without_hashtags.remove('') #removes empty entries from list for keyword_without_hashtag in keywords_without_hashtags: keywordlist_a.append(keyword_without_hashtag) #### this is where the fun begins: for keyword_a in keywordlist_a: for scanned_scource_file in os.scandir(dir_source_files): if keyword_a.lower() in str(scanned_scource_file).lower(): #appending source file titles to a list. the list will become values in the dict "dict_destinationName_and_listOf_srcFiles" file_name = str(os.path.splitext(scanned_scource_file.name)[0]) + str(os.path.splitext(scanned_scource_file.name)[1]) if file_name not in list_value_listOf_srcFiles: # 先去重当前列表内的重复 list_value_listOf_srcFiles.append(file_name) dict_destinationFolder_and_listOf_srcFiles[new_dir_name]=list_value_listOf_srcFiles # 新增跨列表去重逻辑 seen_files = set() for folder in dict_destinationFolder_and_listOf_srcFiles: filtered_files = [] for file in dict_destinationFolder_and_listOf_srcFiles[folder]: if file not in seen_files: filtered_files.append(file) seen_files.add(file) dict_destinationFolder_and_listOf_srcFiles[folder] = filtered_files return dict_destinationFolder_and_listOf_srcFiles ###stackoverflow: duplicate entries from lists within dictionary except for the first entry need to be removed def giveNames_of_destinationFolders_and_srcFiles(): for destination_folder, list_srcFiles in give_dict_destinationFolders_and_listOf_srcFiles().items(): print(f"Following files will be moved to destination folder '{destination_folder}':") for src_file in list_srcFiles: print(f" - {src_file}") print("\n") giveNames_of_destinationFolders_and_srcFiles()
修改说明
- 当前列表内去重:在向
list_value_listOf_srcFiles添加文件时,先判断文件是否已在当前列表中,避免同一列表内的重复条目。 - 跨列表去重:生成原始字典后,遍历每个文件夹的文件列表,用
seen_files集合记录已经出现过的文件,只保留未出现过的文件并更新集合,确保每个文件只在首次出现的文件夹列表中保留。
修正后输出
运行修改后的代码,lord of the rings.txt只会出现在audiobooks的列表中,movies的列表中不再显示该文件:
Following files will be moved to destination folder 'audiobooks': - How to Read People Like a Book -James W. Williams -Full Audiobook (192kbit_AAC).m4a.txt - How to Talk to Anyone 92 Little Tricks for Big Success in Relationships Audiobook (128kbit_AAC).m4a.txt - lord of the rings.txt Following files will be moved to destination folder 'movies':
内容的提问来源于stack exchange,提问作者fbn001
相关产品推荐
相关产品推荐

