54GB大文件SHA-256哈希匹配搜索优化求助(Windows 10)
问题概述
我有一个54GB的hashes.txt文件,每行格式为<compressed_hash>:<original_string>。其中compressed_hash由完整SHA-256哈希的第6、13、20、27位字符组成(例如字符串alon的完整哈希为5a24f03a01d5b10cab6124f3c0e7086994ac9c869fc8e76e1463458f829fc864,存储为0db3:alon)。
现有search.py脚本逻辑为:用户输入完整SHA-256哈希后,先生成对应的压缩哈希,遍历整个hashes.txt搜索匹配项;若找到多个匹配,则重新计算这些字符串的完整SHA-256哈希,匹配成功则输出对应字符串。但当前脚本搜索耗时约1小时,需优化。
现有脚本
search.py
import hashlib import mmap def compress_hash(hash_value): return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27] def search_compressed_hash(hash_input, compressed_file): compressed_input = compress_hash(hash_input) potential_matches = [] with open(compressed_file, "r+b") as file: # Memory-map the file, size 0 means the whole file mmapped_file = mmap.mmap(file.fileno(), 0) # Read through the memory-mapped file line by line for line in iter(mmapped_file.readline, b""): line = line.decode().strip() parts = line.split(":", 1) # Split only on the first colon if len(parts) == 2: # Ensure there are exactly two parts compressed_hash, string = parts if compressed_hash == compressed_input: potential_matches.append(string) mmapped_file.close() return potential_matches def verify_full_hash(potential_matches, hash_input): for string in potential_matches: if hashlib.sha256(string.encode()).hexdigest() == hash_input: return string return None if __name__ == "__main__": while True: hash_input = input("Enter the hash (or type 'exit' to quit): ") if hash_input.lower() == 'exit': break potential_matches = search_compressed_hash(hash_input, "hashes.txt") found_string = verify_full_hash(potential_matches, hash_input) if found_string: print(f"Corresponding string: {found_string}") else: print("String not found for the given hash.")
hash.py
import hashlib import sys import time # Set the interval for saving progress (in seconds) SAVE_INTERVAL = 60 # Save progress every minute BUFFER_SIZE = 1000000 # Number of hashes to buffer before writing to file def generate_hash(string): return hashlib.sha256(string.encode()).hexdigest() def compress_hash(hash_value): return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27] def write_hashes_to_file(start_length): buffer = [] # Buffer to store generated hashes last_save_time = time.time() # Store the last save time for generated_string in generate_strings_and_hashes(start_length): full_hash = generate_hash(generated_string) compressed_hash = compress_hash(full_hash) buffer.append((compressed_hash, generated_string)) if len(buffer) >= BUFFER_SIZE: save_buffer_to_file(buffer) buffer = [] # Clear the buffer after writing to file # Check if it's time to save progress if time.time() - last_save_time >= SAVE_INTERVAL: print("Saving progress...") save_buffer_to_file(buffer) # Save any remaining hashes in buffer buffer = [] # Clear buffer after saving last_save_time = time.time() # Save any remaining hashes in buffer if buffer: save_buffer_to_file(buffer) def save_buffer_to_file(buffer): with open("hashes.txt", "a") as file_hashes: file_hashes.writelines(f"{compressed_hash}:{generated_string}\n" for compressed_hash, generated_string in buffer) def generate_strings_and_hashes(start_length): for length in range(start_length, sys.maxsize): # Use sys.maxsize to simulate infinity current_string = [' '] * length # Initialize with spaces while True: yield ''.join(current_string) if current_string == ['z'] * length: # Stop when all characters reach 'z' break current_string = increment_string(current_string) def increment_string(string_list): index = len(string_list) - 1 while index >= 0: if string_list[index] == 'z': string_list[index] = ' ' index -= 1 else: string_list[index] = chr(ord(string_list[index]) + 1) break if index < 0: string_list.insert(0, ' ') return string_list def load_progress(): # You may not need this function anymore return 1 # Just return a default value if __name__ == "__main__": write_hashes_to_file(load_progress())
优化方案
1. 预处理文件:按压缩哈希排序+二分查找
当前脚本需遍历整个54GB文件,效率极低。先将文件按compressed_hash排序,之后用二分查找快速定位匹配行,避免全量遍历。
操作步骤:
- Windows命令行排序:使用系统自带
sort命令(需预留至少1.5倍文件大小的临时磁盘空间):sort hashes.txt /O sorted_hashes.txt - 修改search.py实现二分查找:
import hashlib import mmap import bisect import os def compress_hash(hash_value): return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27] def search_compressed_hash(hash_input, compressed_file): compressed_input = compress_hash(hash_input) potential_matches = [] target_prefix = f"{compressed_input}:".encode() with open(compressed_file, "r+b") as file: mmapped_file = mmap.mmap(file.fileno(), 0, access=mmap.ACCESS_READ) # 记录所有行的起始偏移量 line_offsets = [] offset = 0 while True: newline_pos = mmapped_file.find(b'\n', offset) if newline_pos == -1: break line_offsets.append(offset) offset = newline_pos + 1 # 二分查找定位第一个匹配行 def get_line_prefix(offset): return mmapped_file[offset:offset+len(target_prefix)].decode() idx = bisect.bisect_left(line_offsets, 0, key=lambda x: get_line_prefix(x)) # 收集所有匹配行 while idx < len(line_offsets): line_start = line_offsets[idx] line = mmapped_file[line_start:].split(b'\n', 1)[0].decode().strip() if not line.startswith(compressed_input + ":"): break _, string = line.split(":", 1) potential_matches.append(string) idx += 1 mmapped_file.close() return potential_matches # verify_full_hash及主逻辑保持不变
2. 改用SQLite数据库存储
将文本文件导入SQLite,利用数据库索引特性实现毫秒级查询,这是效率最高的优化方式之一。
导入脚本(import_to_sqlite.py):
import sqlite3 def create_db(): conn = sqlite3.connect('hashes.db') c = conn.cursor() # 创建表,支持同一压缩哈希对应多个字符串 c.execute('''CREATE TABLE IF NOT EXISTS hashes (compressed_hash TEXT, original_string TEXT)''') # 为压缩哈希创建索引,加速查询 c.execute('CREATE INDEX IF NOT EXISTS idx_compressed_hash ON hashes(compressed_hash)') conn.commit() return conn def import_file(conn, file_path): c = conn.cursor() batch_size = 100000 batch = [] with open(file_path, 'r', encoding='utf-8') as f: for line in f: line = line.strip() if not line: continue compressed_hash, string = line.split(":", 1) batch.append((compressed_hash, string)) if len(batch) >= batch_size: c.executemany('INSERT INTO hashes VALUES (?, ?)', batch) conn.commit() batch = [] if batch: c.executemany('INSERT INTO hashes VALUES (?, ?)', batch) conn.commit() if __name__ == "__main__": conn = create_db() import_file(conn, 'hashes.txt') conn.close()
修改后的search.py:
import hashlib import sqlite3 def compress_hash(hash_value): return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27] def search_compressed_hash(hash_input): compressed_input = compress_hash(hash_input) conn = sqlite3.connect('hashes.db') c = conn.cursor() c.execute('SELECT original_string FROM hashes WHERE compressed_hash = ?', (compressed_input,)) potential_matches = [row[0] for row in c.fetchall()] conn.close() return potential_matches # verify_full_hash及主逻辑保持不变
3. 多进程并行搜索
利用Windows多核CPU,将文件分割为多个块并行搜索,汇总结果后验证。
修改后的search.py:
import hashlib import mmap import os from multiprocessing import Pool, cpu_count def compress_hash(hash_value): return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27] def search_chunk(args): chunk_start, chunk_end, file_path, compressed_input = args potential_matches = [] target_prefix = f"{compressed_input}:".encode() with open(file_path, "r+b") as f: mmapped_file = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) # 定位到块的起始行开头 if chunk_start > 0: mmapped_file.seek(chunk_start) mmapped_file.readline() chunk_start = mmapped_file.tell() while mmapped_file.tell() < chunk_end: line = mmapped_file.readline() if not line: break line = line.strip() if line.startswith(target_prefix): _, string = line.decode().split(":", 1) potential_matches.append(string) mmapped_file.close() return potential_matches def search_compressed_hash(hash_input, compressed_file): compressed_input = compress_hash(hash_input) file_size = os.path.getsize(compressed_file) num_processes = cpu_count() chunk_size = file_size // num_processes chunks = [] for i in range(num_processes): start = i * chunk_size end = start + chunk_size if i != num_processes-1 else file_size chunks.append((start, end, compressed_file, compressed_input)) with Pool(num_processes) as pool: results = pool.map(search_chunk, chunks) # 合并所有结果 potential_matches = [] for res in results: potential_matches.extend(res) return potential_matches # verify_full_hash及主逻辑保持不变
4. 优化生成文件结构
修改hash.py,生成文件时按compressed_hash分组写入,避免后续预处理:
- 在
write_hashes_to_file中,用字典按压缩哈希分组缓冲区内容,定期将同一分组的记录写入对应文件(如hashes_0db3.txt),搜索时直接打开对应分组文件即可。
内容的提问来源于stack exchange,提问作者Alon Alush
相关产品推荐
相关产品推荐

