You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

54GB大文件SHA-256哈希匹配搜索优化求助(Windows 10)

问题概述

我有一个54GB的hashes.txt文件,每行格式为<compressed_hash>:<original_string>。其中compressed_hash由完整SHA-256哈希的第6、13、20、27位字符组成(例如字符串alon的完整哈希为5a24f03a01d5b10cab6124f3c0e7086994ac9c869fc8e76e1463458f829fc864,存储为0db3:alon)。

现有search.py脚本逻辑为:用户输入完整SHA-256哈希后,先生成对应的压缩哈希,遍历整个hashes.txt搜索匹配项;若找到多个匹配,则重新计算这些字符串的完整SHA-256哈希,匹配成功则输出对应字符串。但当前脚本搜索耗时约1小时,需优化。

现有脚本

search.py

import hashlib
import mmap

def compress_hash(hash_value):
    return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27]

def search_compressed_hash(hash_input, compressed_file):
    compressed_input = compress_hash(hash_input)
    potential_matches = []
    
    with open(compressed_file, "r+b") as file:
        # Memory-map the file, size 0 means the whole file
        mmapped_file = mmap.mmap(file.fileno(), 0)
        
        # Read through the memory-mapped file line by line
        for line in iter(mmapped_file.readline, b""):
            line = line.decode().strip()
            parts = line.split(":", 1)  # Split only on the first colon
            if len(parts) == 2:  # Ensure there are exactly two parts
                compressed_hash, string = parts
                if compressed_hash == compressed_input:
                    potential_matches.append(string)
        
        mmapped_file.close()
    
    return potential_matches

def verify_full_hash(potential_matches, hash_input):
    for string in potential_matches:
        if hashlib.sha256(string.encode()).hexdigest() == hash_input:
            return string
    return None

if __name__ == "__main__":
    while True:
        hash_input = input("Enter the hash (or type 'exit' to quit): ")
        if hash_input.lower() == 'exit':
            break
        
        potential_matches = search_compressed_hash(hash_input, "hashes.txt")
        found_string = verify_full_hash(potential_matches, hash_input)
        
        if found_string:
            print(f"Corresponding string: {found_string}")
        else:
            print("String not found for the given hash.")

hash.py

import hashlib
import sys
import time

# Set the interval for saving progress (in seconds)
SAVE_INTERVAL = 60  # Save progress every minute
BUFFER_SIZE = 1000000  # Number of hashes to buffer before writing to file

def generate_hash(string):
    return hashlib.sha256(string.encode()).hexdigest()

def compress_hash(hash_value):
    return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27]

def write_hashes_to_file(start_length):
    buffer = []  # Buffer to store generated hashes
    last_save_time = time.time()  # Store the last save time
    
    for generated_string in generate_strings_and_hashes(start_length):
        full_hash = generate_hash(generated_string)
        compressed_hash = compress_hash(full_hash)
        buffer.append((compressed_hash, generated_string))
        
        if len(buffer) >= BUFFER_SIZE:
            save_buffer_to_file(buffer)
            buffer = []  # Clear the buffer after writing to file
        
        # Check if it's time to save progress
        if time.time() - last_save_time >= SAVE_INTERVAL:
            print("Saving progress...")
            save_buffer_to_file(buffer)  # Save any remaining hashes in buffer
            buffer = []  # Clear buffer after saving
            last_save_time = time.time()
    
    # Save any remaining hashes in buffer
    if buffer:
        save_buffer_to_file(buffer)

def save_buffer_to_file(buffer):
    with open("hashes.txt", "a") as file_hashes:
        file_hashes.writelines(f"{compressed_hash}:{generated_string}\n" for compressed_hash, generated_string in buffer)

def generate_strings_and_hashes(start_length):
    for length in range(start_length, sys.maxsize):  # Use sys.maxsize to simulate infinity
        current_string = [' '] * length  # Initialize with spaces
        while True:
            yield ''.join(current_string)
            if current_string == ['z'] * length:  # Stop when all characters reach 'z'
                break
            current_string = increment_string(current_string)

def increment_string(string_list):
    index = len(string_list) - 1
    
    while index >= 0:
        if string_list[index] == 'z':
            string_list[index] = ' '
            index -= 1
        else:
            string_list[index] = chr(ord(string_list[index]) + 1)
            break
    
    if index < 0:
        string_list.insert(0, ' ')
    
    return string_list

def load_progress():
    # You may not need this function anymore
    return 1  # Just return a default value

if __name__ == "__main__":
    write_hashes_to_file(load_progress())
优化方案

1. 预处理文件:按压缩哈希排序+二分查找

当前脚本需遍历整个54GB文件,效率极低。先将文件按compressed_hash排序,之后用二分查找快速定位匹配行,避免全量遍历。

操作步骤:

  • Windows命令行排序:使用系统自带sort命令(需预留至少1.5倍文件大小的临时磁盘空间):
    sort hashes.txt /O sorted_hashes.txt
    
  • 修改search.py实现二分查找:
    import hashlib
    import mmap
    import bisect
    import os
    
    def compress_hash(hash_value):
        return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27]
    
    def search_compressed_hash(hash_input, compressed_file):
        compressed_input = compress_hash(hash_input)
        potential_matches = []
        target_prefix = f"{compressed_input}:".encode()
    
        with open(compressed_file, "r+b") as file:
            mmapped_file = mmap.mmap(file.fileno(), 0, access=mmap.ACCESS_READ)
            # 记录所有行的起始偏移量
            line_offsets = []
            offset = 0
            while True:
                newline_pos = mmapped_file.find(b'\n', offset)
                if newline_pos == -1:
                    break
                line_offsets.append(offset)
                offset = newline_pos + 1
    
            # 二分查找定位第一个匹配行
            def get_line_prefix(offset):
                return mmapped_file[offset:offset+len(target_prefix)].decode()
            
            idx = bisect.bisect_left(line_offsets, 0, key=lambda x: get_line_prefix(x))
    
            # 收集所有匹配行
            while idx < len(line_offsets):
                line_start = line_offsets[idx]
                line = mmapped_file[line_start:].split(b'\n', 1)[0].decode().strip()
                if not line.startswith(compressed_input + ":"):
                    break
                _, string = line.split(":", 1)
                potential_matches.append(string)
                idx += 1
    
            mmapped_file.close()
        return potential_matches
    
    # verify_full_hash及主逻辑保持不变
    

2. 改用SQLite数据库存储

将文本文件导入SQLite,利用数据库索引特性实现毫秒级查询,这是效率最高的优化方式之一。

导入脚本(import_to_sqlite.py):

import sqlite3

def create_db():
    conn = sqlite3.connect('hashes.db')
    c = conn.cursor()
    # 创建表,支持同一压缩哈希对应多个字符串
    c.execute('''CREATE TABLE IF NOT EXISTS hashes
                 (compressed_hash TEXT, original_string TEXT)''')
    # 为压缩哈希创建索引,加速查询
    c.execute('CREATE INDEX IF NOT EXISTS idx_compressed_hash ON hashes(compressed_hash)')
    conn.commit()
    return conn

def import_file(conn, file_path):
    c = conn.cursor()
    batch_size = 100000
    batch = []
    with open(file_path, 'r', encoding='utf-8') as f:
        for line in f:
            line = line.strip()
            if not line:
                continue
            compressed_hash, string = line.split(":", 1)
            batch.append((compressed_hash, string))
            if len(batch) >= batch_size:
                c.executemany('INSERT INTO hashes VALUES (?, ?)', batch)
                conn.commit()
                batch = []
        if batch:
            c.executemany('INSERT INTO hashes VALUES (?, ?)', batch)
            conn.commit()

if __name__ == "__main__":
    conn = create_db()
    import_file(conn, 'hashes.txt')
    conn.close()

修改后的search.py:

import hashlib
import sqlite3

def compress_hash(hash_value):
    return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27]

def search_compressed_hash(hash_input):
    compressed_input = compress_hash(hash_input)
    conn = sqlite3.connect('hashes.db')
    c = conn.cursor()
    c.execute('SELECT original_string FROM hashes WHERE compressed_hash = ?', (compressed_input,))
    potential_matches = [row[0] for row in c.fetchall()]
    conn.close()
    return potential_matches

# verify_full_hash及主逻辑保持不变

3. 多进程并行搜索

利用Windows多核CPU,将文件分割为多个块并行搜索,汇总结果后验证。

修改后的search.py:

import hashlib
import mmap
import os
from multiprocessing import Pool, cpu_count

def compress_hash(hash_value):
    return hash_value[6] + hash_value[13] + hash_value[20] + hash_value[27]

def search_chunk(args):
    chunk_start, chunk_end, file_path, compressed_input = args
    potential_matches = []
    target_prefix = f"{compressed_input}:".encode()
    with open(file_path, "r+b") as f:
        mmapped_file = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ)
        # 定位到块的起始行开头
        if chunk_start > 0:
            mmapped_file.seek(chunk_start)
            mmapped_file.readline()
            chunk_start = mmapped_file.tell()
        
        while mmapped_file.tell() < chunk_end:
            line = mmapped_file.readline()
            if not line:
                break
            line = line.strip()
            if line.startswith(target_prefix):
                _, string = line.decode().split(":", 1)
                potential_matches.append(string)
        mmapped_file.close()
    return potential_matches

def search_compressed_hash(hash_input, compressed_file):
    compressed_input = compress_hash(hash_input)
    file_size = os.path.getsize(compressed_file)
    num_processes = cpu_count()
    chunk_size = file_size // num_processes
    chunks = []
    for i in range(num_processes):
        start = i * chunk_size
        end = start + chunk_size if i != num_processes-1 else file_size
        chunks.append((start, end, compressed_file, compressed_input))
    
    with Pool(num_processes) as pool:
        results = pool.map(search_chunk, chunks)
    
    # 合并所有结果
    potential_matches = []
    for res in results:
        potential_matches.extend(res)
    return potential_matches

# verify_full_hash及主逻辑保持不变

4. 优化生成文件结构

修改hash.py,生成文件时按compressed_hash分组写入,避免后续预处理:

  • 在write_hashes_to_file中,用字典按压缩哈希分组缓冲区内容,定期将同一分组的记录写入对应文件(如hashes_0db3.txt),搜索时直接打开对应分组文件即可。

内容的提问来源于stack exchange,提问作者Alon Alush

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 17:49:52