Python中高效读取上千个大CSV文件最后100行的方法
高效读取大量CSV文件最后N行的Python方案
针对你需要遍历1000余个10万行级CSV文件、仅读取最后100行并做判断的场景,直接用pd.read_csv加载全量数据确实会浪费大量内存和时间。以下是几种高效的实现方案:
核心思路
避免加载整个文件,从文件末尾反向读取足够的字节,提取最后N行内容后再用pandas解析,大幅减少IO和内存开销。
方案一:手动反向读取最后N行(最节省内存)
通过文件指针定位到末尾,逐步向前读取并统计换行符数量,直到获取目标行数,再将这些行转为DataFrame:
import os import pandas as pd from io import StringIO def get_last_n_csv_lines(file_path, n=100, encoding='utf-8'): # 处理文件不足n行的情况 with open(file_path, 'r', encoding=encoding) as f: # 先获取文件总大小,从末尾开始读 f.seek(0, os.SEEK_END) file_size = f.tell() buffer = [] newline_count = 0 # 每次向前读1KB,直到收集够n行或读到文件开头 while newline_count <= n and file_size > 0: # 计算每次读取的起始位置,最少读1字节 read_size = min(1024, file_size) file_size -= read_size f.seek(file_size) chunk = f.read(read_size) # 将chunk按换行符拆分,倒序处理(因为是从末尾往前读) lines = chunk.split('\n') # 注意:第一个元素可能是不完整的行,需要和上一次的buffer拼接 if buffer: lines[-1] += buffer[0] buffer = lines[::-1] else: buffer = lines[::-1] newline_count = len(buffer) # 取最后n行,注意要反转回来(因为buffer是倒序存储的) last_lines = buffer[-n:] if len(buffer) >= n else buffer # 拼接成完整的CSV内容,注意要包含表头(如果CSV有表头的话) # 先读取表头行 f.seek(0) header = f.readline() # 拼接表头和最后n行 csv_content = header + '\n'.join(last_lines) return pd.read_csv(StringIO(csv_content)) # 遍历目录处理文件 directory = "/tmp/path/to/csv/" out = [] for filename in os.listdir(directory): if not filename.endswith('.csv'): continue file_path = os.path.join(directory, filename) try: df = get_last_n_csv_lines(file_path, n=100) # 检查最后一行所有值是否小于100(注意排除可能的非数值列) if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all(): out.append(filename) except Exception as e: print(f"处理文件 {filename} 出错: {e}") print("符合条件的文件:", out)
方案二:利用pandas的skiprows(需先获取总行数)
如果能快速获取CSV的总行数,可以用skiprows跳过前面的行,只读取最后100行。但需要先遍历文件统计行数,适合行数稳定的场景:
import os import pandas as pd def count_csv_rows(file_path, encoding='utf-8'): with open(file_path, 'r', encoding=encoding) as f: # 跳过表头,统计数据行数 next(f) return sum(1 for _ in f) directory = "/tmp/path/to/csv/" out = [] for filename in os.listdir(directory): if not filename.endswith('.csv'): continue file_path = os.path.join(directory, filename) try: total_rows = count_csv_rows(file_path) # 计算需要跳过的行数,确保至少读取表头+1行 skip_rows = max(0, total_rows - 100) # skiprows参数:跳过前skip_rows行(从0开始计数,表头是第0行) df = pd.read_csv(file_path, skiprows=range(1, skip_rows+1)) if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all(): out.append(filename) except Exception as e: print(f"处理文件 {filename} 出错: {e}")
方案三:并行处理(加速批量文件遍历)
针对1000+文件的场景,用多进程并行处理可以大幅缩短总耗时:
import os import pandas as pd from io import StringIO from concurrent.futures import ProcessPoolExecutor def process_single_file(file_info): filename, directory = file_info if not filename.endswith('.csv'): return None file_path = os.path.join(directory, filename) try: # 复用方案一的get_last_n_csv_lines函数 def get_last_n_csv_lines(file_path, n=100, encoding='utf-8'): with open(file_path, 'r', encoding=encoding) as f: f.seek(0, os.SEEK_END) file_size = f.tell() buffer = [] newline_count = 0 while newline_count <= n and file_size > 0: read_size = min(1024, file_size) file_size -= read_size f.seek(file_size) chunk = f.read(read_size) lines = chunk.split('\n') if buffer: lines[-1] += buffer[0] buffer = lines[::-1] else: buffer = lines[::-1] newline_count = len(buffer) last_lines = buffer[-n:] if len(buffer) >= n else buffer f.seek(0) header = f.readline() csv_content = header + '\n'.join(last_lines) return pd.read_csv(StringIO(csv_content)) df = get_last_n_csv_lines(file_path, n=100) if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all(): return filename except Exception as e: print(f"处理文件 {filename} 出错: {e}") return None directory = "/tmp/path/to/csv/" file_list = [(filename, directory) for filename in os.listdir(directory)] # 用进程池并行处理,进程数根据CPU核心数调整 with ProcessPoolExecutor(max_workers=os.cpu_count()) as executor: results = executor.map(process_single_file, file_list) out = [res for res in results if res is not None] print("符合条件的文件:", out)
注意事项
- 编码处理:如果CSV文件编码不是utf-8,需要指定对应的
encoding参数(比如gbk)。 - 非数值列判断:如果CSV包含非数值列,需要调整判断逻辑,避免类型错误。
- 文件完整性:处理前可先检查文件是否可读、是否为有效CSV。
内容的提问来源于stack exchange,提问作者llm_dude
相关产品推荐
相关产品推荐

