You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python中高效读取上千个大CSV文件最后100行的方法

高效读取大量CSV文件最后N行的Python方案

针对你需要遍历1000余个10万行级CSV文件、仅读取最后100行并做判断的场景,直接用pd.read_csv加载全量数据确实会浪费大量内存和时间。以下是几种高效的实现方案:

核心思路

避免加载整个文件,从文件末尾反向读取足够的字节,提取最后N行内容后再用pandas解析,大幅减少IO和内存开销。

方案一:手动反向读取最后N行(最节省内存)

通过文件指针定位到末尾,逐步向前读取并统计换行符数量,直到获取目标行数,再将这些行转为DataFrame:

import os
import pandas as pd
from io import StringIO

def get_last_n_csv_lines(file_path, n=100, encoding='utf-8'):
    # 处理文件不足n行的情况
    with open(file_path, 'r', encoding=encoding) as f:
        # 先获取文件总大小,从末尾开始读
        f.seek(0, os.SEEK_END)
        file_size = f.tell()
        buffer = []
        newline_count = 0
        # 每次向前读1KB,直到收集够n行或读到文件开头
        while newline_count <= n and file_size > 0:
            # 计算每次读取的起始位置,最少读1字节
            read_size = min(1024, file_size)
            file_size -= read_size
            f.seek(file_size)
            chunk = f.read(read_size)
            # 将chunk按换行符拆分,倒序处理(因为是从末尾往前读)
            lines = chunk.split('\n')
            # 注意:第一个元素可能是不完整的行,需要和上一次的buffer拼接
            if buffer:
                lines[-1] += buffer[0]
                buffer = lines[::-1]
            else:
                buffer = lines[::-1]
            newline_count = len(buffer)
        # 取最后n行,注意要反转回来(因为buffer是倒序存储的)
        last_lines = buffer[-n:] if len(buffer) >= n else buffer
        # 拼接成完整的CSV内容,注意要包含表头(如果CSV有表头的话)
        # 先读取表头行
        f.seek(0)
        header = f.readline()
        # 拼接表头和最后n行
        csv_content = header + '\n'.join(last_lines)
        return pd.read_csv(StringIO(csv_content))

# 遍历目录处理文件
directory = "/tmp/path/to/csv/"
out = []

for filename in os.listdir(directory):
    if not filename.endswith('.csv'):
        continue
    file_path = os.path.join(directory, filename)
    try:
        df = get_last_n_csv_lines(file_path, n=100)
        # 检查最后一行所有值是否小于100(注意排除可能的非数值列)
        if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all():
            out.append(filename)
    except Exception as e:
        print(f"处理文件 {filename} 出错: {e}")

print("符合条件的文件:", out)

方案二:利用pandas的skiprows(需先获取总行数)

如果能快速获取CSV的总行数,可以用skiprows跳过前面的行,只读取最后100行。但需要先遍历文件统计行数,适合行数稳定的场景:

import os
import pandas as pd

def count_csv_rows(file_path, encoding='utf-8'):
    with open(file_path, 'r', encoding=encoding) as f:
        # 跳过表头,统计数据行数
        next(f)
        return sum(1 for _ in f)

directory = "/tmp/path/to/csv/"
out = []

for filename in os.listdir(directory):
    if not filename.endswith('.csv'):
        continue
    file_path = os.path.join(directory, filename)
    try:
        total_rows = count_csv_rows(file_path)
        # 计算需要跳过的行数,确保至少读取表头+1行
        skip_rows = max(0, total_rows - 100)
        # skiprows参数:跳过前skip_rows行(从0开始计数,表头是第0行)
        df = pd.read_csv(file_path, skiprows=range(1, skip_rows+1))
        if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all():
            out.append(filename)
    except Exception as e:
        print(f"处理文件 {filename} 出错: {e}")

方案三:并行处理(加速批量文件遍历)

针对1000+文件的场景,用多进程并行处理可以大幅缩短总耗时:

import os
import pandas as pd
from io import StringIO
from concurrent.futures import ProcessPoolExecutor

def process_single_file(file_info):
    filename, directory = file_info
    if not filename.endswith('.csv'):
        return None
    file_path = os.path.join(directory, filename)
    try:
        # 复用方案一的get_last_n_csv_lines函数
        def get_last_n_csv_lines(file_path, n=100, encoding='utf-8'):
            with open(file_path, 'r', encoding=encoding) as f:
                f.seek(0, os.SEEK_END)
                file_size = f.tell()
                buffer = []
                newline_count = 0
                while newline_count <= n and file_size > 0:
                    read_size = min(1024, file_size)
                    file_size -= read_size
                    f.seek(file_size)
                    chunk = f.read(read_size)
                    lines = chunk.split('\n')
                    if buffer:
                        lines[-1] += buffer[0]
                        buffer = lines[::-1]
                    else:
                        buffer = lines[::-1]
                    newline_count = len(buffer)
                last_lines = buffer[-n:] if len(buffer) >= n else buffer
                f.seek(0)
                header = f.readline()
                csv_content = header + '\n'.join(last_lines)
                return pd.read_csv(StringIO(csv_content))
        
        df = get_last_n_csv_lines(file_path, n=100)
        if df.iloc[-1].apply(lambda x: isinstance(x, (int, float)) and x < 100).all():
            return filename
    except Exception as e:
        print(f"处理文件 {filename} 出错: {e}")
        return None

directory = "/tmp/path/to/csv/"
file_list = [(filename, directory) for filename in os.listdir(directory)]

# 用进程池并行处理,进程数根据CPU核心数调整
with ProcessPoolExecutor(max_workers=os.cpu_count()) as executor:
    results = executor.map(process_single_file, file_list)

out = [res for res in results if res is not None]
print("符合条件的文件:", out)

注意事项

  1. 编码处理:如果CSV文件编码不是utf-8,需要指定对应的encoding参数(比如gbk)。
  2. 非数值列判断:如果CSV包含非数值列,需要调整判断逻辑,避免类型错误。
  3. 文件完整性:处理前可先检查文件是否可读、是否为有效CSV。

内容的提问来源于stack exchange,提问作者llm_dude

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 06:00:19