You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Flask服务器CSV读写异常:Docker部署下Pandas读取报错

问题

我正在开发一个模块,在Flask服务器中调用PubChem REST接口获取数据生成CSV文件,再将其读取为Pandas DataFrame进行后续处理。本地运行时该模块正常,但在Docker中部署运行时,执行df = pd.read_csv(outfile)步骤会抛出pandas.errors.EmptyDataError: No columns to parse from file错误。我尝试过添加header=None, delim_whitespace=True参数,以及使用StringIO方法,均未解决问题,现寻求Flask服务器中正确的CSV读写方案。

相关代码

def bioassay_post(identifier_list, identifier='cid', output='csv'):
    """ uses PubChem's PUG-Rest service to obtain assay summaries for a target set of chemicals
    via a POST request.

    identifier_list: a list of chemical identifiers, the type of identifier should match with that outlined
    in the 'identifier' argument with the default being PubChem Compound Identifier (CID).

    identifier: the chemical identifier (e.g., cid, smiles, etc.) of the chemicals in 'identifier_list'. Can be any chemical identifier
    as outlined in the PubChem PUG-Rest documentation.  Default=cid

    output: the output format (e.g., csv, json, etc.) of the

    """

    # convert list of identifiers to str
    identifier_list = map(str, identifier_list)

    # make the base URL for the PubChem POST Request
    url = 'https://pubchem.ncbi.nlm.nih.gov/rest/pug/compound/{}/assaysummary/{}'.format(identifier, output)

    # encoded_list = [requests.utils.quote(s) for s in identifier_list]

    # encoded_list = requests.utils.quote(','.join(identifier_list))
    headers = {'Content-Type': 'multipart/form-data'}
    data = {identifier: ','.join(identifier_list)}

    response = requests.post(url, data=data)

    return response


def grouper(iterable, n, fillvalue=None):
    """ support function for bioprofile, used to create n equaled-sized
    sets from an iterable.
    """

    args = [iter(iterable)] * n
    return zip_longest(*args, fillvalue=fillvalue)


def generate_bioprofile(identifier_list, identifier='cid', outfile=os.path.join(os.getcwd(),'bioprofile.csv'), chunk=False):
    num_compounds = len(identifier_list)

    # check to see whether
    # the list should be queried
    # in chunks of data or
    # processed as a whole
    if not chunk:
        chunk_size = num_compounds
    else:
        chunk_size = chunk

    counter = 0

    f = open(outfile, 'w', encoding='utf-8')
    header_written = False

    for gp in grouper(identifier_list, chunk_size):
        batch = [cid for cid in gp if cid]
        response = bioassay_post(batch, identifier=identifier, output='csv')

        if response.status_code == 200:
            text = response.text
            if not header_written:
                f.write(text)
                header_written = True
            else:
                header = text.split('\n')[0]
                text = text.replace(header, '')
                f.write(text)
            counter = counter + len(batch)
        # else:
        #     f.close()
        #     os.remove(outfile)
        #     print("Error: {}".format(response.status_code))
        #     # return
    f.close()

    df = pd.read_csv(outfile)
    os.remove(outfile)
    return df

报错信息

2023-11-24 18:02:31 Traceback (most recent call last):
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/rq/worker.py", line 1075, in perform_job
2023-11-24 18:02:31     rv = job.perform()
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/rq/job.py", line 854, in perform
2023-11-24 18:02:31     self._result = self._execute()
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/rq/job.py", line 877, in _execute
2023-11-24 18:02:31     result = self.func(*self.args, **self.kwargs)
2023-11-24 18:02:31   File "/home/toxpro/./app/tasks.py", line 257, in build_bioprofile
2023-11-24 18:02:31     preprofile = bp.generate_bioprofile(identifier_list)
2023-11-24 18:02:31   File "/home/toxpro/./app/bioprofile.py", line 90, in generate_bioprofile
2023-11-24 18:02:31     df = pd.read_csv(outfile, header=None, delim_whitespace=True)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/util/_decorators.py", line 211, in wrapper
2023-11-24 18:02:31     return func(*args, **kwargs)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/util/_decorators.py", line 331, in wrapper
2023-11-24 18:02:31     return func(*args, **kwargs)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/io/parsers/readers.py", line 950, in read_csv
2023-11-24 18:02:31     return _read(filepath_or_buffer, kwds)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/io/parsers/readers.py", line 605, in _read
2023-11-24 18:02:31     parser = TextFileReader(filepath_or_buffer, **kwds)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/io/parsers/readers.py", line 1442, in __init__
2023-11-24 18:02:31     self._engine = self._make_engine(f, self.engine)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/io/parsers/readers.py", line 1753, in _make_engine
2023-11-24 18:02:31     return mapping[engine](f, **self.options)
2023-11-24 18:02:31   File "/home/toxpro/venv/lib/python3.8/site-packages/pandas/io/parsers/c_parser_wrapper.py", line 79, in __init__
2023-11-24 18:02:31     self._reader = parsers.TextReader(src, **kwds)
2023-11-24 18:02:31   File "pandas/_libs/parsers.pyx", line 554, in pandas._libs.parsers.TextReader.__cinit__
2023-11-24 18:02:31 pandas.errors.EmptyDataError: No columns to parse from file

解决方案

问题根源

  1. 错误的请求头:bioassay_post中设置了multipart/form-data头,但PubChem POST接口实际接受application/x-www-form-urlencoded格式,导致请求返回空内容或错误格式。
  2. 文件系统缓存问题:Docker环境下磁盘缓存可能导致写入内容未完全落地,pd.read_csv读取时文件为空。
  3. 异常处理缺失:请求失败时未终止流程,继续生成空文件。
  4. 路径不确定性:os.getcwd()在容器中可能不是预期可写目录,导致文件写入失败但无报错。

修复步骤

1. 修正请求头

移除multipart/form-data头,让requests自动使用默认的application/x-www-form-urlencoded格式,确保PubChem能正确解析请求。

2. 用内存缓冲区替代磁盘文件

使用io.StringIO直接在内存中处理CSV内容,彻底避免磁盘IO的缓存问题,同时提升处理效率。

3. 完善异常处理

请求失败时直接抛出异常,避免后续处理空内容。

4. 优化表头处理

通过分割行来跳过重复表头,避免replace方法可能带来的内容误替换问题。

修改后的代码

import io
import requests
from itertools import zip_longest
import pandas as pd


def bioassay_post(identifier_list, identifier='cid', output='csv'):
    """调用PubChem的PUG-Rest服务,通过POST请求获取目标化学品的分析摘要"""
    identifier_list = list(map(str, identifier_list))
    url = f'https://pubchem.ncbi.nlm.nih.gov/rest/pug/compound/{identifier}/assaysummary/{output}'
    # 移除错误的Content-Type头,使用默认格式
    data = {identifier: ','.join(identifier_list)}
    response = requests.post(url, data=data)
    # 请求失败直接抛出异常
    response.raise_for_status()
    return response


def grouper(iterable, n, fillvalue=None):
    """将可迭代对象分割为n个等大的分组"""
    args = [iter(iterable)] * n
    return zip_longest(*args, fillvalue=fillvalue)


def generate_bioprofile(identifier_list, identifier='cid', chunk=False):
    num_compounds = len(identifier_list)
    chunk_size = num_compounds if not chunk else chunk

    # 使用内存缓冲区替代磁盘文件
    buffer = io.StringIO()
    header_written = False

    for gp in grouper(identifier_list, chunk_size):
        batch = [cid for cid in gp if cid]
        if not batch:
            continue
        response = bioassay_post(batch, identifier=identifier, output='csv')
        text = response.text

        if not header_written:
            buffer.write(text)
            header_written = True
        else:
            # 分割行后跳过表头,避免误替换
            lines = text.split('\n')
            if len(lines) > 1:
                buffer.write('\n'.join(lines[1:]) + '\n')

    # 将缓冲区指针移到开头,读取为DataFrame
    buffer.seek(0)
    df = pd.read_csv(buffer)
    buffer.close()
    return df

额外注意事项

  • 如果必须使用磁盘文件,建议指定明确的可写目录(如/tmp/bioprofile.csv),并在关闭文件前执行强制写入:
    f.flush()
    os.fsync(f.fileno())
    f.close()
    
  • 确保Docker容器网络可访问PubChem API,可在Dockerfile中添加网络连通性验证。

内容的提问来源于stack exchange,提问作者AnthonyTuzhu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.05 10:54:55