You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python中按钮函数读取非Excel文件时Plotly交互式绘图报错(文件为空)

Python中按钮函数读取非Excel文件时Plotly交互式绘图报错(文件为空)

我最近写了一个带GUI的Python程序,核心功能是通过按钮调用filedialog.askopenfilename读取多种格式的文件(xls、xlsx、dpt、xy、txt)到pandas DataFrame里,再用Plotly生成交互式绘图。但现在遇到个头疼的问题:读取Excel文件完全正常,但只要选dpt、xy或者txt格式的文件,就会报错说“文件是空的”。

下面是我的完整代码片段,麻烦帮我排查下问题所在:


导入模块

# GUI Imports
import tkinter as tk
from tkinter import ttk
from tkinter import filedialog
from tkinter import messagebox as mb

# Math / Plotting Imports
import numpy as np
import pandas as pd
from scipy.signal import find_peaks as fp
from scipy.signal import peak_widths as pw
import matplotlib.pyplot as plt
from matplotlib.backends.backend_tkagg import FigureCanvasTkAgg
from BaselineRemoval import BaselineRemoval as BR

# Fancy plotting imports
import plotly.graph_objects as go
from plotly.subplots import make_subplots as sub
import plotly.express as px
import plotly.offline as py
import plotly.io as pio

# Other
import webbrowser
import os
import chardet

pd.options.plotting.backend='plotly'

辅助工具函数

def min_max_normalize(array):
    """Normalizes a NumPy array using min-max scaling.
    Args:
        array: A NumPy array.
    Returns:
        A normalized NumPy array.
    """
    min_val = np.min(array)
    max_val = np.max(array)
    normalized_array = (array - min_val) / (max_val - min_val)
    return normalized_array

def find_index(array, value):
    """Finds the index of the nearest value using NumPy."""
    array = np.asarray(array)
    idx = np.argmin(np.abs(array - value))
    return idx

交互式绘图函数

def open_plot():
    # Actual plotting
    Upper_index = find_index(data.iloc[:,0],float(Upper_x))
    Lower_index = find_index(data.iloc[:,0],float(Lower_x))
    x = data.iloc[:,0].values
    intensity = data.iloc[:,1].values
    bg_data = BR(intensity).ZhangFit() # Removes background
    Norm_bg_data = min_max_normalize(bg_data)
    
    peaks,properties = fp(bg_data, prominence=20) # Detects peaks at indices
    widths, width_heights, left_ips,right_ips = pw(bg_data, peaks, rel_height=0.5) # Calculates FWHM
    filtered_peaks = peaks[(peaks>=Lower_index) & (peaks<=Upper_index)] # Removes peaks outside of window of interest
    
    fig = sub(rows=1,cols=3, subplot_titles=('Raw Data','Background Removed Data','Normalized Data'))
    fig.data = []
    fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=intensity[Lower_index:Upper_index], mode='lines'), row=1,col=1)
    fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=bg_data[Lower_index:Upper_index], mode='lines'), row=1,col=2)
    fig.add_trace(go.Scatter(x=x[filtered_peaks],y=bg_data[filtered_peaks], mode='markers', marker=dict(size=4,color='black',symbol='cross', line=dict(width=0.2))), row=1,col=2)
    fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=Norm_bg_data[Lower_index:Upper_index], mode='lines'), row=1,col=3)
    
    fig.update_layout(showlegend=False,autosize=True, plot_bgcolor='white', margin=dict(l=10,r=10,b=20,t=30))
    fig.update_xaxes(showline=True, linecolor='black', linewidth=2.4, ticks='outside', tickcolor='black')
    fig.update_yaxes(title_text='Intensity (a.u.)',showline=True, linecolor='black', linewidth=2.4, ticks='outside', tickcolor='black')
    
    # Finding Maximum peak posiiton
    peak_idx = find_index(bg_data[Lower_index:Upper_index],max(bg_data[Lower_index:Upper_index]))+Lower_index
    peak_pos = x[find_index(bg_data[Lower_index:Upper_index],max(bg_data[Lower_index:Upper_index]))+Lower_index]
    
    # Calculating FWHM
    Left_FWHM = x[int(left_ips[find_index(peaks,peak_idx)])]
    Right_FWHM = x[int(right_ips[find_index(peaks,peak_idx)])]
    FWHM = round(Right_FWHM-Left_FWHM,1)
    
    # Displaying Peak Details
    fig.add_annotation(x=peak_pos, y=max(intensity[Lower_index:Upper_index]), xref='x1', yref='y1', text='Peak: ' + str(peak_pos) + '<br>FWHM: ' + str(FWHM), arrowhead=1, yshift=5)
    fig.add_annotation(x=peak_pos, y=max(bg_data[Lower_index:Upper_index]), xref='x2', yref='y2', text='Peak: ' + str(peak_pos) + '<br>FWHM: ' + str(FWHM), arrowhead=1, yshift=5)
    fig.add_annotation(x=peak_pos, y=max(Norm_bg_data[Lower_index:Upper_index]), xref='x3', yref='y3', text='Peak: ' + str(peak_pos) + '<br>FWHM: ' + str(FWHM), arrowhead=1, yshift=5)
    
    pio.write_html(fig, 'plot.html')
    webbrowser.open('plot.html')
    
    if len(peaks)==0:
        output_label3.set('No peaks found, but here are your plots.')
        fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=intensity[Lower_index:Upper_index], mode='lines'), row=1,col=1)
        fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=bg_data[Lower_index:Upper_index], mode='lines'), row=1,col=2)
        fig.add_trace(go.Scatter(x=x[Lower_index:Upper_index],y=Norm_bg_data[Lower_index:Upper_index], mode='lines'), row=1,col=3)
        pio.write_html(fig, 'plot.html')
        webbrowser.open('plot.html')

未完成的文件读取函数

def open_file():
    file = filedialog.askopenfilename(filetypes=[("DPT files", "*.dpt"), ("Text files", "*.txt"), ("Excel files", "*.xlsx *.xls"),("XY files", "*.xy")])
    if file:
        with open(file, 'rb') as file:
            result = chardet.detect(file.read())
        global data
        try:
            data = pd.r  # 这里代码没写完,应该是想根据文件类型选择读取方式

问题原因分析

你的核心问题出在文件读取的逻辑上:Excel文件(xls/xlsx)需要用pd.read_excel()读取,但dpt、xy、txt这类文本格式的文件,必须用适合文本文件的读取方法(比如pd.read_csv()或pd.read_table()),而且不同文本文件的分隔符、编码都可能不一样,直接混用读取方法就会导致读取失败,触发“文件为空”的错误。

另外你已经用了chardet检测文件编码,但还没把这个编码信息用到文本文件的读取中,这也是一个小疏漏。

修复后的open_file函数

我帮你重构了open_file函数,加入了按文件后缀分支处理的逻辑,同时用上chardet检测到的编码,还加了异常捕获来提示错误:

def open_file():
    global data
    file_path = filedialog.askopenfilename(
        filetypes=[
            ("DPT files", "*.dpt"), 
            ("Text files", "*.txt"), 
            ("Excel files", "*.xlsx *.xls"),
            ("XY files", "*.xy")
        ]
    )
    if not file_path:
        return  # 用户取消选择文件,直接返回
    
    # 获取文件后缀
    file_ext = os.path.splitext(file_path)[1].lower()
    
    try:
        if file_ext in ['.xls', '.xlsx']:
            # 读取Excel文件
            data = pd.read_excel(file_path)
        elif file_ext in ['.dpt', '.xy', '.txt']:
            # 读取文本类文件,先检测编码
            with open(file_path, 'rb') as f:
                raw_data = f.read()
                encoding = chardet.detect(raw_data)['encoding'] or 'utf-8'
            
            # 尝试用空格或制表符作为分隔符(这类科学数据文件常用这两种)
            data = pd.read_csv(
                file_path,
                encoding=encoding,
                sep=r'\s+',  # 匹配任意空白字符(空格、制表符)
                header=None,  # 假设文件没有表头,根据你的实际情况调整
                skip_blank_lines=True  # 跳过空行
            )
        else:
            mb.showerror("错误", "不支持的文件格式!")
            return
        
        # 检查读取后的数据是否为空
        if data.empty:
            mb.showwarning("警告", "读取的文件内容为空!")
            return
        
        mb.showinfo("成功", "文件读取成功!")
        # 这里可以加调用open_plot的逻辑,或者让用户手动触发绘图
        
    except Exception as e:
        mb.showerror("读取失败", f"文件读取出错:{str(e)}")

几个关键调整点:

  1. 按文件后缀分支:明确区分Excel和文本文件的读取方法,避免混用导致的错误。
  2. 编码适配:把chardet检测到的编码传给pd.read_csv(),解决不同编码文本文件的读取问题。
  3. 分隔符处理:用sep=r'\s+'匹配任意空白字符,适配dpt/xy这类常用空格或制表符分隔的科学数据文件。
  4. 异常捕获:用try-except捕获所有读取异常,并用messagebox给出友好提示,方便你排查问题。
  5. 空数据检查:读取后主动检查DataFrame是否为空,提前给出警告。

内容来源于stack exchange

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.08 08:53:05