You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Streamlit MP3转写总结功能异常问题排查求助

问题分析与解决方案

核心问题根源

1. 小文件总结报错 name 'summarized_text' is not defined

Stage 2 中直接引用了未定义的变量summarized_text,既没有调用已实现的总结函数(summarize_text或gpt_summarize_transcript),也没有将转写结果持久化到st.session_state中供后续阶段使用。

2. 大文件上传报错 name 'transcription' is not defined

大文件的转写逻辑被包裹在st.button("start transcription now")的分支中,点击按钮后页面刷新,transcription作为局部变量会被销毁;同时转写完成后未将结果存入st.session_state,导致进入Stage 2后无法读取该变量。

此外,上传的音频文件、转写结果等关键数据未存入st.session_state,Streamlit每次交互都会重新运行脚本,局部变量会丢失,这是状态管理的核心问题。

修复后的完整代码

import streamlit as st
from pydub import AudioSegment
from pydub.silence import split_on_silence
import os
import openai
from transformers import GPT2TokenizerFast, pipeline
import textwrap
from concurrent.futures import ThreadPoolExecutor
import warnings

warnings.filterwarnings("ignore")

# 获取密码与OpenAI密钥
correct_password = st.secrets["password"]["value"]
password_placeholder = st.empty()
password = password_placeholder.text_input("Enter the password", type="password")
if password != correct_password:
    st.error("密码错误")
    st.stop()

openai.api_key = st.secrets["openai"]["key"]

def split_audio(file_path, min_silence_len=500, silence_thresh=-40, chunk_length=30000):
    st.write("正在分割音频为小片段...")
    progress_bar = st.progress(0)
    
    audio = AudioSegment.from_mp3(file_path)
    chunks = split_on_silence(
        audio,
        min_silence_len=min_silence_len,
        silence_thresh=silence_thresh,
        keep_silence=100
    )
    
    split_chunks = []
    for i, chunk in enumerate(chunks):
        if len(chunk) > chunk_length:
            num_mini_chunks = len(chunk) // chunk_length
            for j in range(num_mini_chunks):
                start_time = j * chunk_length
                end_time = start_time + chunk_length
                split_chunks.append(chunk[start_time:end_time])
        else:
            split_chunks.append(chunk)
        
        progress_bar.progress((i + 1) / len(chunks))
            
    return split_chunks

def count_tokens(input_data, max_tokens=20000, input_type='text'):
    tokenizer = GPT2TokenizerFast.from_pretrained("gpt2")
    if input_type == 'text':
        tokens = tokenizer.tokenize(input_data)
    elif input_type == 'tokens':
        tokens = input_data
    else:
        raise ValueError("input_type必须为'text'或'tokens'")
    return len(tokens)

def truncate_text_by_tokens(text, max_tokens=3000):
    tokenizer = GPT2TokenizerFast.from_pretrained("gpt2")
    tokens = tokenizer.tokenize(text)
    truncated_tokens = tokens[:max_tokens]
    truncated_text = tokenizer.convert_tokens_to_string(truncated_tokens)
    return truncated_text

def summarize_chunk(classifier, chunk):
    summary = classifier(chunk)
    return summary[0]["summary_text"]

def summarize_text(text, model_name="t5-small", max_workers=8):
    classifier = pipeline("summarization", model=model_name)
    chunks = textwrap.wrap(text, width=500, break_long_words=False)
    with ThreadPoolExecutor(max_workers=max_workers) as executor:
        summaries = executor.map(lambda chunk: summarize_chunk(classifier, chunk), chunks)
        summarized_text = " ".join(summaries)
    
    summary_token_len = count_tokens(summarized_text)
    if summary_token_len > 2500:
        summarized_text = truncate_text_by_tokens(summarized_text, max_tokens=2500)
    
    with open("transcript_summary.txt", "w") as file:
        file.write(summarized_text)
    return summarized_text.strip()

def gpt_summarize_transcript(transcript_text):
    token_len = count_tokens(transcript_text)
    response = openai.ChatCompletion.create(
        model="gpt-3.5-turbo",
        messages=[
            {"role": "system", "content": "你是专业的文档总结专家,能将长文本提炼为简洁且全面的摘要。"},
            {"role": "user", "content": f"请总结以下转录文本:\n{transcript_text}"}
        ],
        max_tokens=3800 - token_len,
        n=1,
        stop=None,
        temperature=0.5,
    )
    summary = response['choices'][0]['message']['content']
    with open("transcript_summary.txt", "w") as file:
        file.write(summary)
    return summary.strip()

# 初始化会话状态
if "stage" not in st.session_state:
    st.session_state.stage = 0
if "audio_file" not in st.session_state:
    st.session_state.audio_file = None
if "transcription" not in st.session_state:
    st.session_state.transcription = ""
if "summarized_text" not in st.session_state:
    st.session_state.summarized_text = ""

st.title("音频转录与摘要生成工具")

# 阶段0:上传音频文件
if st.session_state.stage == 0:
    audio_file = st.file_uploader("上传MP3音频文件", type=["mp3"])
    if audio_file is not None:
        st.session_state.audio_file = audio_file
        st.session_state.stage = 1

# 阶段1:音频转录
if st.session_state.stage == 1:
    if st.session_state.audio_file is not None:
        try:
            # 写入临时文件
            with open("temp.mp3", "wb") as f:
                f.write(st.session_state.audio_file.getbuffer())
            
            audio_file_size = os.path.getsize("temp.mp3")
            # 大文件处理(>25MB)
            if audio_file_size > 25 * 1024 * 1024:
                if st.button("开始转录"):
                    with st.spinner("正在分割并转录音频..."):
                        chunks = split_audio("temp.mp3")
                        progress_bar = st.progress(0)
                        transcriptions = []
                        for i, chunk in enumerate(chunks):
                            progress_bar.progress((i + 1) / len(chunks))
                            with open("temp_chunk.mp3", "wb") as f:
                                chunk.export(f, format="mp3")
                            with open("temp_chunk.mp3", "rb") as audio:
                                transcription_chunk = openai.Audio.translate("whisper-1", audio)["text"]
                                transcriptions.append(transcription_chunk)
                        st.session_state.transcription = " ".join(transcriptions)
                    st.write("转录结果:", st.session_state.transcription)
                    st.session_state.stage = 2
            # 小文件处理(<=25MB)
            else:
                with st.spinner("正在转录音频..."):
                    with open("temp.mp3", "rb") as audio:
                        st.session_state.transcription = openai.Audio.translate("whisper-1", audio)["text"]
                st.write("转录结果:", st.session_state.transcription)
                st.session_state.stage = 2

        except Exception as e:
            st.error(f"转录出错:{str(e)}")
    
# 阶段2:生成摘要
if st.session_state.stage == 2:
    st.write("当前转录文本:", st.session_state.transcription)
    if st.button("生成摘要"):
        with st.spinner("正在生成摘要..."):
            # 可切换使用summarize_text或gpt_summarize_transcript
            st.session_state.summarized_text = gpt_summarize_transcript(st.session_state.transcription)
            # st.session_state.summarized_text = summarize_text(st.session_state.transcription)
        st.success("摘要生成完成!")
        st.write("摘要内容:", st.session_state.summarized_text)
        if st.button("重新开始"):
            # 重置会话状态
            st.session_state.stage = 0
            st.session_state.transcription = ""
            st.session_state.summarized_text = ""
            st.rerun()

# 清理临时文件
if os.path.exists("temp.mp3"):
    os.remove("temp.mp3")
if os.path.exists("temp_chunk.mp3"):
    os.remove("temp_chunk.mp3")

关键修复点

  1. 会话状态持久化:将audio_file、transcription、summarized_text全部存入st.session_state,确保页面刷新后数据不丢失。
  2. 大文件转写逻辑修正:转写完成后直接将结果存入st.session_state,并在按钮点击分支内完成阶段切换,避免局部变量丢失。
  3. 总结阶段逻辑补全:调用实际的总结函数生成摘要,将结果存入会话状态后再显示。
  4. 用户体验优化:添加st.spinner提示处理状态,修正进度条计算逻辑,添加重新开始功能。
  5. 临时文件清理:脚本结束时清理生成的临时文件,避免冗余文件堆积。

内容的提问来源于stack exchange,提问作者Patrick Schmidt

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.13 03:17:06