如何实现Llama GGML模型控制台实时逐字符输出摘要并支持中断?
问题
现有Python代码结合GGML格式的Llama模型,实现了指定文件夹内TXT文件的批量总结(包含文本格式规整、模型生成摘要两个部分),代码可正常运行,但无法像ChatGPT那样在控制台实时逐字符输出生成的摘要内容,需要实现该功能以监控生成进度,且支持随时终止进程。
解决方案
利用llama-cpp-python库的流式生成特性,在调用模型时开启stream=True,迭代获取每个生成的token并实时输出,同时收集完整结果用于保存文件。
修改后的完整代码
import os import re from llama_cpp import Llama input_directory = r"C:\Users\Peter-Susan\Desktop\test" output_directory = r"C:\Users\Peter-Susan\Desktop" def join_lines_in_files(directory_path): try: # 获取目录下所有TXT文件 files = [file for file in os.listdir(directory_path) if file.endswith('.txt')] for file in files: file_path = os.path.join(directory_path, file) with open(file_path, 'r', encoding='utf-8') as f: content = f.read() # 规整文本格式:段落内换行替换为空格,移除行首多余空格,统一单词间空格 content = re.sub(r'(?<=\S)\n+', ' ', content) content = re.sub(r'\n\s+', '\n', content) content = re.sub(r'\s+', ' ', content) # 覆盖原文件保存规整后的内容 with open(file_path, 'w', encoding='utf-8') as f: f.write(content) print("文本格式规整完成。") except Exception as e: print("格式处理出错:", str(e)) # 执行文本格式规整 join_lines_in_files(input_directory) # 加载Llama模型 model_path = "./models/llama-2-7b-chat.ggmlv3.q5_K_M.bin" llm = Llama(model_path=model_path, n_ctx=2048, n_threads=7) def process_query(query, max_tokens=2048, temperature=0.1, top_p=0.5, stop=["#"]): try: generated_text = "" # 开启流式生成,逐token获取结果 for token in llm( query, max_tokens=max_tokens, temperature=temperature, top_p=top_p, stop=stop, stream=True # 关键:开启流式输出 ): token_text = token["choices"][0]["text"] generated_text += token_text # 实时输出到控制台,flush确保立即显示 print(token_text, end='', flush=True) print("\n") # 生成结束后换行 return generated_text.strip() except Exception as e: print("\n生成响应出错:", str(e)) return None def get_title_from_path(file_path): return os.path.splitext(os.path.basename(file_path))[0] def process_text_file(input_file_path, output_directory): # 读取文件内容 with open(input_file_path, "r", encoding="utf-8") as file: file_content = file.read() # 构造模型输入提示 header = 'Summarize in detail with at least 50 words: ' input_text = header + file_content print(f"\n开始处理文件: {input_file_path}") print("输入内容:\n", input_text, "\n") print("生成的摘要:\n") # 调用模型生成摘要(实时输出) if input_text: response = process_query(input_text) if response: # 保存摘要到文件 output_file_path = os.path.join( output_directory, f"{get_title_from_path(input_file_path)}_summarized.txt" ) with open(output_file_path, "w", encoding="utf-8") as output_file: output_file.write(response) print(f"\n摘要已保存到: '{output_file_path}'") else: print(f"\n处理文件失败: '{input_file_path}'") else: print(f"\n无法读取文件内容: '{input_file_path}'") if __name__ == "__main__": # 批量处理目录下所有TXT文件 for filename in os.listdir(input_directory): if filename.endswith(".txt"): file_path = os.path.join(input_directory, filename) process_text_file(file_path, output_directory)
关键修改点
- 在
llm()调用中添加stream=True参数,开启流式生成模式 - 迭代遍历模型返回的token流,逐字符输出到控制台(使用
end=''和flush=True确保实时显示) - 同时收集所有生成的token,拼接成完整的摘要文本用于保存
- 优化控制台输出的排版,让进度监控更清晰
终止进程说明
在控制台实时输出过程中,直接按下Ctrl+C即可终止当前文件的生成进程,程序会捕获异常并停止运行。
内容的提问来源于stack exchange,提问作者jackfood
相关产品推荐
相关产品推荐

