Pandas concat首次循环后无法新增行问题求助
我实现了一套基于Python的数据合并流程:通过main.py使用watchdog监听Watson的frames文件变化,触发执行functions.py中的逻辑。functions.py会读取Obsidian日志文件生成DataFrame,调用Watson命令导出最新日志并截取最后一行生成DataFrame,再用pd.concat合并两个DataFrame后导出为Markdown文件。首次执行时concat能正确合并新增行,但后续循环执行时经常无法正常添加行,两个DataFrame格式完全一致,已尝试添加time.sleep()等待日志生成仍未解决,请求排查原因并解决。
相关代码
functions.py
import subprocess import pandas as pd from move import move_to_obsidian pd.options.mode.copy_on_write = False OBSIDIAN_MD_PATH = "D://Gdrive//ObsidianVault//WatsonFrames//duzeltme1.md" OBSIDIAN_PROCESSED_PATH = 'reports/4back_to_obsidian.md' def time_formatting(df): df['start'] = pd.to_datetime(df['start'], format="%d/%m/%y %H:%M:%S", errors="ignore") df['stop'] = pd.to_datetime(df['stop'], format="%d/%m/%y %H:%M:%S", errors="ignore") def get_logs_from_obsidian(): df = pd.read_table(OBSIDIAN_MD_PATH, index_col=0, sep="|") df = df.iloc[1: , :] df = df.reset_index(drop=True) try: df = df.drop(' ', axis=1) except: pass try: df = df.dropna(axis=1) except: pass try: for i in range(0, 100, 1): if 'Unnamed ' in df.columns[i]: print(True) df = df.drop(df.columns[i], axis=1) except: pass df.columns = df.columns.str.strip() time_formatting(df=df) # print('-----Obsidian-----') # print(df['start'].tail(2), '//', '\n', df['stop'].tail(2)) df.to_csv('reports/1obsidianlogs.csv', index=False) df.to_html('reports/1obsidianlogs.html') print('-----Obsidian-----' + '\n') return df # get_logs_from_obsidian() def get_logs_from_watson(): subprocess.run(["watson", "log", "--all", "-s", ">", "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv"], shell=True) df = pd.read_csv("new_report.csv") df['notes'] = ' - ' df['id'] = df['id'].apply(lambda x: f"[[WatsonFrames/ids/{x}]]") time_formatting(df) df1 = df.tail(1) df1.reset_index(drop=True, inplace=True) df1.columns = df.columns.str.strip() # print('-----Watson-----') # print(df1.tail(2)) df.to_csv('reports/2watson_notcut_logs.csv') df1.to_csv('reports/2watsonlogs.csv', index=False) df1.to_html('reports/2watsonlogs.html') print('-----Watson-----') return df1 # get_logs_from_watson() def concatting_two_dfs(first_df=get_logs_from_obsidian(), appended_df=get_logs_from_watson()): result = pd.concat([first_df, appended_df], axis=0) result.reset_index(inplace=True, drop=True) result.to_html('reports/3concatting_two_dfs.html') print(result.tail(5)) return result # concatting_two_dfs() def back_to_md(df=concatting_two_dfs(), buf=OBSIDIAN_PROCESSED_PATH): df.to_markdown(index=False, buf=buf) df.to_html(index=False, buf='reports/4back_to_obsidian.html') done = print(f'---- Result can be viewed in {buf} ----') return done def main(): back_to_md() move_to_obsidian() if __name__ == '__main__': main()
main.py
import sys import time import json from watchdog.observers import Observer from watchdog.events import FileSystemEventHandler import subprocess def exec_main(): from functions import main from backup import aws_backup time.sleep(2) aws_backup() time.sleep(1) main() time.sleep(2) aws_backup() class MyHandler(FileSystemEventHandler): def on_modified(self, event): if event.is_directory: return if event.src_path.endswith("frames"): with open(event.src_path, "r") as file: frames = json.load(file) print(f"New entries in frames.json: {frames[-1]}") print('Waiting a few sec for logs to be created') time.sleep(3) exec_main() # Perform your desired action here if __name__ == "__main__": path = "C://Users//sarpy//AppData//Roaming//watson" event_handler = MyHandler() observer = Observer() observer.schedule(event_handler, path, recursive=False) observer.start() print('Observer started') try: while True: time.sleep(1) except KeyboardInterrupt: observer.stop() observer.join()
问题排查与解决
核心问题1:函数默认参数的惰性求值导致重复使用旧数据
concatting_two_dfs和back_to_md函数使用函数调用作为默认参数,Python中默认参数只会在函数定义时求值一次。后续调用这些函数时,不会重新执行get_logs_from_obsidian()和get_logs_from_watson(),而是一直复用第一次执行得到的DataFrame,导致后续合并时没有读取最新的日志数据。
解决方法:
将默认参数改为None,在函数内部判断后再调用获取数据的函数:
def concatting_two_dfs(first_df=None, appended_df=None): if first_df is None: first_df = get_logs_from_obsidian() if appended_df is None: appended_df = get_logs_from_watson() result = pd.concat([first_df, appended_df], axis=0) result.reset_index(inplace=True, drop=True) result.to_html('reports/3concatting_two_dfs.html') print(result.tail(5)) return result def back_to_md(df=None, buf=OBSIDIAN_PROCESSED_PATH): if df is None: df = concatting_two_dfs() df.to_markdown(index=False, buf=buf) df.to_html(index=False, buf='reports/4back_to_obsidian.html') print(f'---- Result can be viewed in {buf} ----')
核心问题2:Watson日志导出可能存在缓存或未完全写入
使用subprocess.run执行Watson命令时,虽然加了time.sleep(),但无法保证命令执行完成后文件已完全写入磁盘。另外,直接用>重定向输出依赖shell环境,可能存在输出缓冲问题。
解决方法:
- 改用
subprocess的stdout参数直接写入文件,避免shell重定向的缓冲问题:
def get_logs_from_watson(): output_path = "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv" with open(output_path, "w", encoding="utf-8") as f: # 执行Watson命令并将输出直接写入文件 subprocess.run(["watson", "log", "--all", "-s"], stdout=f, check=True, shell=False) df = pd.read_csv(output_path) # 后续逻辑保持不变...
- 添加文件读取前的验证,确保文件存在且内容非空:
import os def get_logs_from_watson(): output_path = "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv" # 删除旧文件避免读取缓存 if os.path.exists(output_path): os.remove(output_path) with open(output_path, "w", encoding="utf-8") as f: subprocess.run(["watson", "log", "--all", "-s"], stdout=f, check=True, shell=False) # 等待文件生成并验证内容 retry_count = 0 while retry_count < 5: if os.path.exists(output_path) and os.path.getsize(output_path) > 0: break time.sleep(1) retry_count += 1 if retry_count >=5: raise Exception("Watson日志文件未生成或内容为空") df = pd.read_csv(output_path) # 后续逻辑保持不变...
核心问题3:合并后未去重导致重复行或未正确更新Obsidian文件
每次合并后需要确保Obsidian的源文件(OBSIDIAN_MD_PATH)是最新的合并结果,否则下次读取时还是旧数据。当前流程中move_to_obsidian()的作用是将生成的Markdown文件移到Obsidian路径,但需要确认该函数是否正确覆盖了旧文件,且没有格式丢失。
验证与优化:
- 检查
move_to_obsidian()函数是否正确将OBSIDIAN_PROCESSED_PATH的内容覆盖到OBSIDIAN_MD_PATH,确保格式与pd.read_table读取的格式一致。 - 在合并时添加去重逻辑,避免重复添加同一行数据:
def concatting_two_dfs(first_df=None, appended_df=None): if first_df is None: first_df = get_logs_from_obsidian() if appended_df is None: appended_df = get_logs_from_watson() # 合并前按id去重,保留最新行 combined = pd.concat([first_df, appended_df], axis=0) # 假设id是唯一标识,根据实际情况调整去重键 result = combined.drop_duplicates(subset=['id'], keep='last') result.reset_index(inplace=True, drop=True) result.to_html('reports/3concatting_two_dfs.html') print(result.tail(5)) return result
内容的提问来源于stack exchange,提问作者Sarp Yy

