You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Pandas concat首次循环后无法新增行问题求助

Watson日志合并流程后续执行无法新增行的排查与解决

我实现了一套基于Python的数据合并流程:通过main.py使用watchdog监听Watson的frames文件变化,触发执行functions.py中的逻辑。functions.py会读取Obsidian日志文件生成DataFrame,调用Watson命令导出最新日志并截取最后一行生成DataFrame,再用pd.concat合并两个DataFrame后导出为Markdown文件。首次执行时concat能正确合并新增行,但后续循环执行时经常无法正常添加行,两个DataFrame格式完全一致,已尝试添加time.sleep()等待日志生成仍未解决,请求排查原因并解决。

相关代码

functions.py

import subprocess
import pandas as pd
from move import move_to_obsidian

pd.options.mode.copy_on_write = False

OBSIDIAN_MD_PATH = "D://Gdrive//ObsidianVault//WatsonFrames//duzeltme1.md"
OBSIDIAN_PROCESSED_PATH = 'reports/4back_to_obsidian.md'

def time_formatting(df):
    
    df['start'] = pd.to_datetime(df['start'], format="%d/%m/%y %H:%M:%S", errors="ignore")
    df['stop'] = pd.to_datetime(df['stop'], format="%d/%m/%y %H:%M:%S", errors="ignore")


def get_logs_from_obsidian():
    
    df = pd.read_table(OBSIDIAN_MD_PATH, 
                       index_col=0, sep="|")
    
    df = df.iloc[1: , :]
    
    df = df.reset_index(drop=True)
    
    try: df = df.drop('    ', axis=1)    
    except: pass
         
    try: df = df.dropna(axis=1)
    except: pass
    
    try: 
        for i in range(0, 100, 1):
            if 'Unnamed ' in df.columns[i]:
                print(True)
                df = df.drop(df.columns[i], axis=1)
    except:
        pass
    
    df.columns = df.columns.str.strip()
    
    time_formatting(df=df)
    
    # print('-----Obsidian-----')
    # print(df['start'].tail(2), '//', '\n', df['stop'].tail(2))
    df.to_csv('reports/1obsidianlogs.csv', index=False)
    df.to_html('reports/1obsidianlogs.html')
    print('-----Obsidian-----' + '\n')
    
    return df

# get_logs_from_obsidian()


def get_logs_from_watson():
    subprocess.run(["watson", "log", "--all", "-s", ">", "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv"], shell=True)

    df = pd.read_csv("new_report.csv")
    
    df['notes'] = ' - '

    df['id'] = df['id'].apply(lambda x: f"[[WatsonFrames/ids/{x}]]")
    
    time_formatting(df)

    df1 = df.tail(1)
        
    df1.reset_index(drop=True, inplace=True)
    
    df1.columns = df.columns.str.strip()
    
    # print('-----Watson-----')
    # print(df1.tail(2))
    df.to_csv('reports/2watson_notcut_logs.csv')
    df1.to_csv('reports/2watsonlogs.csv', index=False)
    df1.to_html('reports/2watsonlogs.html')
    print('-----Watson-----')
    
    return df1

# get_logs_from_watson()
    

def concatting_two_dfs(first_df=get_logs_from_obsidian(), appended_df=get_logs_from_watson()):
    
    result = pd.concat([first_df, appended_df], axis=0)

    result.reset_index(inplace=True, drop=True)
    
    result.to_html('reports/3concatting_two_dfs.html')
    
    print(result.tail(5))

    return result

# concatting_two_dfs()


def back_to_md(df=concatting_two_dfs(), buf=OBSIDIAN_PROCESSED_PATH):
    
    df.to_markdown(index=False, buf=buf)
    df.to_html(index=False, buf='reports/4back_to_obsidian.html')
    
    done = print(f'---- Result can be viewed in {buf} ----')
    
    return done


def main():
    back_to_md()
    move_to_obsidian()
    
if __name__ == '__main__':
    main()

main.py

import sys
import time

import json
from watchdog.observers import Observer
from watchdog.events import FileSystemEventHandler

import subprocess

def exec_main():
    from functions import main
    from backup import aws_backup
    
    time.sleep(2)
    aws_backup()
    time.sleep(1)
    main()
    time.sleep(2)
    aws_backup()

class MyHandler(FileSystemEventHandler):
  def on_modified(self, event):
    if event.is_directory:
        return

    if event.src_path.endswith("frames"):
        with open(event.src_path, "r") as file:
            frames = json.load(file)
            print(f"New entries in frames.json: {frames[-1]}")
            print('Waiting a few sec for logs to be created')
            time.sleep(3)
            exec_main()
            # Perform your desired action here

if __name__ == "__main__":
  path = "C://Users//sarpy//AppData//Roaming//watson"
  event_handler = MyHandler()
  observer = Observer()
  observer.schedule(event_handler, path, recursive=False)
  observer.start()
  print('Observer started')

  try:
      while True:
          time.sleep(1)
  except KeyboardInterrupt:
      observer.stop()

  observer.join()

问题排查与解决

核心问题1:函数默认参数的惰性求值导致重复使用旧数据

concatting_two_dfs和back_to_md函数使用函数调用作为默认参数,Python中默认参数只会在函数定义时求值一次。后续调用这些函数时,不会重新执行get_logs_from_obsidian()和get_logs_from_watson(),而是一直复用第一次执行得到的DataFrame,导致后续合并时没有读取最新的日志数据。

解决方法:
将默认参数改为None,在函数内部判断后再调用获取数据的函数:

def concatting_two_dfs(first_df=None, appended_df=None):
    if first_df is None:
        first_df = get_logs_from_obsidian()
    if appended_df is None:
        appended_df = get_logs_from_watson()
    
    result = pd.concat([first_df, appended_df], axis=0)
    result.reset_index(inplace=True, drop=True)
    result.to_html('reports/3concatting_two_dfs.html')
    print(result.tail(5))
    return result

def back_to_md(df=None, buf=OBSIDIAN_PROCESSED_PATH):
    if df is None:
        df = concatting_two_dfs()
    
    df.to_markdown(index=False, buf=buf)
    df.to_html(index=False, buf='reports/4back_to_obsidian.html')
    print(f'---- Result can be viewed in {buf} ----')

核心问题2:Watson日志导出可能存在缓存或未完全写入

使用subprocess.run执行Watson命令时,虽然加了time.sleep(),但无法保证命令执行完成后文件已完全写入磁盘。另外,直接用>重定向输出依赖shell环境,可能存在输出缓冲问题。

解决方法:

  1. 改用subprocess的stdout参数直接写入文件,避免shell重定向的缓冲问题:
def get_logs_from_watson():
    output_path = "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv"
    with open(output_path, "w", encoding="utf-8") as f:
        # 执行Watson命令并将输出直接写入文件
        subprocess.run(["watson", "log", "--all", "-s"], stdout=f, check=True, shell=False)
    
    df = pd.read_csv(output_path)
    # 后续逻辑保持不变...
  1. 添加文件读取前的验证,确保文件存在且内容非空:
import os
def get_logs_from_watson():
    output_path = "D:/AUIVVII/Udemy/Inspired/Mildew/new_report.csv"
    # 删除旧文件避免读取缓存
    if os.path.exists(output_path):
        os.remove(output_path)
    
    with open(output_path, "w", encoding="utf-8") as f:
        subprocess.run(["watson", "log", "--all", "-s"], stdout=f, check=True, shell=False)
    
    # 等待文件生成并验证内容
    retry_count = 0
    while retry_count < 5:
        if os.path.exists(output_path) and os.path.getsize(output_path) > 0:
            break
        time.sleep(1)
        retry_count += 1
    
    if retry_count >=5:
        raise Exception("Watson日志文件未生成或内容为空")
    
    df = pd.read_csv(output_path)
    # 后续逻辑保持不变...

核心问题3:合并后未去重导致重复行或未正确更新Obsidian文件

每次合并后需要确保Obsidian的源文件(OBSIDIAN_MD_PATH)是最新的合并结果,否则下次读取时还是旧数据。当前流程中move_to_obsidian()的作用是将生成的Markdown文件移到Obsidian路径,但需要确认该函数是否正确覆盖了旧文件,且没有格式丢失。

验证与优化:

  • 检查move_to_obsidian()函数是否正确将OBSIDIAN_PROCESSED_PATH的内容覆盖到OBSIDIAN_MD_PATH,确保格式与pd.read_table读取的格式一致。
  • 在合并时添加去重逻辑,避免重复添加同一行数据:
def concatting_two_dfs(first_df=None, appended_df=None):
    if first_df is None:
        first_df = get_logs_from_obsidian()
    if appended_df is None:
        appended_df = get_logs_from_watson()
    
    # 合并前按id去重,保留最新行
    combined = pd.concat([first_df, appended_df], axis=0)
    # 假设id是唯一标识,根据实际情况调整去重键
    result = combined.drop_duplicates(subset=['id'], keep='last')
    result.reset_index(inplace=True, drop=True)
    
    result.to_html('reports/3concatting_two_dfs.html')
    print(result.tail(5))
    return result

内容的提问来源于stack exchange,提问作者Sarp Yy

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.12 18:33:09