You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

JupyterLab运行网络图代码时出现文件保存错误求助

问题描述

在Anaconda Navigator的JupyterLab 3.2环境中运行Python代码,实现作者-ID网络图生成功能,需循环处理21个TXT输入文件(文件大小从23KB到3782KB不等)。处理第13个837KB的文件时,触发错误:File Save Error for Untitled.ipynb Invalid string length,导致Notebook保存失败。怀疑问题与内存有关,咨询是否需要截断字符串或采用并行处理解决。

原代码

import networkx as nx
import matplotlib.pyplot as plt
import community

def IDauthorsNetwork(ID, authors):
    idn = nx.Graph()
    for i, ids in enumerate(ID):
        for id in ids:
            if not idn.has_node(id):
                idn.add_node(id, bipartite=1, count=1)
            else:
                idn.nodes[id]['count'] += 1
        for a in authors[i]:
            if not idn.has_node(a):
                idn.add_node(a, bipartite=0, count=1)
            else:
                idn.nodes[a]['count'] += 1
            for id in ids:
                idn.add_edge(id, a, count=1)
    return idn

for year in range(2000, 2022):
    with open(f'{year}AUID.txt', 'r', encoding='cp949', errors='ignore') as f:
        lines = f.readlines()

    auts = []
    ids = []
    for i, line in enumerate(lines):
        if line.startswith('AU'):
            aut = [a.strip() for a in line[3:].split(',')]
            auts.append(aut)
        elif line.startswith('ID'):
            id = [i.strip() for i in line[3:].split(';')]
            ids.append(id)

    # Create the ID-author network
    idn = IDauthorsNetwork(ID=ids, authors=auts)

    # Detect communities using the Louvain algorithm
    communities = community.best_partition(idn)

    # Set the node colors based on their bipartite attribute
    node_colors = ['orange' if idn.nodes[n]['bipartite'] == 0 else 'lightblue' for n in idn.nodes()]

    # Set the node sizes based on their count attribute
    node_sizes = [2000 * idn.nodes[n]['count'] for n in idn.nodes()]

    # Set the edge widths based on their count attribute
    edge_widths = [2 * idn.edges[e]['count'] for e in idn.edges]

    # Set the font size for the node labels
    font_size = 8

    # Set the figure size and resolution
    fig = plt.figure(figsize=(20, 20), dpi=800)

    # Draw the graph
    pos = nx.spring_layout(idn, k=0.2, iterations=50)
    nx.draw_networkx_nodes(idn, pos, node_size=node_sizes, node_color=node_colors, alpha=0.8)
    nx.draw_networkx_edges(idn, pos, width=edge_widths, alpha=0.1)
    nx.draw_networkx_labels(idn, pos, font_size=font_size, font_weight='bold', labels={n: n for n in idn.nodes()}, alpha=0.8)

    plt.axis('off')
    plt.savefig(f'{year}world.png')
    plt.show()

输入文件结构示例

AU Urmeneta, J
   Navarrete, A
   Sancho, J
ID NATURAL MINERAL WATERS; BACTERIAL-FLORA

AU Rasmussen, TT
   Kirkeby, LP
   Poulsen, K
   Reinholdt, J
   Kilian, M
ID RNA GENE-SEQUENCES; CORYNEBACTERIUM-ACCOLENS; PHYLOGENETIC ANALYSIS;
   ATOPIC DISEASE; SP-NOV; CHILDREN; BACTERIA; IDENTIFICATION;
   AUREOBACTERIUM; MICROBACTERIUM

...

解决方案分析

一、是否需要截断字符串?

不需要直接截断作者/ID字符串。这个错误本质是Jupyter保存Notebook时,内存中累积的大变量(比如网络图对象、渲染后的图像缓存)导致字符串长度超出限制,而非字符串本身过长。优先从内存优化和代码逻辑修复入手:

  • 原代码未处理缩进的续行(比如AU下的第二行作者、ID下的续行),会导致作者/ID信息不全,甚至误读行生成无效节点/边,额外占用内存。
  • 处理完每个年份的任务后,主动清理大变量,释放内存。

二、并行处理是否有用?

并行处理对这个场景帮助不大,甚至可能加重内存负担。因为每个年份的任务是独立的,但Jupyter的内存管理以单进程为主,多进程并行会额外占用内存。更有效的方式是改用普通Python脚本(.py)运行,避免Notebook保存时的内存累积问题。

三、具体优化代码

修复输入文件的多行处理逻辑,增加内存清理步骤:

import networkx as nx
import matplotlib.pyplot as plt
import community
import gc

def IDauthorsNetwork(ID, authors):
    idn = nx.Graph()
    for i, ids in enumerate(ID):
        # 处理ID节点,跳过空字符串
        for id_str in ids:
            id_str = id_str.strip()
            if not id_str:
                continue
            if idn.has_node(id_str):
                idn.nodes[id_str]['count'] += 1
            else:
                idn.add_node(id_str, bipartite=1, count=1)
        # 处理作者节点和边,跳过空字符串
        for aut in authors[i]:
            aut = aut.strip()
            if not aut:
                continue
            if idn.has_node(aut):
                idn.nodes[aut]['count'] += 1
            else:
                idn.add_node(aut, bipartite=0, count=1)
            for id_str in ids:
                id_str = id_str.strip()
                if not id_str:
                    continue
                if idn.has_edge(id_str, aut):
                    idn.edges[id_str, aut]['count'] += 1
                else:
                    idn.add_edge(id_str, aut, count=1)
    return idn

for year in range(2000, 2022):
    print(f"正在处理 {year}...")
    auts = []
    ids = []
    current_aut = []
    current_id = []
    
    with open(f'{year}AUID.txt', 'r', encoding='cp949', errors='ignore') as f:
        for line in f:
            line = line.rstrip('\n')
            if not line.strip():
                # 空行表示当前AU-ID组结束,加入列表
                if current_aut and current_id:
                    auts.append(current_aut)
                    ids.append(current_id)
                current_aut = []
                current_id = []
                continue
            if line.startswith('AU'):
                # 处理AU开头的行
                aut_part = line[3:].strip()
                if aut_part:
                    current_aut.extend([a.strip() for a in aut_part.split(',')])
            elif line.startswith('   '):
                # 处理缩进的续行(属于当前AU或ID组)
                if current_aut:
                    aut_part = line.strip()
                    current_aut.extend([a.strip() for a in aut_part.split(',')])
                elif current_id:
                    id_part = line.strip()
                    current_id.extend([i.strip() for i in id_part.split(';')])
            elif line.startswith('ID'):
                # 处理ID开头的行
                id_part = line[3:].strip()
                if id_part:
                    current_id.extend([i.strip() for i in id_part.split(';')])
        # 处理文件末尾未结束的AU-ID组
        if current_aut and current_id:
            auts.append(current_aut)
            ids.append(current_id)
    
    # 确保每个AU组对应一个ID组
    assert len(auts) == len(ids), f"{year} 年数据:AU与ID组数不匹配"
    
    # 生成网络图
    idn = IDauthorsNetwork(ID=ids, authors=auts)
    
    # 社区检测
    communities = community.best_partition(idn)
    
    # 绘图参数设置
    node_colors = ['orange' if idn.nodes[n]['bipartite'] == 0 else 'lightblue' for n in idn.nodes()]
    node_sizes = [2000 * idn.nodes[n]['count'] for n in idn.nodes()]
    edge_widths = [2 * idn.edges[e]['count'] for e in idn.edges]
    font_size = 8
    
    # 绘图并保存
    fig = plt.figure(figsize=(20, 20), dpi=800)
    pos = nx.spring_layout(idn, k=0.2, iterations=50)
    nx.draw_networkx_nodes(idn, pos, node_size=node_sizes, node_color=node_colors, alpha=0.8)
    nx.draw_networkx_edges(idn, pos, width=edge_widths, alpha=0.1)
    nx.draw_networkx_labels(idn, pos, font_size=font_size, font_weight='bold', labels={n: n for n in idn.nodes()}, alpha=0.8)
    
    plt.axis('off')
    plt.savefig(f'{year}world.png', bbox_inches='tight')
    plt.close(fig)  # 关闭图像释放内存
    print(f"{year} 处理完成,已保存图片")
    
    # 手动清理大变量,释放内存
    del idn, communities, node_colors, node_sizes, edge_widths, pos, fig
    gc.collect()

四、额外建议

  1. 优先用普通Python脚本运行:Notebook会记录所有变量和输出,内存占用持续累积,脚本运行完即释放所有内存,避免保存错误。
  2. 单独测试第13个文件:检查该文件是否存在格式错误或异常数据(比如超长ID/作者名、重复行),导致生成的网络图节点/边数量远超其他文件。

内容的提问来源于stack exchange,提问作者Namie Amuro

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.27 09:49:54