JupyterLab运行网络图代码时出现文件保存错误求助
问题描述
在Anaconda Navigator的JupyterLab 3.2环境中运行Python代码,实现作者-ID网络图生成功能,需循环处理21个TXT输入文件(文件大小从23KB到3782KB不等)。处理第13个837KB的文件时,触发错误:File Save Error for Untitled.ipynb Invalid string length,导致Notebook保存失败。怀疑问题与内存有关,咨询是否需要截断字符串或采用并行处理解决。
原代码
import networkx as nx import matplotlib.pyplot as plt import community def IDauthorsNetwork(ID, authors): idn = nx.Graph() for i, ids in enumerate(ID): for id in ids: if not idn.has_node(id): idn.add_node(id, bipartite=1, count=1) else: idn.nodes[id]['count'] += 1 for a in authors[i]: if not idn.has_node(a): idn.add_node(a, bipartite=0, count=1) else: idn.nodes[a]['count'] += 1 for id in ids: idn.add_edge(id, a, count=1) return idn for year in range(2000, 2022): with open(f'{year}AUID.txt', 'r', encoding='cp949', errors='ignore') as f: lines = f.readlines() auts = [] ids = [] for i, line in enumerate(lines): if line.startswith('AU'): aut = [a.strip() for a in line[3:].split(',')] auts.append(aut) elif line.startswith('ID'): id = [i.strip() for i in line[3:].split(';')] ids.append(id) # Create the ID-author network idn = IDauthorsNetwork(ID=ids, authors=auts) # Detect communities using the Louvain algorithm communities = community.best_partition(idn) # Set the node colors based on their bipartite attribute node_colors = ['orange' if idn.nodes[n]['bipartite'] == 0 else 'lightblue' for n in idn.nodes()] # Set the node sizes based on their count attribute node_sizes = [2000 * idn.nodes[n]['count'] for n in idn.nodes()] # Set the edge widths based on their count attribute edge_widths = [2 * idn.edges[e]['count'] for e in idn.edges] # Set the font size for the node labels font_size = 8 # Set the figure size and resolution fig = plt.figure(figsize=(20, 20), dpi=800) # Draw the graph pos = nx.spring_layout(idn, k=0.2, iterations=50) nx.draw_networkx_nodes(idn, pos, node_size=node_sizes, node_color=node_colors, alpha=0.8) nx.draw_networkx_edges(idn, pos, width=edge_widths, alpha=0.1) nx.draw_networkx_labels(idn, pos, font_size=font_size, font_weight='bold', labels={n: n for n in idn.nodes()}, alpha=0.8) plt.axis('off') plt.savefig(f'{year}world.png') plt.show()
输入文件结构示例
AU Urmeneta, J Navarrete, A Sancho, J ID NATURAL MINERAL WATERS; BACTERIAL-FLORA AU Rasmussen, TT Kirkeby, LP Poulsen, K Reinholdt, J Kilian, M ID RNA GENE-SEQUENCES; CORYNEBACTERIUM-ACCOLENS; PHYLOGENETIC ANALYSIS; ATOPIC DISEASE; SP-NOV; CHILDREN; BACTERIA; IDENTIFICATION; AUREOBACTERIUM; MICROBACTERIUM ...
解决方案分析
一、是否需要截断字符串?
不需要直接截断作者/ID字符串。这个错误本质是Jupyter保存Notebook时,内存中累积的大变量(比如网络图对象、渲染后的图像缓存)导致字符串长度超出限制,而非字符串本身过长。优先从内存优化和代码逻辑修复入手:
- 原代码未处理缩进的续行(比如AU下的第二行作者、ID下的续行),会导致作者/ID信息不全,甚至误读行生成无效节点/边,额外占用内存。
- 处理完每个年份的任务后,主动清理大变量,释放内存。
二、并行处理是否有用?
并行处理对这个场景帮助不大,甚至可能加重内存负担。因为每个年份的任务是独立的,但Jupyter的内存管理以单进程为主,多进程并行会额外占用内存。更有效的方式是改用普通Python脚本(.py)运行,避免Notebook保存时的内存累积问题。
三、具体优化代码
修复输入文件的多行处理逻辑,增加内存清理步骤:
import networkx as nx import matplotlib.pyplot as plt import community import gc def IDauthorsNetwork(ID, authors): idn = nx.Graph() for i, ids in enumerate(ID): # 处理ID节点,跳过空字符串 for id_str in ids: id_str = id_str.strip() if not id_str: continue if idn.has_node(id_str): idn.nodes[id_str]['count'] += 1 else: idn.add_node(id_str, bipartite=1, count=1) # 处理作者节点和边,跳过空字符串 for aut in authors[i]: aut = aut.strip() if not aut: continue if idn.has_node(aut): idn.nodes[aut]['count'] += 1 else: idn.add_node(aut, bipartite=0, count=1) for id_str in ids: id_str = id_str.strip() if not id_str: continue if idn.has_edge(id_str, aut): idn.edges[id_str, aut]['count'] += 1 else: idn.add_edge(id_str, aut, count=1) return idn for year in range(2000, 2022): print(f"正在处理 {year}...") auts = [] ids = [] current_aut = [] current_id = [] with open(f'{year}AUID.txt', 'r', encoding='cp949', errors='ignore') as f: for line in f: line = line.rstrip('\n') if not line.strip(): # 空行表示当前AU-ID组结束,加入列表 if current_aut and current_id: auts.append(current_aut) ids.append(current_id) current_aut = [] current_id = [] continue if line.startswith('AU'): # 处理AU开头的行 aut_part = line[3:].strip() if aut_part: current_aut.extend([a.strip() for a in aut_part.split(',')]) elif line.startswith(' '): # 处理缩进的续行(属于当前AU或ID组) if current_aut: aut_part = line.strip() current_aut.extend([a.strip() for a in aut_part.split(',')]) elif current_id: id_part = line.strip() current_id.extend([i.strip() for i in id_part.split(';')]) elif line.startswith('ID'): # 处理ID开头的行 id_part = line[3:].strip() if id_part: current_id.extend([i.strip() for i in id_part.split(';')]) # 处理文件末尾未结束的AU-ID组 if current_aut and current_id: auts.append(current_aut) ids.append(current_id) # 确保每个AU组对应一个ID组 assert len(auts) == len(ids), f"{year} 年数据:AU与ID组数不匹配" # 生成网络图 idn = IDauthorsNetwork(ID=ids, authors=auts) # 社区检测 communities = community.best_partition(idn) # 绘图参数设置 node_colors = ['orange' if idn.nodes[n]['bipartite'] == 0 else 'lightblue' for n in idn.nodes()] node_sizes = [2000 * idn.nodes[n]['count'] for n in idn.nodes()] edge_widths = [2 * idn.edges[e]['count'] for e in idn.edges] font_size = 8 # 绘图并保存 fig = plt.figure(figsize=(20, 20), dpi=800) pos = nx.spring_layout(idn, k=0.2, iterations=50) nx.draw_networkx_nodes(idn, pos, node_size=node_sizes, node_color=node_colors, alpha=0.8) nx.draw_networkx_edges(idn, pos, width=edge_widths, alpha=0.1) nx.draw_networkx_labels(idn, pos, font_size=font_size, font_weight='bold', labels={n: n for n in idn.nodes()}, alpha=0.8) plt.axis('off') plt.savefig(f'{year}world.png', bbox_inches='tight') plt.close(fig) # 关闭图像释放内存 print(f"{year} 处理完成,已保存图片") # 手动清理大变量,释放内存 del idn, communities, node_colors, node_sizes, edge_widths, pos, fig gc.collect()
四、额外建议
- 优先用普通Python脚本运行:Notebook会记录所有变量和输出,内存占用持续累积,脚本运行完即释放所有内存,避免保存错误。
- 单独测试第13个文件:检查该文件是否存在格式错误或异常数据(比如超长ID/作者名、重复行),导致生成的网络图节点/边数量远超其他文件。
内容的提问来源于stack exchange,提问作者Namie Amuro
相关产品推荐
相关产品推荐

