You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用HFDataset.from_generator多进程生成数据集遇NameError问题

使用HFDataset.from_generator多进程生成数据集时出现NameError: 'Path'未定义的解决方案

我用HFDataset.from_generator生成数据生成器构建数据集,设置了num_proc多进程参数。已经解决了外部函数和类未定义的问题,但现在出现NameError: name 'Path' is not defined错误——明明已经在Notebook里导入了pathlib.Path、json等库。虽然在函数内部导入相关库能临时解决,但要给所有用到的库重复操作,求更合理的解决办法。

相关代码

train_json_files = glob(paths.TRAIN_JSON_FOLDER + "*.json")
from pathlib import Path

def get_gt_string_and_xy(filepath: Union[str, os.PathLike]) -> Dict[str, str]:
    """
    Get the ground truth string and x-y data from the given JSON file.
    :param filepath: The path to the JSON file
    :return dict: A dictionary containing the ground truth string, x-y data, chart type, id, and source
    """
    filepath = Path(filepath)
    with open(filepath) as fp:
        data = json.load(fp)

    all_x, all_y = process_data_series(data.get("data-series", []))
    chart_type = data.get('chart-type', '')
    chart_str = create_chart_string(chart_type)
    x_str = create_coordinate_string("x", all_x)
    y_str = create_coordinate_string("y", all_y)

    gt_string = chart_str + x_str + y_str

    return {
        "ground_truth": gt_string,
        "x": json.dumps(all_x),
        "y": json.dumps(all_y),
        "chart-type": chart_type,
        "id": filepath.stem,
        "source": data.get("source", ''),
    }
def gen_data(files: List[str], paths:paths, get_gt_string_and_xy:callable ) -> Dict[str, str]:
    """
    This function takes a list of json files and returns a generator that yields a
    dictionary with the ground truth string and the path to the image.
    :param files (list): A list of json files
    :return generator: A generator that yields a dictionary with the ground truth string and the path to the corresponding image.
    """
    for f in files:
        # Extract image ID from the file path
        image_id = f.split("/")[-1].split(".")[0]
        # Construct the image path based on the ID
        image_path = paths.TRAIN_IMAGES_FOLDER + image_id + ".jpg"
        # Yield a dictionary containing ground truth string, image path, and other information
        yield {
            **get_gt_string_and_xy(f),
            "image_path": image_path,
        }

ds = HFDataset.from_generator(
    gen_data, gen_kwargs={"files": train_json_files,"paths":paths,"get_gt_string_and_xy":get_gt_string_and_xy}, num_proc=config.NUM_PROCESS
)
print(f"Ground Truth string: \n {ds['ground_truth'][0]}")

报错信息

---------------------------------------------------------------------------
RemoteTraceback                           Traceback (most recent call last)
RemoteTraceback: 
"""
Traceback (most recent call last):
  File "c:\Users\FR00CSS0000000040678\AppData\Local\Programs\Python\Python311\Lib\site-packages\datasets\builder.py", line 1726, in _prepare_split_single
    for key, record in generator:
  File "c:\Users\FR00CSS0000000040678\AppData\Local\Programs\Python\Python311\Lib\site-packages\datasets\packaged_modules\generator\generator.py", line 30, in _generate_examples
    for idx, ex in enumerate(self.config.generator(**gen_kwargs)):
  File "C:\Users\FR00CSS0000000040678\AppData\Local\Temp\ipykernel_22828\3924248845.py", line 18, in gen_data
  File "C:\Users\FR00CSS0000000040678\AppData\Local\Temp\ipykernel_22828\332669057.py", line 28, in get_gt_string_and_xy
NameError: name 'Path' is not defined

The above exception was the direct cause of the following exception:

Traceback (most recent call last):
  File "c:\Users\FR00CSS0000000040678\AppData\Local\Programs\Python\Python311\Lib\site-packages\multiprocess\pool.py", line 125, in worker
    result = (True, func(*args, **kwds))
                    ^^^^^^^^^^^^^^^^^^^
  File "c:\Users\FR00CSS0000000040678\AppData\Local\Programs\Python\Python311\Lib\site-packages\datasets\utils\py_utils.py", line 614, in _write_generator_to_queue
    for i, result in enumerate(func(**kwargs)):
  File "c:\Users\FR00CSS0000000040678\AppData\Local\Programs\Python\Python311\Lib\site-packages\datasets\builder.py", line 1762, in _prepare_split_single
    raise DatasetGenerationError("An error occurred while generating the dataset") from e
datasets.exceptions.DatasetGenerationError: An error occurred while generating the dataset
"""
...
    772     return self._value
    773 else:
--> 774     raise self._value

DatasetGenerationError: An error occurred while generating the dataset

解决方案

方案一:确保核心导入在全局作用域且位置靠前

多进程模式下,子进程不会继承Notebook主进程的所有导入上下文,尤其是当导入语句放在函数定义之后时,子进程可能无法加载到这些模块。把所有需要的导入放在代码最开头,确保子进程启动时能优先加载:

# 全局导入,放在所有函数定义之前
from pathlib import Path
import json
from typing import Union, Dict, List, Callable
# 其他依赖库也统一放在这里

train_json_files = glob(paths.TRAIN_JSON_FOLDER + "*.json")

def get_gt_string_and_xy(filepath: Union[str, os.PathLike]) -> Dict[str, str]:
    # 函数内容不变

方案二:避免传递函数作为参数

你当前把get_gt_string_and_xy作为参数传给gen_data,这会增加子进程解析函数依赖的复杂度。直接在gen_data内部调用该函数,无需通过参数传递:

def gen_data(files: List[str], paths) -> Dict[str, str]:
    for f in files:
        image_id = f.split("/")[-1].split(".")[0]
        image_path = paths.TRAIN_IMAGES_FOLDER + image_id + ".jpg"
        yield {
            **get_gt_string_and_xy(f),
            "image_path": image_path,
        }

# 调用时移除get_gt_string_and_xy参数
ds = HFDataset.from_generator(
    gen_data, gen_kwargs={"files": train_json_files, "paths": paths}, num_proc=config.NUM_PROCESS
)

方案三:统一初始化子进程(进阶)

自定义子进程的初始化函数,在每个子进程启动时统一导入依赖库,确保所有子进程上下文一致:

def init_worker():
    """子进程初始化函数,统一导入依赖"""
    global Path, json
    from pathlib import Path
    import json

# 若datasets库支持,可通过底层参数传递初始化函数
# 或者手动创建带初始化的进程池来生成数据集
# 示例(需结合datasets的API调整):
from multiprocessing import Pool

with Pool(processes=config.NUM_PROCESS, initializer=init_worker) as pool:
    # 这里可以自定义数据集生成逻辑,替代HFDataset.from_generator的默认进程管理
    pass

内容的提问来源于stack exchange,提问作者Shadowpulse

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.01 11:22:07