You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在HDF5文件系统中为组创建属性并访问?组结构搭建与属性访问异常排查

修复HDF5中h5md组属性丢失与box组缺失的问题

你遇到的两个核心问题都是结构配置和路径错误导致的,下面是修正后的完整代码,以及关键修改的说明:

修正后的完整代码

import struct
import numpy as np
import h5py
import re

# First part convert the .gro -> .h5 .
csv_file = 'com'
fmtstring = '7s 8s 5s 7s 7s 7s'
fieldstruct = struct.Struct(fmtstring)
parse = fieldstruct.unpack_from

#define a np.dtype for gro array/dataset (hard-coded for now)
gro_dt = np.dtype([('col1', 'S7'), ('col2', 'S8'), ('col3', int), ('col4', float), ('col5', float), ('col6', float)])

with open(csv_file, 'r') as f, \
     h5py.File('xaa.h5', 'w') as hdf:
    # --------------------------
    # 1. 修正h5md组结构与属性配置
    # --------------------------
    # 创建/h5md根组,直接添加version属性
    h5md_grp = hdf.require_group('h5md')
    h5md_grp.attrs['version'] = 1.0
    # 创建h5md下的creator和author子组(可自行添加子组的属性)
    creator_grp = h5md_grp.require_group('creator')
    author_grp = h5md_grp.require_group('author')
    # 示例:给creator子组加属性,根据需求调整
    # creator_grp.attrs['name'] = 'Your Creator Name'
    # author_grp.attrs['name'] = 'Your Author Name'

    # --------------------------
    # 2. 创建并配置particles/lipids/box组
    # --------------------------
    lipids_grp = hdf.require_group('particles/lipids')
    box_grp = lipids_grp.require_group('box')
    # 添加box组的属性
    box_grp.attrs['dimension'] = 3
    box_grp.attrs['boundary'] = np.array(["periodic", "periodic", "periodic"], dtype='S')
    # 创建可动态扩展的edges数据集
    ds_edges = box_grp.create_dataset(
        'edges',
        dtype=np.float32,
        shape=(0, 3),
        maxshape=(None, 3),
        compression='gzip',
        shuffle=True
    )

    # 原有粒子位置相关代码保留,调整路径为particles/lipids/positions
    particles_pos_grp = lipids_grp.require_group('positions')
    ds_time = particles_pos_grp.create_dataset('time', dtype="f", shape=(0,), maxshape=(None,), compression='gzip', shuffle=True)
    ds_step = particles_pos_grp.create_dataset('step', dtype=np.uint64, shape=(0,), maxshape=(None,), compression='gzip', shuffle=True)
    ds_value = None

    step = 0
    while True:
        header = f.readline()
        if not header:
            print("End Of File")
            break
        m = re.search("t= *(.*)$", header)
        time = float(m.group(1)) if m else 0.0

        # get number of data rows, i.e., number of particles
        nparticles_line = f.readline()
        if not nparticles_line:
            break
        nparticles = int(nparticles_line)

        # read data lines and store in array
        arr = np.empty(shape=(nparticles, 3), dtype=np.float32)
        for row in range(nparticles):
            line = f.readline()
            if not line:
                break
            fields = parse(line.encode('utf-8'))
            arr[row] = np.array((float(fields[3]), float(fields[4]), float(fields[5])))

        if nparticles > 0:
            # create a resizable dataset upon the first iteration
            if not ds_value:
                ds_value = particles_pos_grp.create_dataset(
                    'value',
                    dtype=np.float32,
                    shape=(0, nparticles, 3),
                    maxshape=(None, nparticles, 3),
                    chunks=(1, nparticles, 3),
                    compression='gzip',
                    shuffle=True
                )

            # append this sample to the datasets
            ds_time.resize(step + 1, axis=0)
            ds_step.resize(step + 1, axis=0)
            ds_value.resize(step + 1, axis=0)

            ds_time[step] = time
            ds_step[step] = step
            ds_value[step] = arr

        # 读取footer行的box尺寸并追加到edges数据集
        footer = f.readline()
        if footer:
            box_dims = np.array(list(map(float, footer.strip().split())), dtype=np.float32)
            # 确保读取到3个维度的尺寸
            if len(box_dims) == 3:
                ds_edges.resize(step + 1, axis=0)
                ds_edges[step] = box_dims

        step += 1

# --------------------------
# 验证读取修正后的HDF5文件
# --------------------------
with h5py.File('xaa.h5', 'r') as ff:
    print('Items in the base directory: ', list(ff.keys()))
    
    # 读取h5md组的version属性
    h5md_grp = ff['h5md']
    print(f"/h5md组的version属性值: {h5md_grp.attrs['version']}")
    
    # 读取box组的属性和edges数据集
    box_grp = ff['particles/lipids/box']
    print(f"/particles/lipids/box组的dimension属性: {box_grp.attrs['dimension']}")
    print(f"/particles/lipids/box组的boundary属性: {box_grp.attrs['boundary']}")
    print(f"edges数据集的形状: {box_grp['edges'].shape}")
    print(f"最后一帧的box尺寸: {box_grp['edges'][-1] if box_grp['edges'].shape[0]>0 else '无数据'}")

关键修改点解释

  1. h5md组结构修正:
    • 原来的代码错误地把version属性加到了h5md/version/author/creator这个不存在的深层组上,现在改为直接给/h5md根组添加version属性,再创建它的子组creator和author,完全符合需求中的结构。
  2. box组完整配置:
    • 在particles/lipids下创建box组,添加了dimension(值为3)和boundary(周期边界字符串数组)两个属性。
    • 创建了edges数据集,设置maxshape=(None,3)支持动态扩展,每帧读取footer行的box尺寸后追加到数据集中,满足模拟过程中尺寸变化的需求。
  3. 读取逻辑修正:
    • 读取时直接访问/h5md组的attrs就能获取到version属性,无需复杂的路径查找。

内容的提问来源于stack exchange,提问作者Mahesh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.30 12:37:46