You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

远程集群Jupyter Notebook保存失败:[Errno 28] No space left on device

问题:磁盘空间不足错误但df -h显示空间充足
  • 在远程集群Jupyter Notebook运行论文项目代码,无法保存内容,持续触发错误:[Errno 28] No space left on device
  • 执行df -h检查所有分区,使用率均远低于100%,无法定位问题根源
  • 错误在循环调用函数时触发,尝试%reset -f清除变量无效,排查后锁定问题出在Profile_Fitter类的fit_single_detector方法中调用的resample_on_wavelength函数

完整类代码

# Define the Profile_Fitter class to find the best SEMIMAJOR_AXIS vs FWHM relation for a whole frame (16 detectors)
import numpy as np
import gelsa
from gelsa.sgs import datastore
from tqdm import tqdm

class Profile_Fitter:
    def __init__(self, pt_id_s, cat):
        self.pt_id_s = pt_id_s
        self.cat = cat

    def load_data_to_frame(self, pt_id):
        DS = datastore.DataStore(username='', password='',
                        cachedir='/scratch/astro/benjamin.granett/datastore'
                        )
        # Load the data related to the pointing_id
        file_list = DS.load_sir_pack("DR1_R1", pointing_id_list=[pt_id])
        # Create a Gelsa object
        G = gelsa.Gelsa(config_file="../gelsa-spectra/calib/gelsa_config.json", calibdir="../gelsa-spectra/calib/", zero_order_catalog=None)
        # Define the frame associated to the pointing
        frame = G.load_spec_frame(**file_list[0])

        return frame
        
    def smjax_range(self, smjax_data, center, num=20):
    # Finds the interval around one semi-major axis center, containing "num" objects (20 by default)
        smjax_data = np.asarray(smjax_data)
        # - Compute distance of each object from the center
        dist = np.abs(smjax_data - center)  
        # - Sort by distance and keep the "num" closest objects
        idx = np.argsort(dist)[:num]
        idx = np.sort(idx)
        # - Get the "num" closest objects to "center"
        closest_obj = smjax_data[idx]
        # - Return those and their indices
        return closest_obj, idx

    def fit_single_detector(self, det_n, smjax_centers, pt_id, num_per_range=20, batch_size=5):
        """
        Fit the SEMIMAJOR_AXIS vs FWHM relation for one detector, processing sources in memory-safe batches.
        
        Args:
            det_n (int): Detector number (0-15)
            smjax_centers (list/array): Centers of semi-major axis ranges
            pt_id (int/str): Pointing ID
            num_per_range (int): Number of sources per semi-major axis range
            batch_size (int): Number of sources to process at once (memory control)
        
        Returns:
            tuple: (slope, intercept) from linear regression
        """
        import gc
        from sklearn.linear_model import LinearRegression
        from scipy.signal import peak_widths, find_peaks
        from tqdm import tqdm
        import numpy as np
    
        if not (0 <= det_n < 16):
            print("Detector number not valid\n")
            return
    
        frame = self.load_data_to_frame(pt_id)
        print(f"\nCurrently fitting detector {det_n} in pointing {pt_id}")
    
        # Get sources on this detector
        x1, y1, det1 = frame.radec_to_pixel(self.cat['RIGHT_ASCENSION'], self.cat['DECLINATION'], wavelength=12000)
        x2, y2, det2 = frame.radec_to_pixel(self.cat['RIGHT_ASCENSION'], self.cat['DECLINATION'], wavelength=19000)
        on_detector = (det1 == det_n) & (det2 == det_n)
    
        ra_on_detector = self.cat[on_detector]['RIGHT_ASCENSION']
        dec_on_detector = self.cat[on_detector]['DECLINATION']
    
        # Prepare semi-major axis ranges
        smjax_range_values = []
        indices = []
        for c in smjax_centers:
            masked, index = self.smjax_range(self.cat[on_detector]["SEMIMAJOR_AXIS"], center=c, num=num_per_range)
            smjax_range_values.append(masked)
            indices.append(index)
    
        # Initialize spectra array
        spectra = np.zeros((len(smjax_centers), num_per_range, 11))  # 11 pixels vertically
    
        # Loop over semi-major axis ranges
        for i, indix in enumerate(indices):
            temp = self.cat[on_detector][indix]  # sources in current range
            n_sources = len(temp)
    
            # Process sources in small batches
            for start in range(0, n_sources, batch_size):
                end = min(start + batch_size, n_sources)
                batch_ra = temp["RIGHT_ASCENSION"][start:end]
                batch_dec = temp["DECLINATION"][start:end]
    
                # Process each source in batch
                for j, (ra, dec) in enumerate(zip(batch_ra, batch_dec)):
                    im, var, norm, pix_bins = frame.resample_on_wavelength(ra, dec, wave_range=frame.params['wavelength_range'], super_sample=1)
                    spectra[i, start+j, :] = im.sum(axis=1)
                    
                    # Free memory from this iteration
                    del im, var, norm, pix_bins
                    gc.collect()
    
        # Compute average spectra
        avg_spectra = spectra.mean(axis=1)
    
        # Compute FWHM for main peak
        fwhm = np.zeros(len(avg_spectra))
        for i, av in enumerate(avg_spectra):
            peaks, _ = find_peaks(av)
            if len(peaks) == 0:
                fwhm[i] = np.nan
                continue
            results_half = peak_widths(av, peaks, rel_height=0.5)
            max_peak_fwhm = results_half[0][np.argmax(av[peaks])]
            fwhm[i] = max_peak_fwhm
    
        # Linear fit ignoring NaNs
        valid = ~np.isnan(fwhm)
        reg = LinearRegression().fit(np.array(smjax_centers)[valid].reshape(-1, 1), fwhm[valid])

        return reg.coef_[0], reg.intercept_
        
    def fit_full_frame(self, smjax_centers, pt_id, num_per_range):
        fit_coeffs = np.zeros((16, 2))
        for j in range(16):
            m_, q_ = fit_single_detector(j, smjax_centers, pt_id, num_per_range)
            fit_coeffs[j, :] = m_, q_

        return fit_coeffs
排查解决方向
  • 检查inode使用情况:磁盘空间充足但inode耗尽会触发该错误,执行df -i查看各分区inode使用率,若有分区接近100%,清理大量小文件(如缓存、临时文件)
  • 检查临时目录:程序或Jupyter可能在/tmp等临时目录写入大量临时文件,执行df -h /tmp查看占用情况,若已满则清理该目录,或通过设置TMPDIR环境变量指定其他可用目录
  • 排查resample_on_wavelength函数:该函数可能存在未正确清理的临时文件或未关闭的文件句柄,检查其实现逻辑,确认是否有大量生成临时文件且未自动删除的情况
  • 重启Jupyter内核:会话累积的临时资源可能导致异常,重启内核释放占用的资源
  • 检查用户磁盘配额:集群可能对用户设置了磁盘配额,执行quota -s查看个人配额使用情况,若已达上限联系集群管理员调整

内容的提问来源于stack exchange,提问作者Nicolò Fiaba

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.12 05:14:52