远程集群Jupyter Notebook保存失败:[Errno 28] No space left on device
问题:磁盘空间不足错误但
df -h显示空间充足 - 在远程集群Jupyter Notebook运行论文项目代码,无法保存内容,持续触发错误:
[Errno 28] No space left on device - 执行
df -h检查所有分区,使用率均远低于100%,无法定位问题根源 - 错误在循环调用函数时触发,尝试
%reset -f清除变量无效,排查后锁定问题出在Profile_Fitter类的fit_single_detector方法中调用的resample_on_wavelength函数
完整类代码
# Define the Profile_Fitter class to find the best SEMIMAJOR_AXIS vs FWHM relation for a whole frame (16 detectors) import numpy as np import gelsa from gelsa.sgs import datastore from tqdm import tqdm class Profile_Fitter: def __init__(self, pt_id_s, cat): self.pt_id_s = pt_id_s self.cat = cat def load_data_to_frame(self, pt_id): DS = datastore.DataStore(username='', password='', cachedir='/scratch/astro/benjamin.granett/datastore' ) # Load the data related to the pointing_id file_list = DS.load_sir_pack("DR1_R1", pointing_id_list=[pt_id]) # Create a Gelsa object G = gelsa.Gelsa(config_file="../gelsa-spectra/calib/gelsa_config.json", calibdir="../gelsa-spectra/calib/", zero_order_catalog=None) # Define the frame associated to the pointing frame = G.load_spec_frame(**file_list[0]) return frame def smjax_range(self, smjax_data, center, num=20): # Finds the interval around one semi-major axis center, containing "num" objects (20 by default) smjax_data = np.asarray(smjax_data) # - Compute distance of each object from the center dist = np.abs(smjax_data - center) # - Sort by distance and keep the "num" closest objects idx = np.argsort(dist)[:num] idx = np.sort(idx) # - Get the "num" closest objects to "center" closest_obj = smjax_data[idx] # - Return those and their indices return closest_obj, idx def fit_single_detector(self, det_n, smjax_centers, pt_id, num_per_range=20, batch_size=5): """ Fit the SEMIMAJOR_AXIS vs FWHM relation for one detector, processing sources in memory-safe batches. Args: det_n (int): Detector number (0-15) smjax_centers (list/array): Centers of semi-major axis ranges pt_id (int/str): Pointing ID num_per_range (int): Number of sources per semi-major axis range batch_size (int): Number of sources to process at once (memory control) Returns: tuple: (slope, intercept) from linear regression """ import gc from sklearn.linear_model import LinearRegression from scipy.signal import peak_widths, find_peaks from tqdm import tqdm import numpy as np if not (0 <= det_n < 16): print("Detector number not valid\n") return frame = self.load_data_to_frame(pt_id) print(f"\nCurrently fitting detector {det_n} in pointing {pt_id}") # Get sources on this detector x1, y1, det1 = frame.radec_to_pixel(self.cat['RIGHT_ASCENSION'], self.cat['DECLINATION'], wavelength=12000) x2, y2, det2 = frame.radec_to_pixel(self.cat['RIGHT_ASCENSION'], self.cat['DECLINATION'], wavelength=19000) on_detector = (det1 == det_n) & (det2 == det_n) ra_on_detector = self.cat[on_detector]['RIGHT_ASCENSION'] dec_on_detector = self.cat[on_detector]['DECLINATION'] # Prepare semi-major axis ranges smjax_range_values = [] indices = [] for c in smjax_centers: masked, index = self.smjax_range(self.cat[on_detector]["SEMIMAJOR_AXIS"], center=c, num=num_per_range) smjax_range_values.append(masked) indices.append(index) # Initialize spectra array spectra = np.zeros((len(smjax_centers), num_per_range, 11)) # 11 pixels vertically # Loop over semi-major axis ranges for i, indix in enumerate(indices): temp = self.cat[on_detector][indix] # sources in current range n_sources = len(temp) # Process sources in small batches for start in range(0, n_sources, batch_size): end = min(start + batch_size, n_sources) batch_ra = temp["RIGHT_ASCENSION"][start:end] batch_dec = temp["DECLINATION"][start:end] # Process each source in batch for j, (ra, dec) in enumerate(zip(batch_ra, batch_dec)): im, var, norm, pix_bins = frame.resample_on_wavelength(ra, dec, wave_range=frame.params['wavelength_range'], super_sample=1) spectra[i, start+j, :] = im.sum(axis=1) # Free memory from this iteration del im, var, norm, pix_bins gc.collect() # Compute average spectra avg_spectra = spectra.mean(axis=1) # Compute FWHM for main peak fwhm = np.zeros(len(avg_spectra)) for i, av in enumerate(avg_spectra): peaks, _ = find_peaks(av) if len(peaks) == 0: fwhm[i] = np.nan continue results_half = peak_widths(av, peaks, rel_height=0.5) max_peak_fwhm = results_half[0][np.argmax(av[peaks])] fwhm[i] = max_peak_fwhm # Linear fit ignoring NaNs valid = ~np.isnan(fwhm) reg = LinearRegression().fit(np.array(smjax_centers)[valid].reshape(-1, 1), fwhm[valid]) return reg.coef_[0], reg.intercept_ def fit_full_frame(self, smjax_centers, pt_id, num_per_range): fit_coeffs = np.zeros((16, 2)) for j in range(16): m_, q_ = fit_single_detector(j, smjax_centers, pt_id, num_per_range) fit_coeffs[j, :] = m_, q_ return fit_coeffs
排查解决方向
- 检查inode使用情况:磁盘空间充足但inode耗尽会触发该错误,执行
df -i查看各分区inode使用率,若有分区接近100%,清理大量小文件(如缓存、临时文件) - 检查临时目录:程序或Jupyter可能在
/tmp等临时目录写入大量临时文件,执行df -h /tmp查看占用情况,若已满则清理该目录,或通过设置TMPDIR环境变量指定其他可用目录 - 排查
resample_on_wavelength函数:该函数可能存在未正确清理的临时文件或未关闭的文件句柄,检查其实现逻辑,确认是否有大量生成临时文件且未自动删除的情况 - 重启Jupyter内核:会话累积的临时资源可能导致异常,重启内核释放占用的资源
- 检查用户磁盘配额:集群可能对用户设置了磁盘配额,执行
quota -s查看个人配额使用情况,若已达上限联系集群管理员调整
内容的提问来源于stack exchange,提问作者Nicolò Fiaba
相关产品推荐
相关产品推荐

