论坛内容审查程序报错:no space left on device 求助
论坛内容审查程序报错:no space left on device 解决及代码修复
我正在尝试根据违禁词文件对论坛内容进行审查,将违禁词替换为*后重写论坛文件,但运行程序时出现错误:no space left on device。期望输出格式为1个标题、1个空行、日期、用户名、一行消息,示例如下:
Torchlight Forum
1996-09-12T16:30:16
Plato
Are ****** real?
1996-09-12T16:30:54
Socrates
Now let me **** in a figure how far our ****** is enlightened or unenlightened
以下是我的Python代码:
import sys def argumentcheck(): i = 1 arguments = ['','','','',''] taskcheck = False logcheck = False forumcheck = False wordscheck = False peoplecheck = False try: while i<=10: if sys.argv[i]=='-task': arguments[0] = sys.argv[i+1] taskcheck = True i+=2 elif sys.argv[i]=='-log': arguments[1] = sys.argv[i+1] logcheck = True i+=2 elif sys.argv[i]=='-forum': arguments[2] = sys.argv[i+1] forumcheck = True i+=2 elif sys.argv[i]=='-words': arguments[3] = sys.argv[i+1] wordscheck = True i+=2 elif sys.argv[i]=='-people': arguments[4] = sys.argv[i+1] peoplecheck = True i+=2 else: print(f'Wrong argument name {sys.argv[i]}') exit() except IndexError: if not taskcheck: print('No task arguments provided.') exit() elif not logcheck: print('No log arguments provided.') exit() elif not forumcheck: print('No forum arguments provided.') exit() elif not wordscheck: print('No words arguments provided.') exit() elif not peoplecheck: print('No people arguments provided.') exit() else: print('Something is missing, don\'t know what.') exit() if (arguments[0] != 'rank_people') and (arguments[0] != 'validate_forum') and (arguments[0] != 'censor_forum') and (arguments[0] != 'evaluate_forum'): print('Task argument is invalid.') exit() try: forumf = open(arguments[2]) except: print(f'{arguments[2]} cannot be read.') exit() try: wordsf = open(arguments[3]) except: print(f'{arguments[3]} cannot be read.') exit() try: peoplef = open(arguments[4]) except: print(f'{arguments[4]} cannot be read.') exit() print('Moderator program starting...') return arguments def namecheck(name): name = name.replace(' ','').replace('-','') if name.isalpha(): return True else: return False def is_chronological(x,y): j = 0 year1 = '' year2 = '' month1 = '' month2 = '' day1 = '' day2 = '' hour1 = '' hour2 = '' min1 = '' min2 = '' sec1 = '' sec2 = '' while j < len(y): if j < 4: year1 += x[j] year2 += y[j] elif j == 4: try: year1 = int(year1) year2 = int(year2) if x[j] != '-' or y[j] != '-': return 'invalid format' except: return 'invalid format' elif j < 7: month1 += x[j] month2 += y[j] elif j == 7: try: month1 = int(month1) month2 = int(month2) if x[j] != '-' or y[j] != '-': return 'invalid format' except: return 'invalid format' elif j < 10: day1 += x[j] day2 += y[j] elif j == 10: try: day1 = int(day1) day2 = int(day2) if x[j] != 'T' or y[j] != 'T': return 'invalid format' except: return 'invalid format' elif j < 13: hour1 += x[j] hour2 += y[j] elif j == 13: try: hour1 = int(hour1) hour2 = int(hour2) if x[j] != ':' or y[j] != ':': return 'invalid format' except: return 'invalid format' elif j < 16: min1 += x[j] min2 += y[j] elif j == 16: try: min1 = int(min1) min2 = int(min2) if x[j] != ':' or y[j] != ':': return 'invalid format' except: return 'invalid format' elif j < 19: sec1 += x[j] sec2 += y[j] if j == 18: try: sec1 = int(sec1) sec2 = int(sec2) except: return 'invalid format' j+=1 if year2 > year1: return 'T' elif year2 == year1: if month2 > month1: return 'T' elif month2 == month1: if day2 > day1: return 'T' elif day2 == day1: if hour2 > hour1: return 'T' elif hour2 == hour1: if min2 > min1: return 'T' elif min2 == min1: if sec2 > sec1: return 'T' return 'F' def validate_forum(): '''Part 4: Validate forum file''' logf = open(argument[1],'a') forumf = open(argument[2]) header1 = forumf.readline() header2 = forumf.readline() if (header1 == '\n') or (header2 != '\n'): print('Error: forum file read. The forum file header is incorrectly formatted', file=logf) exit() post_before = False i = 3 c_post = '0000-00-00T00:00:00' c_rep = '0000-00-00T00:00:00' while True: date_time = forumf.readline() user_name = forumf.readline() content = forumf.readline() if date_time == '' and user_name == '' and content == '': break if not post_before and (date_time.startswith('\t')): print(f'Error: forum file read. The reply is placed before a post on line {i}', file = logf) exit() else: post_before = True if date_time.startswith('\t'): date_time = date_time.replace('\t','',1) if len(date_time.strip('\n')) != 19: print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf) exit() checking1 = is_chronological(c_post,date_time) checking2 = is_chronological(c_rep,date_time) if checking1 == 'invalid format' or checking2 == 'invalid format': print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf) exit() elif checking1 == 'T' and checking2 == 'T': c_rep = date_time else: print(f'Error: forum file read. The reply is out of chronological order on line {i}',file = logf) exit() if user_name.startswith('\t'): user_name = user_name.replace('\t','',1) else: print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf) exit() if not namecheck(user_name.strip('\n')): print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf) exit() if not content.startswith('\t') or not content.endswith('\n'): print(f'Error: forum file read. The post has an invalid format on line {i+2}',file = logf) else: if len(date_time.strip('\n')) != 19: print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf) exit() checking1 = is_chronological(c_post,date_time) if checking1 == 'invalid format': print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf) exit() elif checking1 == 'T': c_post = date_time else: print(f'Error: forum file read. The post is out of chronological order on line {i}',file = logf) exit() if not namecheck(user_name.strip('\n')): print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf) exit() if not content.endswith('\n'): print(f'Error: forum file read. The post has an invalid format on line {i+2}',file = logf) i+=3 def validate_wordfile(): wordf = open(argument[3]) logf = open(argument[1],'a') header1 = wordf.readline() header2 = wordf.readline() if (header1 == '\n') or (header2 != '\n'): print('Error: words file read. The words file header is incorrectly formatted', file=logf) exit() i = 3 banned_word = [] while True: word = wordf.readline() if word == '': break if not word.endswith('\n'): print(f'Error: words file read. The banned word is invalid on line {i}',file = logf) exit() if word.strip() == '': print(f'Error: words file read. The banned word is invalid on line {i}',file = logf) exit() banned_word.append(word.strip('\n').lower()) i+=1 return banned_word def censor_forum(): banned_word = validate_wordfile() validate_forum() forumf = open(argument[2]) lines_list = [] cha_list = (' ','\t','\n',',','.',"'",'"','!','?','(',')') while True: line = forumf.readline() if line == '': break lines_list.append(line) i = 4 forumf = open(argument[2],'w') while i<len(lines_list): content = lines_list[i] j = -1 k = 0 censored = '' while j<len(content): while k<len(cha_list): censored += content[j] if content[j] == cha_list[k]: a = 0 while a<len(banned_word): b = 0 ban_check = True while b<len(banned_word[a]): if content[j+b+1] != banned_word[a][b]: ban_check = False break b+=1 if ban_check == True: c = 0 while c<len(cha_list): if content[j+b+2] == cha_list: j += len(banned_word[a])-1 censored += ('*' * len(banned_word[a])) c+=1 a+=1 k+=1 j+=1 lines_list[i] = censored i+=3 i=0 while i<len(lines_list): print(lines_list[i],file = forumf) argument = argumentcheck() elif argument[0] == 'censor_forum': censor_forum()
问题分析与修复方案
1. 错误根源
no space left on device表面是磁盘空间不足,但代码逻辑缺陷会触发或加剧这个问题:
- 文件操作顺序错误:打开原文件并以
w模式清空后才开始处理内容,若处理过程中出现无限循环,会持续写入大量无效数据占满磁盘。 - 无限循环逻辑:内容审查的嵌套循环中,
k未在每次j循环时重置,且j起始值为-1导致索引逻辑混乱,可能触发无限循环不断生成内容。 - 违禁词匹配逻辑错误:匹配逻辑不完整,仅检查分隔符后的违禁词,且未正确跳过原词字符,还存在索引越界的风险。
2. 具体修复步骤
(1)调整文件操作顺序
先完成所有内容的审查处理,再打开文件写入,避免提前清空原文件:
def censor_forum(): banned_word = validate_wordfile() validate_forum() # 先读取所有内容 with open(argument[2], 'r') as forumf: lines_list = forumf.readlines() cha_list = {' ', '\t', '\n', ',', '.', "'", '"', '!', '?', '(', ')'} i = 4 while i < len(lines_list): content = lines_list[i] censored = [] j = 0 content_len = len(content) while j < content_len: # 检查当前位置是否是分隔符,或者是内容开头 is_separator = j == 0 or content[j-1] in cha_list if is_separator: # 尝试匹配违禁词 matched = False for word in banned_word: word_len = len(word) if j + word_len > content_len: continue # 大小写不敏感匹配 if content[j:j+word_len].lower() == word: # 检查结尾是否是分隔符或内容结束 if j + word_len == content_len or content[j+word_len] in cha_list: censored.append('*' * word_len) j += word_len matched = True break if not matched: censored.append(content[j]) j += 1 else: censored.append(content[j]) j += 1 lines_list[i] = ''.join(censored) i += 3 # 所有处理完成后再写入原文件 with open(argument[2], 'w') as forumf: forumf.writelines(lines_list)
(2)修复循环逻辑与违禁词匹配
- 用
set存储分隔符,提升查找效率 - 实现大小写不敏感的完整匹配,检查词前后是否为分隔符(或边界)
- 避免嵌套循环导致的逻辑混乱,简化匹配流程
(3)修复主程序语法错误
原代码末尾的elif语法错误,应改为if:
argument = argumentcheck() if argument[0] == 'censor_forum': censor_forum()
3. 额外建议
- 处理大文件时,不要一次性读取所有行到内存,可逐行读取处理并写入临时文件,最后替换原文件,减少内存占用和磁盘风险。
- 定期检查磁盘空间,避免因磁盘满导致程序异常。
内容的提问来源于stack exchange,提问作者DN78
相关产品推荐
相关产品推荐

