You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

论坛内容审查程序报错:no space left on device 求助

论坛内容审查程序报错:no space left on device 解决及代码修复

我正在尝试根据违禁词文件对论坛内容进行审查,将违禁词替换为*后重写论坛文件,但运行程序时出现错误:no space left on device。期望输出格式为1个标题、1个空行、日期、用户名、一行消息,示例如下:

Torchlight Forum

1996-09-12T16:30:16
Plato
Are ****** real?
1996-09-12T16:30:54
Socrates
Now let me **** in a figure how far our ****** is enlightened or unenlightened

以下是我的Python代码:

import sys

def argumentcheck():
    i = 1
    arguments = ['','','','','']
    taskcheck = False
    logcheck = False
    forumcheck = False
    wordscheck = False
    peoplecheck = False

    try:
        while i<=10:
            if sys.argv[i]=='-task':
                arguments[0] = sys.argv[i+1]
                taskcheck = True
                i+=2
            elif sys.argv[i]=='-log':
                arguments[1] = sys.argv[i+1]
                logcheck = True
                i+=2
            elif sys.argv[i]=='-forum':
                arguments[2] = sys.argv[i+1]
                forumcheck = True
                i+=2
            elif sys.argv[i]=='-words':
                arguments[3] = sys.argv[i+1]
                wordscheck = True
                i+=2
            elif sys.argv[i]=='-people':
                arguments[4] = sys.argv[i+1]
                peoplecheck = True
                i+=2
            else:
                print(f'Wrong argument name {sys.argv[i]}')
                exit()
    except IndexError:
        if not taskcheck:
            print('No task arguments provided.')
            exit()
        elif not logcheck:
            print('No log arguments provided.')
            exit()
        elif not forumcheck:
            print('No forum arguments provided.')
            exit()
        elif not wordscheck:
            print('No words arguments provided.')
            exit()
        elif not peoplecheck:
            print('No people arguments provided.')
            exit()
        else:
            print('Something is missing, don\'t know what.')
            exit()
    if (arguments[0] != 'rank_people') and (arguments[0] != 'validate_forum') and (arguments[0] != 'censor_forum') and (arguments[0] != 'evaluate_forum'):
        print('Task argument is invalid.')
        exit()
    try:
        forumf = open(arguments[2])
    except:
        print(f'{arguments[2]} cannot be read.')
        exit()
    try:
        wordsf = open(arguments[3])
    except:
        print(f'{arguments[3]} cannot be read.')
        exit()
    try:
        peoplef = open(arguments[4])
    except:
        print(f'{arguments[4]} cannot be read.')
        exit()
    print('Moderator program starting...')
    return arguments

def namecheck(name):
    name = name.replace(' ','').replace('-','')
    if name.isalpha():
        return True
    else:
        return False


def is_chronological(x,y):
    j = 0
    year1 = ''
    year2 = ''
    month1 = ''
    month2 = ''
    day1 = ''
    day2 = ''
    hour1 = ''
    hour2 = ''
    min1 = ''
    min2 = ''
    sec1 = ''
    sec2 = ''

    while j < len(y):
        if j < 4:
            year1 += x[j]
            year2 += y[j]
        elif j == 4:
            try:
                year1 = int(year1)
                year2 = int(year2)
                if x[j] != '-' or y[j] != '-':
                    return 'invalid format'
            except:
                return 'invalid format'
        elif j < 7:
            month1 += x[j]
            month2 += y[j]
        elif j == 7:
            try:
                month1 = int(month1)
                month2 = int(month2)
                if x[j] != '-' or y[j] != '-':
                    return 'invalid format'
            except:
                return 'invalid format'
        elif j < 10:
            day1 += x[j]
            day2 += y[j]
        elif j == 10:
            try:
                day1 = int(day1)
                day2 = int(day2)
                if x[j] != 'T' or y[j] != 'T':
                    return 'invalid format'
            except:
                return 'invalid format'
        elif j < 13:
            hour1 += x[j]
            hour2 += y[j]
        elif j == 13:
            try:
                hour1 = int(hour1)
                hour2 = int(hour2)
                if x[j] != ':' or y[j] != ':':
                    return 'invalid format'
            except:
                return 'invalid format'
        elif j < 16:
            min1 += x[j]
            min2 += y[j]
        elif j == 16:
            try:
                min1 = int(min1)
                min2 = int(min2)
                if x[j] != ':' or y[j] != ':':
                    return 'invalid format'
            except:
                return 'invalid format'
        elif j < 19:
            sec1 += x[j]
            sec2 += y[j]
            if j == 18:
                try:
                    sec1 = int(sec1)
                    sec2 = int(sec2)
                except:
                    return 'invalid format'
        j+=1
    if year2 > year1:
        return 'T'
    elif year2 == year1:
        if month2 > month1:
            return 'T'
        elif month2 == month1:
            if day2 > day1:
                return 'T'
            elif day2 == day1:
                if hour2 > hour1:
                    return 'T'
                elif hour2 == hour1:
                    if min2 > min1:
                        return 'T'
                    elif min2 == min1:
                        if sec2 > sec1:
                            return 'T'
        return 'F'
        

def validate_forum():
    '''Part 4: Validate forum file'''
    logf = open(argument[1],'a')
    forumf = open(argument[2])
    header1 = forumf.readline()
    header2 = forumf.readline()
    if (header1 == '\n') or (header2 != '\n'):
        print('Error: forum file read. The forum file header is incorrectly formatted', file=logf)
        exit()
    post_before = False
    i = 3
    c_post = '0000-00-00T00:00:00'
    c_rep = '0000-00-00T00:00:00'
    while True:
        date_time = forumf.readline()
        user_name = forumf.readline()
        content = forumf.readline()
        if date_time == '' and user_name == '' and content == '':
            break
        if not post_before and (date_time.startswith('\t')):
            print(f'Error: forum file read. The reply is placed before a post on line {i}', file = logf)
            exit()
        else:
            post_before = True
        if date_time.startswith('\t'):
            date_time = date_time.replace('\t','',1)
            if len(date_time.strip('\n')) != 19:
                print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf)
                exit()
            checking1 = is_chronological(c_post,date_time)
            checking2 = is_chronological(c_rep,date_time)
            if checking1 == 'invalid format' or checking2 == 'invalid format':
                print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf)
                exit()
            elif checking1 == 'T' and checking2 == 'T':
                c_rep = date_time
            else:
                print(f'Error: forum file read. The reply is out of chronological order on line {i}',file = logf)
                exit()
            if user_name.startswith('\t'):
                user_name = user_name.replace('\t','',1)
            else:
                print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf)
                exit()
            if not namecheck(user_name.strip('\n')):
                print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf)
                exit()
            if not content.startswith('\t') or not content.endswith('\n'):
                print(f'Error: forum file read. The post has an invalid format on line {i+2}',file = logf)
        else:
            if len(date_time.strip('\n')) != 19:
                print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf)
                exit()
            checking1 = is_chronological(c_post,date_time)
            if checking1 == 'invalid format':
                print(f'Error: forum file read. The datetime string is invalid on line {i}',file = logf)
                exit()
            elif checking1 == 'T':
                c_post = date_time
            else:
                print(f'Error: forum file read. The post is out of chronological order on line {i}',file = logf)
                exit()
            if not namecheck(user_name.strip('\n')):
                print(f'Error: forum file read. The user\'s name is invalid on line {i+1}',file = logf)
                exit()
            if not content.endswith('\n'):
                print(f'Error: forum file read. The post has an invalid format on line {i+2}',file = logf)
        i+=3

def validate_wordfile():
    wordf = open(argument[3])
    logf = open(argument[1],'a')
    header1 = wordf.readline()
    header2 = wordf.readline()
    if (header1 == '\n') or (header2 != '\n'):
        print('Error: words file read. The words file header is incorrectly formatted', file=logf)
        exit()
    i = 3
    banned_word = []
    while True:
        word = wordf.readline()
        if word == '':
            break
        if not word.endswith('\n'):
            print(f'Error: words file read. The banned word is invalid on line {i}',file = logf)
            exit()
        if word.strip() == '':
            print(f'Error: words file read. The banned word is invalid on line {i}',file = logf)
            exit()
        banned_word.append(word.strip('\n').lower())
        i+=1
    return banned_word

def censor_forum():
    banned_word = validate_wordfile()
    validate_forum()
    forumf = open(argument[2])
    lines_list = []
    cha_list = (' ','\t','\n',',','.',"'",'"','!','?','(',')')
    while True:
        line = forumf.readline()
        if line == '':
            break
        lines_list.append(line)
    i = 4
    forumf = open(argument[2],'w')
    while i<len(lines_list):
        content = lines_list[i]
        j = -1
        k = 0
        censored = ''
        while j<len(content):
            while k<len(cha_list):
                censored += content[j]
                if content[j] == cha_list[k]:
                    a = 0
                    while a<len(banned_word):
                        b = 0
                        ban_check = True
                        while b<len(banned_word[a]):
                            if content[j+b+1] != banned_word[a][b]:
                                ban_check = False
                                break
                            b+=1
                        if ban_check == True:
                            c = 0
                            while c<len(cha_list):
                                if content[j+b+2] == cha_list:
                                    j += len(banned_word[a])-1
                                    censored += ('*' * len(banned_word[a]))
                                c+=1
                        a+=1
                k+=1
            j+=1
        lines_list[i] = censored
        i+=3
    i=0
    while i<len(lines_list):
        print(lines_list[i],file = forumf)
argument = argumentcheck()
elif argument[0] == 'censor_forum':
    censor_forum()

问题分析与修复方案

1. 错误根源

no space left on device表面是磁盘空间不足,但代码逻辑缺陷会触发或加剧这个问题:

  • 文件操作顺序错误:打开原文件并以w模式清空后才开始处理内容,若处理过程中出现无限循环,会持续写入大量无效数据占满磁盘。
  • 无限循环逻辑:内容审查的嵌套循环中,k未在每次j循环时重置,且j起始值为-1导致索引逻辑混乱,可能触发无限循环不断生成内容。
  • 违禁词匹配逻辑错误:匹配逻辑不完整,仅检查分隔符后的违禁词,且未正确跳过原词字符,还存在索引越界的风险。

2. 具体修复步骤

(1)调整文件操作顺序

先完成所有内容的审查处理,再打开文件写入,避免提前清空原文件:

def censor_forum():
    banned_word = validate_wordfile()
    validate_forum()
    # 先读取所有内容
    with open(argument[2], 'r') as forumf:
        lines_list = forumf.readlines()
    
    cha_list = {' ', '\t', '\n', ',', '.', "'", '"', '!', '?', '(', ')'}
    i = 4
    while i < len(lines_list):
        content = lines_list[i]
        censored = []
        j = 0
        content_len = len(content)
        while j < content_len:
            # 检查当前位置是否是分隔符,或者是内容开头
            is_separator = j == 0 or content[j-1] in cha_list
            if is_separator:
                # 尝试匹配违禁词
                matched = False
                for word in banned_word:
                    word_len = len(word)
                    if j + word_len > content_len:
                        continue
                    # 大小写不敏感匹配
                    if content[j:j+word_len].lower() == word:
                        # 检查结尾是否是分隔符或内容结束
                        if j + word_len == content_len or content[j+word_len] in cha_list:
                            censored.append('*' * word_len)
                            j += word_len
                            matched = True
                            break
                if not matched:
                    censored.append(content[j])
                    j += 1
            else:
                censored.append(content[j])
                j += 1
        lines_list[i] = ''.join(censored)
        i += 3
    
    # 所有处理完成后再写入原文件
    with open(argument[2], 'w') as forumf:
        forumf.writelines(lines_list)
(2)修复循环逻辑与违禁词匹配
  • 用set存储分隔符,提升查找效率
  • 实现大小写不敏感的完整匹配,检查词前后是否为分隔符(或边界)
  • 避免嵌套循环导致的逻辑混乱,简化匹配流程
(3)修复主程序语法错误

原代码末尾的elif语法错误,应改为if:

argument = argumentcheck()
if argument[0] == 'censor_forum':
    censor_forum()

3. 额外建议

  • 处理大文件时,不要一次性读取所有行到内存,可逐行读取处理并写入临时文件,最后替换原文件,减少内存占用和磁盘风险。
  • 定期检查磁盘空间,避免因磁盘满导致程序异常。

内容的提问来源于stack exchange,提问作者DN78

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.16 03:20:13