You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

信息检索代码运行报错:UnicodeDecodeError解码失败求助

问题描述

我开发的信息检索代码用于从语料库的txt文件中提取信息,初期运行正常(检索效果一般但能工作),现在每次运行都报错:

UnicodeDecodeError: 'utf-8' codec can't decode byte 0x80 in position 3131: invalid start byte

代码如下:

import nltk
import sys
import os
import math
from nltk.tokenize import wordpunct_tokenize
import string
import path

FILE_MATCHES = 1
SENTENCE_MATCHES = 1


def main():

    # Check command-line arguments
    if len(sys.argv) != 2:
        sys.exit("Usage: python questions.py corpus")
    else:
        print("Hi! I hope you are having a good time. Thank you for contacting Appen today.")

    # Calculate IDF values across files
    files = load_files(sys.argv[1])

   
    file_words = {
        filename: tokenize(files[filename])
        for filename in files
    }
    file_idfs = compute_idfs(file_words)

    # Prompt user for query
    
    query = set(tokenize(input( "What can I help you with today?: ")))

    # Determine top file matches according to TF-IDF
    filenames = top_files(query, file_words, file_idfs, n=FILE_MATCHES)

    # Extract sentences from top files
    sentences = dict()
    for filename in filenames:
        for passage in files[filename].split("\n"):
            for sentence in nltk.sent_tokenize(passage):
                tokens = tokenize(sentence)
                if tokens:
                    sentences[sentence] = tokens

    # Compute IDF values across sentences
    idfs = compute_idfs(sentences)

    # Determine top sentence matches
    matches = top_sentences(query, sentences, idfs, n=SENTENCE_MATCHES)
    for match in matches:
        print(match)

def load_files(directory):
    """
    Given a directory name, return a dictionary mapping the filename of each
    `.txt` file inside that directory to the file's contents as a string.
    """
    corpus = {}
    abad = os.path.join(os.getcwd(), directory)

    allfiles = os.listdir(directory)

    for filename in allfiles:

        with open((os.path.join(abad, filename)), mode='r', encoding="utf-8") as f:
            doc = f.read().rstrip("\n")
            corpus[filename] = doc

    return corpus


def tokenize(document):
    """
    Given a document (represented as a string), return a list of all of the
    words in that document, in order.
    Process document by coverting all words to lowercase, and removing any
    punctuation or English stopwords.
    """
    token = nltk.tokenize.word_tokenize(document.lower())
    
    document = [x for x in token if x not in string.punctuation and x not in nltk.corpus.stopwords.words("english")]

    return document


def compute_idfs(documents):
    """
    Given a dictionary of `documents` that maps names of documents to a list
    of words, return a dictionary that maps words to their IDF values.
    Any word that appears in at least one of the documents should be in the
    resulting dictionary.
    """
    words = set()
    for filename in documents:
        words.update(documents[filename])

    # Calculate IDFs
    idfs = dict()
    for word in words:
        f = sum(word in documents[filename] for filename in documents)
        idf = math.log(len(documents) / f)
        idfs[word] = idf

    return idfs


def top_files(query, files, idfs, n):
    """
    Given a `query` (a set of words), `files` (a dictionary mapping names of
    files to a list of their words), and `idfs` (a dictionary mapping words
    to their IDF values), return a list of the filenames of the the `n` top
    files that match the query, ranked according to tf-idf.
    """
    #set up tfids dictionary
    tfidfs = {}


    #for the name and content in the files items check if the word is in them and if so add to score
    for filename, filecon in files.items():
        score = 0
        for word in query:
            if word in filecon:
                score += filecon.count(word)* idfs[word]
        if score != 0:
            tfidfs[filename] = score

    # Sort and get top n TF-IDFs for each file
    print("Give me a moment to check that for you.")
    sort = [k for k, v in sorted(tfidfs.items(), key=lambda y: y[1], reverse=True)]

    return sort[:n]


def top_sentences(query, sentences, idfs, n):
    """
    Given a `query` (a set of words), `sentences` (a dictionary mapping
    sentences to a list of their words), and `idfs` (a dictionary mapping words
    to their IDF values), return a list of the `n` top sentences that match
    the query, ranked according to idf. If there are ties, preference should
    be given to sentences that have a higher query term density.
    """
    #set up tfids dictionary
    tfidfs = [] 

    #for the name and content in the sentences check if the word is in them and if so add to score
    for sentence in sentences:
        idf = 0
        match = 0
        for s in query:
            if s in sentences[sentence]:  # if query is in the sentence, add IDFS and record a match
                idf += idfs[s]
                match += 1

        density = float(match)/len(sentences[sentence])  # calculate 'matching word measure'

        tfidfs.append((sentence, idf, density))
    # Sort and get top n TF-IDFs for sentence
    tfidfs.sort(key=lambda x: (x[1], x[2]), reverse=True)
    sort = [x[0] for x in tfidfs]

    return sort [:n]

if __name__ == "__main__":
    main()

我已查阅网络相关解决方案并尝试,但均未解决问题,希望得到帮助。

解决方案
  • 问题根源:报错是因为语料库中存在非UTF-8编码的文件,0x80字节不属于UTF-8的有效起始字节,导致解码失败。
  • 修复方案1:指定正确编码
    先排查文件实际编码(比如Windows常用的gbk、cp1252等),修改load_files函数中的打开编码:
    # 示例:尝试cp1252编码(Windows常见)
    with open(os.path.join(abad, filename), mode='r', encoding="cp1252") as f:
    
  • 修复方案2:容错解码
    如果不确定文件编码,可添加errors参数忽略或替换错误字节,保证程序能继续运行:
    with open(os.path.join(abad, filename), mode='r', encoding="utf-8", errors='ignore') as f:
    # 或者用'replace'把错误字节替换成�
    # with open(os.path.join(abad, filename), mode='r', encoding="utf-8", errors='replace') as f:
    
  • 修复方案3:批量检测文件编码
    可以用chardet库批量检测文件编码,先安装:
    pip install chardet
    
    然后修改load_files函数自动适配编码:
    import chardet
    
    def load_files(directory):
        corpus = {}
        abad = os.path.join(os.getcwd(), directory)
        allfiles = os.listdir(directory)
        for filename in allfiles:
            file_path = os.path.join(abad, filename)
            # 检测文件编码
            with open(file_path, 'rb') as f:
                result = chardet.detect(f.read())
            # 用检测到的编码打开文件
            with open(file_path, mode='r', encoding=result['encoding']) as f:
                doc = f.read().rstrip("\n")
                corpus[filename] = doc
        return corpus
    
  • 额外建议:过滤非txt文件,避免加载其他格式文件导致错误:
    # 在load_files函数中添加判断
    allfiles = [f for f in os.listdir(directory) if f.endswith('.txt')]
    

内容的提问来源于stack exchange,提问作者lauraraexo

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.30 01:48:19