You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何基于现有不良词过滤代码扩展实现良/不良词组合校验?

扩展不良词过滤功能以支持良词校验

我已经完成仅校验不良词的过滤功能,现在需要扩展支持良词校验。需求规则为:当句子包含至少一个良词且无不良词时返回True,否则返回False。现有代码如下,请问如何整合实现该功能?

import re
from dataclasses import dataclass
from typing import AbstractSet, Iterable

from database import Database

WORD_BREAKER = re.compile(r"[-'\w]+")


@dataclass(frozen=True)
class BadWords:
    """
    Bad words database.
    """
    combos: AbstractSet[AbstractSet[str]]

    @staticmethod
    def new(bad_words_collection: Iterable[str]) -> "BadWords":
        """
        Create a new BadWords instance.

        :param bad_words_collection: Iterable[str] - The bad words collection
        :return: BadWords - The bad words instance
        """

        return BadWords(frozenset(
            frozenset(WORD_BREAKER.findall(phrase))
            for phrase in bad_words_collection
        ))

    def matches(self, sentence: str) -> bool:
        """
        Check if the sentence contains any bad words.

        :param sentence: str - The sentence to check
        :return: bool - True if the sentence contains any bad words
        """

        words = frozenset(WORD_BREAKER.findall(sentence.casefold()))
        return any(combo <= words for combo in self.combos)


def get_badword_db() -> BadWords:
    """
    Get the bad words database.

    :return: BadWords - The bad words database
    """
    # Mockdata could be:
    # ['test', 't-shirts', 'world', 'beautiful day']
    return BadWords.new([i[0] for i in Database.get_keyword()])


test_match = get_badword_db().matches('test')

期望新增的良词示例:

['stackoverflow', 'python']

实现思路与代码整合

我们可以复用现有BadWords的逻辑结构,创建对应的GoodWords类,再编写统一的校验函数整合判断逻辑,具体实现如下:

完整整合代码

import re
from dataclasses import dataclass
from typing import AbstractSet, Iterable

from database import Database

WORD_BREAKER = re.compile(r"[-'\w]+")


@dataclass(frozen=True)
class BadWords:
    """
    Bad words database.
    """
    combos: AbstractSet[AbstractSet[str]]

    @staticmethod
    def new(bad_words_collection: Iterable[str]) -> "BadWords":
        """
        Create a new BadWords instance.

        :param bad_words_collection: Iterable[str] - The bad words collection
        :return: BadWords - The bad words instance
        """

        return BadWords(frozenset(
            frozenset(WORD_BREAKER.findall(phrase))
            for phrase in bad_words_collection
        ))

    def matches(self, sentence: str) -> bool:
        """
        Check if the sentence contains any bad words.

        :param sentence: str - The sentence to check
        :return: bool - True if the sentence contains any bad words
        """

        words = frozenset(WORD_BREAKER.findall(sentence.casefold()))
        return any(combo <= words for combo in self.combos)


@dataclass(frozen=True)
class GoodWords:
    """
    Good words database.
    """
    combos: AbstractSet[AbstractSet[str]]

    @staticmethod
    def new(good_words_collection: Iterable[str]) -> "GoodWords":
        """
        Create a new GoodWords instance.

        :param good_words_collection: Iterable[str] - The good words collection
        :return: GoodWords - The good words instance
        """

        return GoodWords(frozenset(
            frozenset(WORD_BREAKER.findall(phrase))
            for phrase in good_words_collection
        ))

    def matches(self, sentence: str) -> bool:
        """
        Check if the sentence contains any good words.

        :param sentence: str - The sentence to check
        :return: bool - True if the sentence contains any good words
        """

        words = frozenset(WORD_BREAKER.findall(sentence.casefold()))
        return any(combo <= words for combo in self.combos)


def get_badword_db() -> BadWords:
    """
    Get the bad words database.

    :return: BadWords - The bad words database
    """
    # Mockdata could be:
    # ['test', 't-shirts', 'world', 'beautiful day']
    return BadWords.new([i[0] for i in Database.get_keyword()])


def get_goodword_db() -> GoodWords:
    """
    Get the good words database.

    :return: GoodWords - The good words database
    """
    # Mockdata could be:
    # ['stackoverflow', 'python']
    return GoodWords.new([i[0] for i in Database.get_good_keyword()])


def check_sentence(sentence: str) -> bool:
    """
    Check if the sentence meets the requirement: contains at least one good word and no bad words.

    :param sentence: str - The sentence to check
    :return: bool - True if meets requirement, else False
    """
    has_bad_word = get_badword_db().matches(sentence)
    has_good_word = get_goodword_db().matches(sentence)
    return not has_bad_word and has_good_word


# 测试示例
test1 = check_sentence('I love python')  # 无不良词+含良词 → True
test2 = check_sentence('test python')    # 含不良词 → False
test3 = check_sentence('hello world')    # 无不良词但无良词 → False
print(test1, test2, test3)

关键说明

  1. 复用结构:GoodWords类完全复用BadWords的逻辑,保证代码一致性,后续若需调整分词或匹配规则,只需修改一处即可。
  2. 核心逻辑:check_sentence函数封装需求规则,先判断是否存在不良词,再判断是否包含良词,仅当两个条件同时满足(无不良词+有良词)时返回True。
  3. 数据库适配:get_goodword_db函数假设数据库提供get_good_keyword()方法获取良词集合,可根据实际数据库结构调整实现。

内容的提问来源于stack exchange,提问作者PythonNewbie

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.20 03:15:43