如何基于现有不良词过滤代码扩展实现良/不良词组合校验?
扩展不良词过滤功能以支持良词校验
我已经完成仅校验不良词的过滤功能,现在需要扩展支持良词校验。需求规则为:当句子包含至少一个良词且无不良词时返回True,否则返回False。现有代码如下,请问如何整合实现该功能?
import re from dataclasses import dataclass from typing import AbstractSet, Iterable from database import Database WORD_BREAKER = re.compile(r"[-'\w]+") @dataclass(frozen=True) class BadWords: """ Bad words database. """ combos: AbstractSet[AbstractSet[str]] @staticmethod def new(bad_words_collection: Iterable[str]) -> "BadWords": """ Create a new BadWords instance. :param bad_words_collection: Iterable[str] - The bad words collection :return: BadWords - The bad words instance """ return BadWords(frozenset( frozenset(WORD_BREAKER.findall(phrase)) for phrase in bad_words_collection )) def matches(self, sentence: str) -> bool: """ Check if the sentence contains any bad words. :param sentence: str - The sentence to check :return: bool - True if the sentence contains any bad words """ words = frozenset(WORD_BREAKER.findall(sentence.casefold())) return any(combo <= words for combo in self.combos) def get_badword_db() -> BadWords: """ Get the bad words database. :return: BadWords - The bad words database """ # Mockdata could be: # ['test', 't-shirts', 'world', 'beautiful day'] return BadWords.new([i[0] for i in Database.get_keyword()]) test_match = get_badword_db().matches('test')
期望新增的良词示例:
['stackoverflow', 'python']
实现思路与代码整合
我们可以复用现有BadWords的逻辑结构,创建对应的GoodWords类,再编写统一的校验函数整合判断逻辑,具体实现如下:
完整整合代码
import re from dataclasses import dataclass from typing import AbstractSet, Iterable from database import Database WORD_BREAKER = re.compile(r"[-'\w]+") @dataclass(frozen=True) class BadWords: """ Bad words database. """ combos: AbstractSet[AbstractSet[str]] @staticmethod def new(bad_words_collection: Iterable[str]) -> "BadWords": """ Create a new BadWords instance. :param bad_words_collection: Iterable[str] - The bad words collection :return: BadWords - The bad words instance """ return BadWords(frozenset( frozenset(WORD_BREAKER.findall(phrase)) for phrase in bad_words_collection )) def matches(self, sentence: str) -> bool: """ Check if the sentence contains any bad words. :param sentence: str - The sentence to check :return: bool - True if the sentence contains any bad words """ words = frozenset(WORD_BREAKER.findall(sentence.casefold())) return any(combo <= words for combo in self.combos) @dataclass(frozen=True) class GoodWords: """ Good words database. """ combos: AbstractSet[AbstractSet[str]] @staticmethod def new(good_words_collection: Iterable[str]) -> "GoodWords": """ Create a new GoodWords instance. :param good_words_collection: Iterable[str] - The good words collection :return: GoodWords - The good words instance """ return GoodWords(frozenset( frozenset(WORD_BREAKER.findall(phrase)) for phrase in good_words_collection )) def matches(self, sentence: str) -> bool: """ Check if the sentence contains any good words. :param sentence: str - The sentence to check :return: bool - True if the sentence contains any good words """ words = frozenset(WORD_BREAKER.findall(sentence.casefold())) return any(combo <= words for combo in self.combos) def get_badword_db() -> BadWords: """ Get the bad words database. :return: BadWords - The bad words database """ # Mockdata could be: # ['test', 't-shirts', 'world', 'beautiful day'] return BadWords.new([i[0] for i in Database.get_keyword()]) def get_goodword_db() -> GoodWords: """ Get the good words database. :return: GoodWords - The good words database """ # Mockdata could be: # ['stackoverflow', 'python'] return GoodWords.new([i[0] for i in Database.get_good_keyword()]) def check_sentence(sentence: str) -> bool: """ Check if the sentence meets the requirement: contains at least one good word and no bad words. :param sentence: str - The sentence to check :return: bool - True if meets requirement, else False """ has_bad_word = get_badword_db().matches(sentence) has_good_word = get_goodword_db().matches(sentence) return not has_bad_word and has_good_word # 测试示例 test1 = check_sentence('I love python') # 无不良词+含良词 → True test2 = check_sentence('test python') # 含不良词 → False test3 = check_sentence('hello world') # 无不良词但无良词 → False print(test1, test2, test3)
关键说明
- 复用结构:
GoodWords类完全复用BadWords的逻辑,保证代码一致性,后续若需调整分词或匹配规则,只需修改一处即可。 - 核心逻辑:
check_sentence函数封装需求规则,先判断是否存在不良词,再判断是否包含良词,仅当两个条件同时满足(无不良词+有良词)时返回True。 - 数据库适配:
get_goodword_db函数假设数据库提供get_good_keyword()方法获取良词集合,可根据实际数据库结构调整实现。
内容的提问来源于stack exchange,提问作者PythonNewbie
相关产品推荐
相关产品推荐

