43 lines
718 B
Python
43 lines
718 B
Python
import re
|
||
|
||
from app.moderation.profanity.dictionary import ProfanityDictionary
|
||
from app.moderation.profanity.lemmatizer import Lemmatizer
|
||
|
||
class ProfanityDetector:
|
||
|
||
def __init__(
|
||
self,
|
||
dictionary: ProfanityDictionary,
|
||
lemmatizer: Lemmatizer
|
||
):
|
||
|
||
self.dictionary = dictionary
|
||
self.lemmatizer = lemmatizer
|
||
|
||
def detect(
|
||
self,
|
||
text: str
|
||
) -> list[str]:
|
||
|
||
words = re.findall(
|
||
r"[а-яА-ЯёЁ]+",
|
||
text.lower()
|
||
)
|
||
|
||
result = []
|
||
|
||
for word in words:
|
||
|
||
lemma = self.lemmatizer.normalize(word)
|
||
|
||
|
||
for bad_word in self.dictionary.words:
|
||
|
||
if lemma.startswith(bad_word):
|
||
|
||
result.append(word)
|
||
|
||
break
|
||
|
||
|
||
return result |