added image ai-moderation
This commit is contained in:
0
ai-moderation/app/moderation/profanity/__init__.py
Normal file
0
ai-moderation/app/moderation/profanity/__init__.py
Normal file
51
ai-moderation/app/moderation/profanity/detector.py
Normal file
51
ai-moderation/app/moderation/profanity/detector.py
Normal file
@@ -0,0 +1,51 @@
|
||||
import re
|
||||
|
||||
from app.moderation.profanity.dictionary import ProfanityDictionary
|
||||
from app.moderation.profanity.lemmatizer import Lemmatizer
|
||||
|
||||
|
||||
|
||||
class ProfanityDetector:
|
||||
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
dictionary: ProfanityDictionary,
|
||||
lemmatizer: Lemmatizer
|
||||
):
|
||||
|
||||
self.dictionary = dictionary
|
||||
self.lemmatizer = lemmatizer
|
||||
|
||||
|
||||
|
||||
def detect(
|
||||
self,
|
||||
text: str
|
||||
) -> list[str]:
|
||||
|
||||
|
||||
words = re.findall(
|
||||
r"[а-яА-ЯёЁ]+",
|
||||
text.lower()
|
||||
)
|
||||
|
||||
|
||||
result = []
|
||||
|
||||
|
||||
for word in words:
|
||||
|
||||
lemma = self.lemmatizer.normalize(word)
|
||||
|
||||
|
||||
for bad_word in self.dictionary.words:
|
||||
|
||||
if lemma.startswith(bad_word):
|
||||
|
||||
result.append(word)
|
||||
|
||||
break
|
||||
|
||||
|
||||
return result
|
||||
49
ai-moderation/app/moderation/profanity/dictionary.py
Normal file
49
ai-moderation/app/moderation/profanity/dictionary.py
Normal file
@@ -0,0 +1,49 @@
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
class ProfanityDictionary:
|
||||
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
file_path: str
|
||||
):
|
||||
|
||||
self.words = self.load(file_path)
|
||||
|
||||
|
||||
|
||||
def load(
|
||||
self,
|
||||
file_path: str
|
||||
) -> set[str]:
|
||||
|
||||
path = Path(file_path)
|
||||
|
||||
|
||||
if not path.exists():
|
||||
raise FileNotFoundError(
|
||||
f"Profanity dictionary not found: {file_path}"
|
||||
)
|
||||
|
||||
|
||||
with open(
|
||||
path,
|
||||
"r",
|
||||
encoding="utf-8"
|
||||
) as file:
|
||||
|
||||
return {
|
||||
line.strip().lower().replace("\ufeff", "")
|
||||
for line in file
|
||||
if line.strip()
|
||||
}
|
||||
|
||||
|
||||
|
||||
def contains(
|
||||
self,
|
||||
word: str
|
||||
) -> bool:
|
||||
|
||||
return word in self.words
|
||||
20
ai-moderation/app/moderation/profanity/lemmatizer.py
Normal file
20
ai-moderation/app/moderation/profanity/lemmatizer.py
Normal file
@@ -0,0 +1,20 @@
|
||||
import pymorphy3
|
||||
|
||||
|
||||
class Lemmatizer:
|
||||
|
||||
|
||||
def __init__(self):
|
||||
|
||||
self.morph = pymorphy3.MorphAnalyzer()
|
||||
|
||||
|
||||
|
||||
def normalize(
|
||||
self,
|
||||
word: str
|
||||
) -> str:
|
||||
|
||||
result = self.morph.parse(word)
|
||||
|
||||
return result[0].normal_form
|
||||
Reference in New Issue
Block a user