mirror of
https://github.com/fromchat-messenger/web.git
synced 2026-09-22 19:15:08 +03:00
Restructure backend into microservices, add envelope encryption, DM files, and message editing
This commit is contained in:
@@ -0,0 +1,694 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from threading import RLock
|
||||
from typing import Iterable, List, Set, Tuple
|
||||
|
||||
from better_profanity import Profanity
|
||||
|
||||
BLOCKLIST_PATH = Path("data/profanity/blocklist.json")
|
||||
BLOCKLIST_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
_CUSTOM_RU_TERMS: Set[str] = {
|
||||
"бляд", "блять", "бля", "сука", "суки", "сучка", "мразь", "ебан",
|
||||
"ебать", "ебёт", "ебет", "ебаная", "ебаная", "уёбок", "уебок", "уебище", "пизда",
|
||||
"пиздец", "хуй", "хуя", "хуе", "хуё", "хуйня", "хер", "гондон",
|
||||
"долбоёб", "долбоеб", "дебил", "член", "проститутка", "проститутки",
|
||||
"урод", "хуесос", "хуесосы", "хуесосов", "хуесоса", "сос", "пидор",
|
||||
"пидоры", "пидорас", "пидорасы", "пидорасов",
|
||||
}
|
||||
|
||||
_ADULT_TERMS: Set[str] = {
|
||||
"порно", "порнуха", "эротика", "эротический", "секс", "сексуальный",
|
||||
"инцест", "порнография", "порностудия", "порновидео", "порносайт",
|
||||
"сексчат", "сексчатик", "секслайв", "сексвидео",
|
||||
}
|
||||
|
||||
_STATIC_TERMS: Set[str] = set(term.lower() for term in (_CUSTOM_RU_TERMS | _ADULT_TERMS))
|
||||
|
||||
# Words that should never be flagged as profanity (whitelist)
|
||||
_WHITELIST: Set[str] = {
|
||||
"говно", # Allow this word
|
||||
}
|
||||
|
||||
# Phrase patterns - these will be applied to normalized text (without special chars)
|
||||
_PHRASE_PATTERNS: Tuple[re.Pattern[str], ...] = (
|
||||
re.compile(r"\bmax\s+is\s+better\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bмакс\s+лучше\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bfromchat\s+г[ао]вно\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bфромчат\s+г[ао]вно\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\b18\+\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bxxx\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bайфон\s+топ\b", re.IGNORECASE | re.UNICODE),
|
||||
re.compile(r"\bсамсунг\s+г[ао]вно\b", re.IGNORECASE | re.UNICODE),
|
||||
)
|
||||
|
||||
# Patterns to check in original text (before normalization) to catch visual bypasses
|
||||
# These patterns check for special character combinations that visually form letters
|
||||
_ORIGINAL_TEXT_PATTERNS: Tuple[re.Pattern[str], ...] = (
|
||||
# Catch "}{" used to visually form "х" followed by "С0С" or similar patterns
|
||||
# This catches "хуесос" written as "}{¥€С0С" or variations
|
||||
# Matches: }{ + any characters (including special chars) + С/с + 0 + С/с
|
||||
# The pattern allows any characters between to catch special chars like ¥€
|
||||
re.compile(r"}\{.*?[сcСC].*?[0оoОO].*?[сcСC]", re.IGNORECASE | re.UNICODE),
|
||||
# Also catch "}{" followed by "уесос" with 0 instead of о
|
||||
re.compile(r"}\{.*?[уyУY].*?[еeЕE].*?[сcСC].*?[0оoОO].*?[сcСC]", re.IGNORECASE | re.UNICODE),
|
||||
)
|
||||
|
||||
# Map for normalizing homoglyphs (similar-looking characters)
|
||||
# Maps English/Latin characters to their Cyrillic equivalents and vice versa
|
||||
# Also includes Greek, full-width, and other Unicode variants
|
||||
_LEET_MAP = {
|
||||
# Numbers to letters
|
||||
"0": "о",
|
||||
"1": "и",
|
||||
"3": "е",
|
||||
"4": "а",
|
||||
# Latin to Cyrillic (lowercase)
|
||||
"a": "а",
|
||||
"c": "с",
|
||||
"e": "е",
|
||||
"f": "ф",
|
||||
"g": "г",
|
||||
"i": "и",
|
||||
"m": "м",
|
||||
"n": "н",
|
||||
"o": "о",
|
||||
"p": "п",
|
||||
"s": "с",
|
||||
"t": "т",
|
||||
"u": "у",
|
||||
"v": "в",
|
||||
"x": "х",
|
||||
"y": "у",
|
||||
"z": "з", # English 'z' to Cyrillic 'з'
|
||||
# Latin to Cyrillic (uppercase)
|
||||
"A": "а",
|
||||
"C": "с",
|
||||
"E": "е",
|
||||
"F": "ф",
|
||||
"G": "г",
|
||||
"I": "и",
|
||||
"M": "м",
|
||||
"N": "н",
|
||||
"O": "о",
|
||||
"P": "п",
|
||||
"S": "с",
|
||||
"T": "т",
|
||||
"U": "у",
|
||||
"V": "в",
|
||||
"X": "х",
|
||||
"Y": "у",
|
||||
"Z": "з", # English 'Z' to Cyrillic 'з'
|
||||
# Greek letters that look like Cyrillic/Latin
|
||||
"α": "а", # Greek alpha
|
||||
"Α": "а",
|
||||
"ο": "о", # Greek omicron
|
||||
"Ο": "о",
|
||||
"ρ": "р", # Greek rho (looks like Cyrillic р)
|
||||
"Ρ": "р",
|
||||
"υ": "у", # Greek upsilon
|
||||
"Υ": "у",
|
||||
"χ": "х", # Greek chi
|
||||
"Χ": "х",
|
||||
"ε": "е", # Greek epsilon
|
||||
"Ε": "е",
|
||||
"ι": "и", # Greek iota
|
||||
"Ι": "и",
|
||||
"ν": "н", # Greek nu
|
||||
"Ν": "н",
|
||||
"μ": "м", # Greek mu
|
||||
"Μ": "м",
|
||||
"π": "п", # Greek pi
|
||||
"Π": "п",
|
||||
"τ": "т", # Greek tau
|
||||
"Τ": "т",
|
||||
"γ": "г", # Greek gamma
|
||||
"Γ": "г",
|
||||
"σ": "с", # Greek sigma
|
||||
"Σ": "с",
|
||||
"φ": "ф", # Greek phi
|
||||
"Φ": "ф",
|
||||
# Full-width Latin characters
|
||||
"a": "а",
|
||||
"A": "а",
|
||||
"c": "с",
|
||||
"C": "с",
|
||||
"e": "е",
|
||||
"E": "е",
|
||||
"f": "ф",
|
||||
"F": "ф",
|
||||
"g": "г",
|
||||
"G": "г",
|
||||
"i": "и",
|
||||
"I": "и",
|
||||
"m": "м",
|
||||
"M": "м",
|
||||
"n": "н",
|
||||
"N": "н",
|
||||
"o": "о",
|
||||
"O": "о",
|
||||
"p": "п",
|
||||
"P": "п",
|
||||
"s": "с",
|
||||
"S": "с",
|
||||
"t": "т",
|
||||
"T": "т",
|
||||
"u": "у",
|
||||
"U": "у",
|
||||
"v": "в",
|
||||
"V": "в",
|
||||
"x": "х",
|
||||
"X": "х",
|
||||
"y": "у",
|
||||
"Y": "у",
|
||||
"z": "з", # Full-width 'z' to Cyrillic 'з'
|
||||
"Z": "з",
|
||||
# Cyrillic to canonical Cyrillic (identity mappings)
|
||||
"а": "а",
|
||||
"с": "с",
|
||||
"е": "е",
|
||||
"ё": "е",
|
||||
"ф": "ф",
|
||||
"г": "г",
|
||||
"и": "и",
|
||||
"м": "м",
|
||||
"н": "н",
|
||||
"о": "о",
|
||||
"п": "п",
|
||||
"т": "т",
|
||||
"у": "у",
|
||||
"ү": "у", # Cyrillic capital U (U+04AE)
|
||||
"Ү": "у", # Cyrillic capital U (U+04AE)
|
||||
"в": "в",
|
||||
"х": "х",
|
||||
"р": "р",
|
||||
"з": "з", # Cyrillic 'з'
|
||||
"д": "д", # Cyrillic 'д'
|
||||
"б": "б", # Cyrillic 'б'
|
||||
"л": "л", # Cyrillic 'л'
|
||||
"я": "я", # Cyrillic 'я'
|
||||
"н": "н", # Already mapped, but explicit
|
||||
# Special characters
|
||||
"@": "а",
|
||||
# Multi-character visual bypasses (handled separately in preprocessing)
|
||||
# "}{" visually forms "х" - handled in _preprocess_visual_bypasses
|
||||
}
|
||||
|
||||
_RAW_PHRASE_GROUPS: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
|
||||
("generic", ("айфон", "топ")),
|
||||
("generic", ("самсунг", "говно")),
|
||||
)
|
||||
|
||||
_SENSITIVE_PHRASE_PATH = Path("data/profanity/sensitive_phrases.json")
|
||||
_PHRASE_CACHE: dict[str, Tuple[Tuple[str, ...], ...]] = {}
|
||||
|
||||
|
||||
def _preprocess_visual_bypasses(text: str) -> str:
|
||||
"""
|
||||
Preprocess text to convert multi-character visual bypasses to their intended letters.
|
||||
This handles cases like "}{" visually forming "х".
|
||||
"""
|
||||
result = text
|
||||
# Convert "}{" to "х" (visual bypass for Cyrillic х)
|
||||
# The curly braces visually form the letter х when placed together
|
||||
result = result.replace("}{", "х")
|
||||
return result
|
||||
|
||||
|
||||
def _normalize_char(ch: str) -> str:
|
||||
"""Normalize a single character, mapping homoglyphs to canonical form."""
|
||||
# First try direct mapping (preserves case for non-mapped chars)
|
||||
if ch in _LEET_MAP:
|
||||
return _LEET_MAP[ch]
|
||||
# Then try lowercase mapping
|
||||
lower = ch.lower()
|
||||
if lower in _LEET_MAP:
|
||||
return _LEET_MAP[lower]
|
||||
# If no mapping and character is ASCII letter, return lowercase
|
||||
# This preserves English words like "fromchat" as-is
|
||||
if ch.isascii() and ch.isalpha():
|
||||
return lower
|
||||
# For other characters, return lowercase for consistency
|
||||
return lower
|
||||
|
||||
|
||||
def _normalize_token(token: str) -> str:
|
||||
"""Normalize a token by mapping all homoglyphs."""
|
||||
return "".join(_normalize_char(ch) for ch in token)
|
||||
|
||||
|
||||
def _normalize_text_for_profanity(text: str) -> str:
|
||||
"""
|
||||
Normalize entire text by mapping homoglyphs to canonical forms.
|
||||
This prevents bypasses like using English 'u' instead of Russian 'у'.
|
||||
"""
|
||||
return "".join(_normalize_char(ch) for ch in text)
|
||||
|
||||
|
||||
def _strip_zero_width_chars(text: str) -> str:
|
||||
"""
|
||||
Remove zero-width characters that could be used to bypass filters.
|
||||
"""
|
||||
# Zero-width space, zero-width non-joiner, zero-width joiner, etc.
|
||||
zero_width_chars = [
|
||||
'\u200B', # Zero-width space
|
||||
'\u200C', # Zero-width non-joiner
|
||||
'\u200D', # Zero-width joiner
|
||||
'\uFEFF', # Zero-width no-break space
|
||||
'\u2060', # Word joiner
|
||||
'\u2061', # Function application
|
||||
'\u2062', # Invisible times
|
||||
'\u2063', # Invisible separator
|
||||
'\u2064', # Invisible plus
|
||||
]
|
||||
result = text
|
||||
for zw_char in zero_width_chars:
|
||||
result = result.replace(zw_char, '')
|
||||
return result
|
||||
|
||||
|
||||
def _extract_alphanumeric_with_mapping(text: str, preserve_spaces: bool = False) -> tuple[str, list[int]]:
|
||||
"""
|
||||
Extract only alphanumeric characters from text and create a mapping
|
||||
from normalized positions to original positions.
|
||||
|
||||
Args:
|
||||
preserve_spaces: If True, preserve spaces in the normalized text (for phrase matching)
|
||||
|
||||
Returns:
|
||||
(normalized_text, position_map) where position_map[i] is the original
|
||||
position of the i-th character in normalized_text
|
||||
"""
|
||||
# First preprocess visual bypasses (like "}{" -> "х")
|
||||
text = _preprocess_visual_bypasses(text)
|
||||
|
||||
# Then normalize Unicode (composed vs decomposed)
|
||||
normalized_unicode = unicodedata.normalize('NFKC', text)
|
||||
|
||||
# For phrase matching, convert zero-width chars to spaces instead of stripping
|
||||
if preserve_spaces:
|
||||
zero_width_chars = ['\u200B', '\u200C', '\u200D', '\uFEFF', '\u2060', '\u2061', '\u2062', '\u2063', '\u2064']
|
||||
for zw_char in zero_width_chars:
|
||||
normalized_unicode = normalized_unicode.replace(zw_char, ' ')
|
||||
else:
|
||||
# Strip zero-width characters
|
||||
normalized_unicode = _strip_zero_width_chars(normalized_unicode)
|
||||
|
||||
normalized = []
|
||||
position_map = []
|
||||
|
||||
for i, ch in enumerate(normalized_unicode):
|
||||
# Check if character is alphanumeric (including Cyrillic)
|
||||
if ch.isalnum():
|
||||
# For phrase matching, preserve ASCII letters as-is (just lowercase)
|
||||
# to allow English words in patterns to match
|
||||
if preserve_spaces and ch.isascii() and ch.isalpha():
|
||||
normalized.append(ch.lower())
|
||||
else:
|
||||
# Normalize this character (homoglyphs, Cyrillic, etc.)
|
||||
normalized.append(_normalize_char(ch))
|
||||
position_map.append(i)
|
||||
elif preserve_spaces:
|
||||
# For phrase matching, treat any whitespace or non-alphanumeric as word separator
|
||||
if ch.isspace() or not ch.isalnum():
|
||||
# Normalize to single space to allow patterns to match
|
||||
if normalized and normalized[-1] != ' ': # Don't add consecutive spaces
|
||||
normalized.append(' ')
|
||||
position_map.append(i)
|
||||
|
||||
return "".join(normalized), position_map
|
||||
|
||||
|
||||
def _check_profanity_substrings(normalized_text: str, profane_words: Set[str]) -> list[tuple[int, int]]:
|
||||
"""
|
||||
Check for profane words as substrings or subsequences in normalized text.
|
||||
This catches cases like "хуй" in "хууй" (with extra characters).
|
||||
Returns list of (start, end) positions where profanity is found.
|
||||
"""
|
||||
spans = []
|
||||
normalized_lower = normalized_text.lower()
|
||||
|
||||
for word in profane_words:
|
||||
word_lower = word.lower()
|
||||
|
||||
# First try exact substring match
|
||||
start = 0
|
||||
while True:
|
||||
pos = normalized_lower.find(word_lower, start)
|
||||
if pos == -1:
|
||||
break
|
||||
spans.append((pos, pos + len(word_lower)))
|
||||
start = pos + 1
|
||||
|
||||
# Also check if profane word appears as a subsequence (allowing extra chars)
|
||||
# This catches cases like "хуй" in "хууй" or "х}{¥€уй" -> "хууй"
|
||||
# Now applies to ALL words, not just length >= 4, to prevent bypasses
|
||||
word_chars = list(word_lower)
|
||||
text_chars = list(normalized_lower)
|
||||
|
||||
# Stricter span limits based on word length to prevent false positives
|
||||
# Shorter words get much stricter limits
|
||||
if len(word_lower) <= 3:
|
||||
max_span_ratio = 1.3 # Very strict for 3-char words (e.g., "хуй")
|
||||
elif len(word_lower) == 4:
|
||||
max_span_ratio = 1.4 # Strict for 4-char words
|
||||
elif len(word_lower) <= 5:
|
||||
max_span_ratio = 1.5 # Moderate for 5-char words
|
||||
else:
|
||||
max_span_ratio = 1.8 # Slightly more lenient for longer words
|
||||
|
||||
# Try to find the word as a subsequence
|
||||
i = 0 # position in text
|
||||
j = 0 # position in word
|
||||
seq_start = None
|
||||
|
||||
while i < len(text_chars) and j < len(word_chars):
|
||||
if text_chars[i] == word_chars[j]:
|
||||
if seq_start is None:
|
||||
seq_start = i
|
||||
j += 1
|
||||
if j == len(word_chars):
|
||||
# Found the word as subsequence
|
||||
seq_end = i + 1
|
||||
# Check if the span is reasonable (not too long)
|
||||
span_length = seq_end - seq_start
|
||||
max_allowed_span = int(len(word_lower) * max_span_ratio)
|
||||
if span_length <= max_allowed_span:
|
||||
# Only add if it's not already covered by exact match
|
||||
if (seq_start, seq_end) not in spans:
|
||||
spans.append((seq_start, seq_end))
|
||||
# Reset to find next occurrence - continue from after the end of this match
|
||||
next_start = seq_start + 1
|
||||
seq_start = None
|
||||
j = 0
|
||||
i = next_start
|
||||
continue
|
||||
i += 1
|
||||
|
||||
return spans
|
||||
|
||||
|
||||
def _check_profanity_in_normalized(normalized_text: str) -> bool:
|
||||
"""
|
||||
Check if normalized text contains profanity.
|
||||
Uses both better_profanity library and substring matching for better detection.
|
||||
|
||||
Returns True if profanity is found.
|
||||
"""
|
||||
if not normalized_text:
|
||||
return False
|
||||
|
||||
# Check normalized text for profanity using better_profanity
|
||||
censored = _profanity.censor(normalized_text, censor_char="\\*")
|
||||
|
||||
# Check if better_profanity found anything
|
||||
if "*" in censored:
|
||||
return True
|
||||
|
||||
# Also check for profane words as substrings (to catch cases like "хуй" in "хууй" or "хуйня")
|
||||
profane_words = _STATIC_TERMS
|
||||
substring_spans = _check_profanity_substrings(normalized_text, profane_words)
|
||||
|
||||
# If we found any substring matches, there's profanity
|
||||
if substring_spans:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def _tokenize_with_spans(text: str) -> List[Tuple[int, int, str]]:
|
||||
tokens: List[Tuple[int, int, str]] = []
|
||||
start: int | None = None
|
||||
buffer: List[str] = []
|
||||
|
||||
for idx, ch in enumerate(text):
|
||||
if ch.isalnum() or ch in {"@", "#", "_"}:
|
||||
if start is None:
|
||||
start = idx
|
||||
buffer.append(ch)
|
||||
else:
|
||||
if buffer and start is not None:
|
||||
token_raw = "".join(buffer)
|
||||
tokens.append((start, idx, _normalize_token(token_raw)))
|
||||
buffer.clear()
|
||||
start = None
|
||||
if buffer and start is not None:
|
||||
token_raw = "".join(buffer)
|
||||
tokens.append((start, len(text), _normalize_token(token_raw)))
|
||||
return tokens
|
||||
|
||||
|
||||
def _edit_distance_limited(a: str, b: str, max_distance: int = 1) -> bool:
|
||||
if a == b:
|
||||
return True
|
||||
if max_distance <= 0:
|
||||
return False
|
||||
if abs(len(a) - len(b)) > max_distance:
|
||||
return False
|
||||
|
||||
previous = list(range(len(b) + 1))
|
||||
for i, ca in enumerate(a, 1):
|
||||
current = [i]
|
||||
best = current[0]
|
||||
for j, cb in enumerate(b, 1):
|
||||
insert_cost = current[j - 1] + 1
|
||||
delete_cost = previous[j] + 1
|
||||
replace_cost = previous[j - 1] + (0 if ca == cb else 1)
|
||||
cost = min(insert_cost, delete_cost, replace_cost)
|
||||
current.append(cost)
|
||||
if cost < best:
|
||||
best = cost
|
||||
if best > max_distance:
|
||||
return False
|
||||
previous = current
|
||||
return previous[-1] <= max_distance
|
||||
|
||||
|
||||
def _load_sensitive_phrases() -> List[Tuple[str, ...]]:
|
||||
if not _SENSITIVE_PHRASE_PATH.exists():
|
||||
return []
|
||||
try:
|
||||
payload = json.loads(_SENSITIVE_PHRASE_PATH.read_text(encoding="utf-8"))
|
||||
phrases: List[Tuple[str, ...]] = []
|
||||
if isinstance(payload, list):
|
||||
for entry in payload:
|
||||
if isinstance(entry, list) and entry:
|
||||
normalized = tuple(str(part).strip() for part in entry if str(part).strip())
|
||||
if normalized:
|
||||
phrases.append(normalized)
|
||||
return phrases
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def _get_phrases(group: str) -> Tuple[Tuple[str, ...], ...]:
|
||||
if group not in _PHRASE_CACHE:
|
||||
base = [phrase for key, phrase in _RAW_PHRASE_GROUPS if key == group]
|
||||
if group == "sensitive":
|
||||
base.extend(_load_sensitive_phrases())
|
||||
_PHRASE_CACHE[group] = tuple(
|
||||
tuple(_normalize_token(part) for part in phrase)
|
||||
for phrase in base
|
||||
)
|
||||
return _PHRASE_CACHE[group]
|
||||
|
||||
|
||||
def _find_fuzzy_phrase_spans(text: str, group: str = "generic") -> List[Tuple[int, int]]:
|
||||
tokens = _tokenize_with_spans(text)
|
||||
if not tokens:
|
||||
return []
|
||||
|
||||
spans: List[Tuple[int, int]] = []
|
||||
normalized_phrases = _get_phrases(group)
|
||||
|
||||
for index in range(len(tokens)):
|
||||
for phrase in normalized_phrases:
|
||||
if index + len(phrase) > len(tokens):
|
||||
continue
|
||||
matches = True
|
||||
for offset, target in enumerate(phrase):
|
||||
token = tokens[index + offset][2]
|
||||
if not _edit_distance_limited(token, target):
|
||||
matches = False
|
||||
break
|
||||
if matches:
|
||||
span_start = tokens[index][0]
|
||||
span_end = tokens[index + len(phrase) - 1][1]
|
||||
spans.append((span_start, span_end))
|
||||
return spans
|
||||
|
||||
_dictionary_lock = RLock()
|
||||
_blocklist_signature: Tuple[str, ...] | None = None
|
||||
_profanity = Profanity()
|
||||
|
||||
|
||||
def _normalize_words(words: Iterable[str]) -> Set[str]:
|
||||
normalized: Set[str] = set()
|
||||
for raw in words:
|
||||
if not raw:
|
||||
continue
|
||||
cleaned = re.sub(r"\s+", " ", str(raw)).strip().lower()
|
||||
if cleaned:
|
||||
normalized.add(cleaned)
|
||||
return normalized
|
||||
|
||||
|
||||
def _load_blocklist() -> Set[str]:
|
||||
if not BLOCKLIST_PATH.exists():
|
||||
return set()
|
||||
try:
|
||||
data = json.loads(BLOCKLIST_PATH.read_text(encoding="utf-8"))
|
||||
if isinstance(data, list):
|
||||
return _normalize_words(data)
|
||||
except Exception:
|
||||
pass
|
||||
return set()
|
||||
|
||||
|
||||
def _write_blocklist(words: Iterable[str]) -> None:
|
||||
BLOCKLIST_PATH.write_text(
|
||||
json.dumps(sorted(words), ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8"
|
||||
)
|
||||
|
||||
|
||||
def _rebuild_dictionary(force: bool = False) -> None:
|
||||
global _profanity, _blocklist_signature
|
||||
with _dictionary_lock:
|
||||
blocklist_list = sorted(_load_blocklist())
|
||||
signature = tuple(blocklist_list)
|
||||
if not force and _blocklist_signature == signature and _blocklist_signature is not None:
|
||||
return
|
||||
|
||||
profanity = Profanity()
|
||||
profanity.load_censor_words()
|
||||
# Remove whitelisted words from the default word list
|
||||
try:
|
||||
for word in _WHITELIST:
|
||||
profanity.remove_censor_words([word])
|
||||
except AttributeError:
|
||||
# If remove_censor_words doesn't exist, we'll handle it in post-processing
|
||||
pass
|
||||
combined = set(_STATIC_TERMS)
|
||||
combined.update(blocklist_list)
|
||||
# Remove whitelisted words from our custom terms
|
||||
combined -= _WHITELIST
|
||||
if combined:
|
||||
profanity.add_censor_words(list(combined))
|
||||
|
||||
_profanity = profanity
|
||||
_blocklist_signature = signature
|
||||
|
||||
|
||||
def _check_phrase_patterns(text: str) -> bool:
|
||||
"""
|
||||
Check if text matches any phrase patterns.
|
||||
Returns True if any pattern matches.
|
||||
"""
|
||||
# Normalize text for phrase matching (remove special chars but preserve spaces)
|
||||
normalized_text, _ = _extract_alphanumeric_with_mapping(text, preserve_spaces=True)
|
||||
normalized_lower = normalized_text.lower()
|
||||
|
||||
# Check phrase patterns
|
||||
for pattern in _PHRASE_PATTERNS:
|
||||
if pattern.search(normalized_lower):
|
||||
return True
|
||||
|
||||
# Check fuzzy phrase spans
|
||||
if _find_fuzzy_phrase_spans(normalized_lower, "generic"):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def contains_profanity(text: str) -> bool:
|
||||
"""
|
||||
Check if text contains profanity.
|
||||
Returns True if profanity is detected.
|
||||
"""
|
||||
if not text:
|
||||
return False
|
||||
|
||||
_rebuild_dictionary()
|
||||
|
||||
# Check original text patterns first (before normalization) to catch visual bypasses
|
||||
# like "}{" used to form "х"
|
||||
for pattern in _ORIGINAL_TEXT_PATTERNS:
|
||||
if pattern.search(text):
|
||||
return True
|
||||
|
||||
# Check phrase patterns
|
||||
if _check_phrase_patterns(text):
|
||||
return True
|
||||
|
||||
# Normalize text for whitelist matching (to handle special characters)
|
||||
normalized_for_whitelist, _ = _extract_alphanumeric_with_mapping(text)
|
||||
normalized_for_whitelist_lower = normalized_for_whitelist.lower()
|
||||
|
||||
# Check if text contains whitelisted words - if the entire text is a whitelisted word, skip profanity check
|
||||
for whitelist_word in _WHITELIST:
|
||||
normalized_whitelist, _ = _extract_alphanumeric_with_mapping(whitelist_word)
|
||||
normalized_whitelist_lower = normalized_whitelist.lower()
|
||||
|
||||
# Check if the normalized text exactly matches a whitelisted word
|
||||
if normalized_for_whitelist_lower == normalized_whitelist_lower:
|
||||
return False
|
||||
|
||||
# Extract only alphanumeric characters and normalize homoglyphs
|
||||
# This removes special characters, emojis, etc. that could be used to bypass the filter
|
||||
normalized_text, _ = _extract_alphanumeric_with_mapping(text)
|
||||
|
||||
# Check profanity on normalized text (without special characters)
|
||||
return _check_profanity_in_normalized(normalized_text)
|
||||
|
||||
|
||||
def contains_sensitive_phrase(text: str) -> bool:
|
||||
if not text:
|
||||
return False
|
||||
if _find_fuzzy_phrase_spans(text, "sensitive"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def get_blocklist() -> List[str]:
|
||||
with _dictionary_lock:
|
||||
return sorted(_load_blocklist())
|
||||
|
||||
|
||||
def add_to_blocklist(words: Iterable[str]) -> Tuple[List[str], List[str]]:
|
||||
normalized = _normalize_words(words)
|
||||
if not normalized:
|
||||
return [], get_blocklist()
|
||||
|
||||
with _dictionary_lock:
|
||||
current = _load_blocklist()
|
||||
added = sorted(normalized - current)
|
||||
if not added:
|
||||
return [], sorted(current)
|
||||
|
||||
updated = sorted(current | normalized)
|
||||
_write_blocklist(updated)
|
||||
_rebuild_dictionary(force=True)
|
||||
return added, updated
|
||||
|
||||
|
||||
def remove_from_blocklist(words: Iterable[str]) -> Tuple[List[str], List[str]]:
|
||||
normalized = _normalize_words(words)
|
||||
if not normalized:
|
||||
return [], get_blocklist()
|
||||
|
||||
with _dictionary_lock:
|
||||
current = _load_blocklist()
|
||||
removed = sorted(word for word in normalized if word in current)
|
||||
if not removed:
|
||||
return [], sorted(current)
|
||||
|
||||
updated = sorted(current - normalized)
|
||||
_write_blocklist(updated)
|
||||
_rebuild_dictionary(force=True)
|
||||
return removed, updated
|
||||
|
||||
Reference in New Issue
Block a user