You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
135 lines
4.2 KiB
135 lines
4.2 KiB
"""
|
|
Comprehensive Arabic & Persian Text Normalization Utility for Search.
|
|
Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification,
|
|
Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers.
|
|
"""
|
|
|
|
import re
|
|
import unicodedata
|
|
from typing import Any, Iterable, List, Optional
|
|
|
|
# 1. Arabic Harakat / Diacritics / Quranic annotations:
|
|
# \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc.
|
|
# \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن
|
|
# \u06D6 - \u06ED : Quranic pause marks, small high signs, etc.
|
|
ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]')
|
|
|
|
# 2. Tatweel / Kashida (ـ)
|
|
TATWEEL_REGEX = re.compile(r'\u0640')
|
|
|
|
# 3. HTML tags remover (e.g. <p>, <br>, <span>, etc.)
|
|
HTML_TAGS_REGEX = re.compile(r'<[^>]+>')
|
|
|
|
# 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.)
|
|
ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]')
|
|
|
|
# 5. Normalization translation map
|
|
ARABIC_NORMALIZATION_MAP = str.maketrans({
|
|
# Alef variations -> Plain Alef
|
|
'أ': 'ا',
|
|
'إ': 'ا',
|
|
'آ': 'ا',
|
|
'ٱ': 'ا',
|
|
|
|
# Yeh & Alef Maksura -> Standard Yeh
|
|
'ي': 'ی',
|
|
'ى': 'ی',
|
|
'ئ': 'ی',
|
|
|
|
# Kaf -> Standard Kaf
|
|
'ك': 'ک',
|
|
|
|
# Teh Marbuta & Heh with Yeh -> Standard Heh
|
|
'ة': 'ه',
|
|
'ۀ': 'ه',
|
|
|
|
# Waw with Hamza -> Standard Waw
|
|
'ؤ': 'و',
|
|
|
|
# Eastern Arabic and Persian digits -> Standard Latin digits
|
|
'٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4',
|
|
'٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9',
|
|
'۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4',
|
|
'۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9',
|
|
})
|
|
|
|
|
|
def normalize_for_search(text: Optional[str]) -> str:
|
|
"""
|
|
Normalizes a text string for fuzzy / invariant Arabic and Persian search:
|
|
- Strips HTML tags
|
|
- Unicode NFKC normalization
|
|
- Removes zero-width and invisible control characters
|
|
- Strips all diacritics / tashkeel and dagger alef (\u0670)
|
|
- Strips tatweel / kashida (\u0640)
|
|
- Unifies all Alef forms (أ, إ, آ, ٱ -> ا)
|
|
- Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی)
|
|
- Unifies Kaf (ك -> ک)
|
|
- Unifies Teh Marbuta (ة, ۀ -> ه)
|
|
- Unifies Waw with Hamza (ؤ -> و)
|
|
- Converts Arabic & Persian numerals to ASCII (0-9)
|
|
- Collapses multiple whitespace characters and lowercases English text
|
|
"""
|
|
if not text or not isinstance(text, str):
|
|
return ""
|
|
|
|
# 1. Remove HTML tags if present
|
|
cleaned = HTML_TAGS_REGEX.sub(' ', text)
|
|
|
|
# 2. Unicode normalization (NFKC)
|
|
cleaned = unicodedata.normalize('NFKC', cleaned)
|
|
|
|
# 3. Remove zero-width / control characters
|
|
cleaned = ZERO_WIDTH_REGEX.sub('', cleaned)
|
|
|
|
# 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef
|
|
cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned)
|
|
|
|
# 5. Remove Tatweel / Kashida
|
|
cleaned = TATWEEL_REGEX.sub('', cleaned)
|
|
|
|
# 6. Apply character unification map
|
|
cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP)
|
|
|
|
# 7. Lowercase English characters and collapse whitespace
|
|
cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower()
|
|
|
|
return cleaned
|
|
|
|
|
|
def extract_searchable_strings(data: Any) -> List[str]:
|
|
"""
|
|
Recursively extracts all text values from nested lists, dicts,
|
|
or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]).
|
|
"""
|
|
results: List[str] = []
|
|
|
|
if data is None:
|
|
return results
|
|
|
|
if isinstance(data, str):
|
|
val = data.strip()
|
|
if val:
|
|
results.append(val)
|
|
elif isinstance(data, dict):
|
|
for k, v in data.items():
|
|
results.extend(extract_searchable_strings(v))
|
|
elif isinstance(data, (list, tuple, set)):
|
|
for item in data:
|
|
results.extend(extract_searchable_strings(item))
|
|
|
|
return results
|
|
|
|
|
|
def build_search_blob(*components: Any) -> str:
|
|
"""
|
|
Extracts all text components, normalizes them, and joins them with spaces
|
|
into a single indexed search text blob.
|
|
"""
|
|
all_raw_strings: List[str] = []
|
|
for c in components:
|
|
all_raw_strings.extend(extract_searchable_strings(c))
|
|
|
|
# Join and normalize in one pass
|
|
joined = " ".join(all_raw_strings)
|
|
return normalize_for_search(joined)
|