You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 
 
 
 

135 lines
4.2 KiB

"""
Comprehensive Arabic & Persian Text Normalization Utility for Search.
Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification,
Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers.
"""
import re
import unicodedata
from typing import Any, Iterable, List, Optional
# 1. Arabic Harakat / Diacritics / Quranic annotations:
# \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc.
# \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن
# \u06D6 - \u06ED : Quranic pause marks, small high signs, etc.
ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]')
# 2. Tatweel / Kashida (ـ)
TATWEEL_REGEX = re.compile(r'\u0640')
# 3. HTML tags remover (e.g. <p>, <br>, <span>, etc.)
HTML_TAGS_REGEX = re.compile(r'<[^>]+>')
# 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.)
ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]')
# 5. Normalization translation map
ARABIC_NORMALIZATION_MAP = str.maketrans({
# Alef variations -> Plain Alef
'أ': 'ا',
'إ': 'ا',
'آ': 'ا',
'ٱ': 'ا',
# Yeh & Alef Maksura -> Standard Yeh
'ي': 'ی',
'ى': 'ی',
'ئ': 'ی',
# Kaf -> Standard Kaf
'ك': 'ک',
# Teh Marbuta & Heh with Yeh -> Standard Heh
'ة': 'ه',
'ۀ': 'ه',
# Waw with Hamza -> Standard Waw
'ؤ': 'و',
# Eastern Arabic and Persian digits -> Standard Latin digits
'٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4',
'٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9',
'۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4',
'۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9',
})
def normalize_for_search(text: Optional[str]) -> str:
"""
Normalizes a text string for fuzzy / invariant Arabic and Persian search:
- Strips HTML tags
- Unicode NFKC normalization
- Removes zero-width and invisible control characters
- Strips all diacritics / tashkeel and dagger alef (\u0670)
- Strips tatweel / kashida (\u0640)
- Unifies all Alef forms (أ, إ, آ, ٱ -> ا)
- Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی)
- Unifies Kaf (ك -> ک)
- Unifies Teh Marbuta (ة, ۀ -> ه)
- Unifies Waw with Hamza (ؤ -> و)
- Converts Arabic & Persian numerals to ASCII (0-9)
- Collapses multiple whitespace characters and lowercases English text
"""
if not text or not isinstance(text, str):
return ""
# 1. Remove HTML tags if present
cleaned = HTML_TAGS_REGEX.sub(' ', text)
# 2. Unicode normalization (NFKC)
cleaned = unicodedata.normalize('NFKC', cleaned)
# 3. Remove zero-width / control characters
cleaned = ZERO_WIDTH_REGEX.sub('', cleaned)
# 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef
cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned)
# 5. Remove Tatweel / Kashida
cleaned = TATWEEL_REGEX.sub('', cleaned)
# 6. Apply character unification map
cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP)
# 7. Lowercase English characters and collapse whitespace
cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower()
return cleaned
def extract_searchable_strings(data: Any) -> List[str]:
"""
Recursively extracts all text values from nested lists, dicts,
or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]).
"""
results: List[str] = []
if data is None:
return results
if isinstance(data, str):
val = data.strip()
if val:
results.append(val)
elif isinstance(data, dict):
for k, v in data.items():
results.extend(extract_searchable_strings(v))
elif isinstance(data, (list, tuple, set)):
for item in data:
results.extend(extract_searchable_strings(item))
return results
def build_search_blob(*components: Any) -> str:
"""
Extracts all text components, normalizes them, and joins them with spaces
into a single indexed search text blob.
"""
all_raw_strings: List[str] = []
for c in components:
all_raw_strings.extend(extract_searchable_strings(c))
# Join and normalize in one pass
joined = " ".join(all_raw_strings)
return normalize_for_search(joined)