""" Comprehensive Arabic & Persian Text Normalization Utility for Search. Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification, Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers. """ import re import unicodedata from typing import Any, Iterable, List, Optional # 1. Arabic Harakat / Diacritics / Quranic annotations: # \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc. # \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن # \u06D6 - \u06ED : Quranic pause marks, small high signs, etc. ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]') # 2. Tatweel / Kashida (ـ) TATWEEL_REGEX = re.compile(r'\u0640') # 3. HTML tags remover (e.g.

,
, , etc.) HTML_TAGS_REGEX = re.compile(r'<[^>]+>') # 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.) ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]') # 5. Normalization translation map ARABIC_NORMALIZATION_MAP = str.maketrans({ # Alef variations -> Plain Alef 'أ': 'ا', 'إ': 'ا', 'آ': 'ا', 'ٱ': 'ا', # Yeh & Alef Maksura -> Standard Yeh 'ي': 'ی', 'ى': 'ی', 'ئ': 'ی', # Kaf -> Standard Kaf 'ك': 'ک', # Teh Marbuta & Heh with Yeh -> Standard Heh 'ة': 'ه', 'ۀ': 'ه', # Waw with Hamza -> Standard Waw 'ؤ': 'و', # Eastern Arabic and Persian digits -> Standard Latin digits '٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4', '٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9', '۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4', '۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9', }) def normalize_for_search(text: Optional[str]) -> str: """ Normalizes a text string for fuzzy / invariant Arabic and Persian search: - Strips HTML tags - Unicode NFKC normalization - Removes zero-width and invisible control characters - Strips all diacritics / tashkeel and dagger alef (\u0670) - Strips tatweel / kashida (\u0640) - Unifies all Alef forms (أ, إ, آ, ٱ -> ا) - Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی) - Unifies Kaf (ك -> ک) - Unifies Teh Marbuta (ة, ۀ -> ه) - Unifies Waw with Hamza (ؤ -> و) - Converts Arabic & Persian numerals to ASCII (0-9) - Collapses multiple whitespace characters and lowercases English text """ if not text or not isinstance(text, str): return "" # 1. Remove HTML tags if present cleaned = HTML_TAGS_REGEX.sub(' ', text) # 2. Unicode normalization (NFKC) cleaned = unicodedata.normalize('NFKC', cleaned) # 3. Remove zero-width / control characters cleaned = ZERO_WIDTH_REGEX.sub('', cleaned) # 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned) # 5. Remove Tatweel / Kashida cleaned = TATWEEL_REGEX.sub('', cleaned) # 6. Apply character unification map cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP) # 7. Lowercase English characters and collapse whitespace cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower() return cleaned def extract_searchable_strings(data: Any) -> List[str]: """ Recursively extracts all text values from nested lists, dicts, or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]). """ results: List[str] = [] if data is None: return results if isinstance(data, str): val = data.strip() if val: results.append(val) elif isinstance(data, dict): for k, v in data.items(): results.extend(extract_searchable_strings(v)) elif isinstance(data, (list, tuple, set)): for item in data: results.extend(extract_searchable_strings(item)) return results def build_search_blob(*components: Any) -> str: """ Extracts all text components, normalizes them, and joins them with spaces into a single indexed search text blob. """ all_raw_strings: List[str] = [] for c in components: all_raw_strings.extend(extract_searchable_strings(c)) # Join and normalize in one pass joined = " ".join(all_raw_strings) return normalize_for_search(joined)