From 44c2f59328879cca0a4412dfe70f6a4e93191f49 Mon Sep 17 00:00:00 2001 From: mohsentaba Date: Thu, 8 Oct 2026 10:51:45 +0330 Subject: [PATCH] feat(search): implement Arabic and Persian search normalization with shadow search column - Add utils/text_normalizer.py providing robust Arabic and Persian orthographic normalization (diacritics/tashkeel removal, unification of Alef variants, Yeh/Alef Maksura, Kaf, Teh Marbuta, Waw Hamza, Tatweel, and digit normalization). - Implement recursive text extraction from multilingual JSON structures and models. - Add normalized_text shadow column across Hadis, HadisCorrection, HadisInterpretation, Transmitters, BookReference, and HadisCategory models with automated save-time synchronization. - Apply database migrations (0048 and 0049). - Add populate_normalized_search management command for backfilling existing records. - Update search querysets in hadis, transmitter, book reference, and category views to filter using normalized text queries. - Add comprehensive documentation in docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md. --- .../commands/populate_normalized_search.py | 95 ++++++ ...zed_text_hadis_normalized_text_and_more.py | 43 +++ ..._bookreference_normalized_text_and_more.py | 43 +++ apps/hadis/models/category.py | 6 + apps/hadis/models/hadis.py | 30 ++ apps/hadis/models/reference.py | 11 + apps/hadis/models/transmitter.py | 10 + apps/hadis/views/category.py | 6 + apps/hadis/views/hadis.py | 28 +- apps/hadis/views/reference.py | 12 +- apps/hadis/views/reference_v2.py | 3 + apps/hadis/views/transmitter.py | 9 + docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md | 283 ++++++++++++++++++ utils/text_normalizer.py | 135 +++++++++ 14 files changed, 700 insertions(+), 14 deletions(-) create mode 100644 apps/hadis/management/commands/populate_normalized_search.py create mode 100644 apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py create mode 100644 apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py create mode 100644 docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md create mode 100644 utils/text_normalizer.py diff --git a/apps/hadis/management/commands/populate_normalized_search.py b/apps/hadis/management/commands/populate_normalized_search.py new file mode 100644 index 0000000..d8b48a4 --- /dev/null +++ b/apps/hadis/management/commands/populate_normalized_search.py @@ -0,0 +1,95 @@ +from django.core.management.base import BaseCommand +from apps.hadis.models import ( + Hadis, + Transmitters, + BookReference, + HadisCategory, + HadisCorrection, + HadisInterpretation +) +from utils.text_normalizer import build_search_blob + + +class Command(BaseCommand): + help = "Backfill normalized_text on Hadis, Transmitters, Categories, BookReferences, Corrections, and Interpretations" + + def handle(self, *args, **options): + self.stdout.write(self.style.SUCCESS("Starting search normalization backfill...")) + + # 1. Hadis + self.stdout.write("1. Normalizing Hadis records...") + hadiths = list(Hadis.objects.all().only('id', 'text', 'title', 'title_narrator', 'translation', 'normalized_text')) + for h in hadiths: + h.normalized_text = build_search_blob( + h.text, + h.title, + h.title_narrator, + h.translation + ) + Hadis.objects.bulk_update(hadiths, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(hadiths)} Hadis records.")) + + # 2. Transmitters + self.stdout.write("2. Normalizing Transmitters records...") + transmitters = list(Transmitters.objects.all().only('id', 'full_name', 'kunya', 'known_as', 'nickname', 'normalized_text')) + for t in transmitters: + t.normalized_text = build_search_blob( + t.full_name, + t.kunya, + t.known_as, + t.nickname + ) + Transmitters.objects.bulk_update(transmitters, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(transmitters)} Transmitters records.")) + + # 3. BookReference + self.stdout.write("3. Normalizing BookReference records...") + books = list(BookReference.objects.select_related('author').all().only('id', 'title', 'description', 'author__name', 'normalized_text')) + for b in books: + author_name = b.author.name if b.author else "" + b.normalized_text = build_search_blob( + b.title, + b.description, + author_name + ) + BookReference.objects.bulk_update(books, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(books)} BookReference records.")) + + # 4. HadisCategory + self.stdout.write("4. Normalizing HadisCategory records...") + categories = list(HadisCategory.objects.all().only('id', 'title', 'description', 'normalized_text')) + for c in categories: + c.normalized_text = build_search_blob( + c.title, + c.description + ) + HadisCategory.objects.bulk_update(categories, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(categories)} HadisCategory records.")) + + # 5. HadisCorrection + self.stdout.write("5. Normalizing HadisCorrection records...") + corrections = list(HadisCorrection.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text')) + for corr in corrections: + corr.normalized_text = build_search_blob( + corr.text, + corr.title, + corr.narrator, + corr.translation + ) + HadisCorrection.objects.bulk_update(corrections, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(corrections)} HadisCorrection records.")) + + # 6. HadisInterpretation + self.stdout.write("6. Normalizing HadisInterpretation records...") + interpretations = list(HadisInterpretation.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text')) + for interp in interpretations: + interp.normalized_text = build_search_blob( + interp.text, + interp.title, + interp.narrator, + interp.translation + ) + HadisInterpretation.objects.bulk_update(interpretations, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(interpretations)} HadisInterpretation records.")) + + self.stdout.write(self.style.SUCCESS("All search normalization fields backfilled successfully!")) diff --git a/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py b/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py new file mode 100644 index 0000000..b7792a0 --- /dev/null +++ b/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py @@ -0,0 +1,43 @@ +# Generated by Django 4.2.30 on 2026-10-08 10:37 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('hadis', '0047_transmitteroriginaltext_text_to_textfield'), + ] + + operations = [ + migrations.AddField( + model_name='bookreference', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadis', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadiscategory', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadiscorrection', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadisinterpretation', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='transmitters', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + ] diff --git a/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py b/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py new file mode 100644 index 0000000..52ab625 --- /dev/null +++ b/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py @@ -0,0 +1,43 @@ +# Generated by Django 4.2.30 on 2026-10-08 10:40 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('hadis', '0048_bookreference_normalized_text_hadis_normalized_text_and_more'), + ] + + operations = [ + migrations.AlterField( + model_name='bookreference', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadis', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadiscategory', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadiscorrection', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadisinterpretation', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='transmitters', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + ] diff --git a/apps/hadis/models/category.py b/apps/hadis/models/category.py index 03c3739..cb84941 100644 --- a/apps/hadis/models/category.py +++ b/apps/hadis/models/category.py @@ -80,6 +80,7 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel): order = models.IntegerField(default=0, verbose_name=_('order')) xmind_file = models.FileField(upload_to='hadis/xmind_files/', verbose_name=_('xmind file'), null=True, blank=True) slug = models.SlugField(max_length=255, null=True, blank=True,allow_unicode=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) content_type = None language = None language_id = None @@ -111,6 +112,11 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel): slug_source_field = 'title' def save(self, *args, **kwargs): + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.title, + self.description + ) self.full_clean() super().save(*args, **kwargs) diff --git a/apps/hadis/models/hadis.py b/apps/hadis/models/hadis.py index 7407bc2..f3f1e6e 100644 --- a/apps/hadis/models/hadis.py +++ b/apps/hadis/models/hadis.py @@ -182,6 +182,7 @@ class Hadis(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) embedded_in = models.JSONField(default=list, blank=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) def __str__(self): title = None @@ -231,6 +232,15 @@ class Hadis(LowercaseSlugMixin, models.Model): if (old_instance.text != self.text or old_instance.translation != self.translation): self.embedded_in = [] # Reset! + + # Populate normalized_text for fast and accurate Arabic/Persian search + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.title_narrator, + self.translation + ) super().save(*args, **kwargs) @@ -417,6 +427,7 @@ class HadisCorrection(models.Model): updated_at = models.DateTimeField(auto_now=True, verbose_name=_("updated at")) share_link = models.CharField(max_length=255, verbose_name=_('share link'), null=True, blank=True) embedded_in = models.JSONField(default=list, blank=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: verbose_name = _("Hadis Correction") @@ -442,6 +453,15 @@ class HadisCorrection(models.Model): old_instance.translation != self.translation): self.embedded_in = [] # Reset! + # Populate normalized_text for fast search + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.narrator, + self.translation + ) + super().save(*args, **kwargs) def get_title(self, lang): @@ -474,6 +494,7 @@ class HadisInterpretation(models.Model): text = models.TextField(verbose_name=_('Original Text')) translation = models.JSONField(verbose_name=_('Translation'), default=list) priority = models.IntegerField(default=0, verbose_name=_('priority')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) created_at = models.DateTimeField(auto_now_add=True) updated_at = models.DateTimeField(auto_now=True) @@ -486,6 +507,15 @@ class HadisInterpretation(models.Model): # 👈 ست کردن اسلاگ تفسیر برابر با اسلاگ کتگوری if self.category and self.category.slug: self.slug = self.category.slug + + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.narrator, + self.translation + ) + super().save(*args, **kwargs) # --- FOR CORRECTIONS --- diff --git a/apps/hadis/models/reference.py b/apps/hadis/models/reference.py index 55d5ec7..c475d76 100644 --- a/apps/hadis/models/reference.py +++ b/apps/hadis/models/reference.py @@ -143,6 +143,7 @@ class BookReference(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: indexes = [ @@ -157,6 +158,16 @@ class BookReference(LowercaseSlugMixin, models.Model): return self.title[0].get('text') or self.title[0].get('title') or f"Reference {self.id}" return f"Reference {self.id}" + def save(self, *args, **kwargs): + from utils.text_normalizer import build_search_blob + author_name = self.author.name if self.author else "" + self.normalized_text = build_search_blob( + self.title, + self.description, + author_name + ) + super().save(*args, **kwargs) + def _get_json_field(self, field_name: str, lang: Optional[str]=None , fallback: str = "en"): """ Generic getter for JSONField in our [{text, language_code}] format. diff --git a/apps/hadis/models/transmitter.py b/apps/hadis/models/transmitter.py index 49a9f75..9d2d6fa 100644 --- a/apps/hadis/models/transmitter.py +++ b/apps/hadis/models/transmitter.py @@ -207,6 +207,7 @@ class Transmitters(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: indexes = [ @@ -222,6 +223,15 @@ class Transmitters(LowercaseSlugMixin, models.Model): self.legacy_id = None if not self.legacy_number: self.legacy_number = None + + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.full_name, + self.kunya, + self.known_as, + self.nickname + ) + super().save(*args, **kwargs) @property diff --git a/apps/hadis/views/category.py b/apps/hadis/views/category.py index b72a168..0a0095a 100644 --- a/apps/hadis/views/category.py +++ b/apps/hadis/views/category.py @@ -477,7 +477,10 @@ class CategoriesView(ListAPIView): ) search_query = (self.request.query_params.get('search') or '').strip() if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(slug__icontains=search_query) | Q(description__icontains=search_query) @@ -514,7 +517,10 @@ class CategoriesBySectView(ListAPIView): ) search_query = (self.request.query_params.get('search') or '').strip() if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(slug__icontains=search_query) | Q(description__icontains=search_query) diff --git a/apps/hadis/views/hadis.py b/apps/hadis/views/hadis.py index f19d920..9ecaea5 100644 --- a/apps/hadis/views/hadis.py +++ b/apps/hadis/views/hadis.py @@ -361,12 +361,18 @@ class HadisListView(ListAPIView): # 👇 1. Apply Search Filter search_query = self.request.query_params.get('search', None) if search_query: - search_conditions = Q(text__icontains=search_query) + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) + + search_conditions = Q(normalized_text__icontains=norm_q) + search_conditions |= Q(text__icontains=search_query) search_conditions |= Q(title__icontains=search_query) search_conditions |= Q(title_narrator__icontains=search_query) search_conditions |= Q(translation__icontains=search_query) search_conditions |= Q(tags__title__icontains=search_query) + search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q) search_conditions |= Q(references__book_reference__title__icontains=search_query) + search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q) search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query) search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query) queryset = queryset.filter(search_conditions) @@ -625,21 +631,25 @@ class HadisMainListView(ListAPIView): def apply_search_filter(self, queryset, search_query): """ - Apply search filter across multiple fields including JSONFields. - Searches in: title, title_narrator, text, translation, tags, book references, transmitters + Apply search filter across multiple fields including normalized text and JSONFields. + Searches in: normalized_text, title, title_narrator, text, translation, tags, book references, transmitters """ from django.db.models import Q + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) - # Basic search conditions - search_conditions = Q(text__icontains=search_query) + # Primary normalized search (matches Arabic/Persian/English invariantly) + search_conditions = Q(normalized_text__icontains=norm_q) - # For JSONFields, search in the JSON string representation - # This will find matches in the "text" values within the JSON arrays + # Fallback raw search + search_conditions |= Q(text__icontains=search_query) search_conditions |= Q(title__icontains=search_query) search_conditions |= Q(title_narrator__icontains=search_query) search_conditions |= Q(translation__icontains=search_query) search_conditions |= Q(tags__title__icontains=search_query) + search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q) search_conditions |= Q(references__book_reference__title__icontains=search_query) + search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q) search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query) search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query) @@ -919,7 +929,11 @@ class CategoryHadisCorrectionsView(ListAPIView): # Optional search query filter search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) search_conditions = ( + Q(normalized_text__icontains=norm_q) | + Q(hadis__normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(text__icontains=search_query) | Q(narrator__icontains=search_query) | diff --git a/apps/hadis/views/reference.py b/apps/hadis/views/reference.py index 11b4e1e..3a36c96 100644 --- a/apps/hadis/views/reference.py +++ b/apps/hadis/views/reference.py @@ -36,15 +36,13 @@ class BookReferencesView(ListAPIView): def apply_search_filter(self, queryset, search_query): """ - Apply search filter across book titles (JSONField). - Searches in: title + Apply search filter across book titles (JSONField) and normalized text. + Searches in: normalized_text, title """ - # Search conditions - search_conditions = Q() + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) - # For JSONFields, search in the JSON string representation - # This will find matches in the "text" values within the JSON arrays - search_conditions |= Q(title__icontains=search_query) + search_conditions = Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) return queryset.filter(search_conditions) diff --git a/apps/hadis/views/reference_v2.py b/apps/hadis/views/reference_v2.py index d2abb24..d3b2157 100644 --- a/apps/hadis/views/reference_v2.py +++ b/apps/hadis/views/reference_v2.py @@ -163,7 +163,10 @@ class BookReferenceV2ListView(generics.ListAPIView): search = self.request.query_params.get('search') if search: from django.db.models import Q + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search) qs = qs.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search) | Q(description__icontains=search) | Q(author__name__icontains=search) diff --git a/apps/hadis/views/transmitter.py b/apps/hadis/views/transmitter.py index e8a8109..93f848b 100644 --- a/apps/hadis/views/transmitter.py +++ b/apps/hadis/views/transmitter.py @@ -42,7 +42,10 @@ class TransmitterView(ListAPIView): # 1. Apply search filter (Searching across name-related JSONFields, slug, legacy_id) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) q_filter = ( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | @@ -421,7 +424,10 @@ class NarratorTeachersView(ListAPIView): search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | @@ -471,7 +477,10 @@ class NarratorStudentsView(ListAPIView): search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | diff --git a/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md b/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md new file mode 100644 index 0000000..540011b --- /dev/null +++ b/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md @@ -0,0 +1,283 @@ +# دستورالعمل جامع نرمال‌سازی جستجو برای متون عربی و اسلامی +### (Comprehensive Arabic & Persian Search Normalization Specification) + +--- + +## ۱. مقدمه و بیان مسئله (Executive Summary & Problem Statement) + +در پایگاه‌های داده و سامانه‌های متنی اسلامی و حدیثی، جستجوی متنی با چالش‌های بنیادین زبان‌شناختی و فنی مواجه است: + +1. **گوناگونی رسم‌الخط و کدگذاری یونیکد (Unicode Homoglyphs):** نویسه‌هایی مانند الف همزه‌دار (`أ`، `إ`)، الف ممدوده (`آ`)، الف وصل (`ٱ`) و الف ساده (`ا`) دارای کدهای یونیکد کاملاً متفاوتی هستند. +2. **اعراب و اعجام (Harakat / Tashkeel):** متون احادیث و اسناد دینی معمولاً با حرکت‌گذاری کامل (مانند `إِنَّمَا الأَعْمَالُ`) در پایگاه داده ذخیره شده‌اند، در حالی که کاربران جستجوهای خود را بدون اعراب (`انما الاعمال` یا `انما`) تایپ می‌کنند. در مقایسه‌های متنی پیش‌فرض دیتابیس (`LIKE` یا `ILIKE` و `icontains`)، وجود اعراب در میان حروف کلمه باعث می‌شود کلمه کاربر تطبیق داده نشود و نتیجه «هیچ رکوردی یافت نشد» برگردد. +3. **تداخل کاراکترهای فارسی و عربی:** در کیبوردهای کاربران (موبایل و دسکتاپ)، حروفی چون «ی» و «ي»، «ک» و «ك»، «ه» و «ة» مدام جابه‌جا می‌شوند. +4. **علائم قرآنی و وقوف:** نمادهایی مانند «الف خنجری» (`ٰ` مانند `رَحْمٰنِ`)، علائم وقف و نمادهای تعظیم اگر نرمال نشوند، جستجو را مسدود می‌کنند. + +> **هدف:** هر متنی که کاربر با هر شکلی از کیبورد (با اعراب، بدون اعراب، با همزه یا بدون همزه، فارسی یا عربی) جستجو کرد، سیستم باید دقیقاً مفهوم آن را درک کرده و تمامی رکوردهای منطبق را بدون وابستگی به شکل ظاهری نگارش پیدا کند. + +--- + +## ۲. دسته‌بندی جامع کاراکترها و حالات نرمال‌سازی (Normalization Taxonomy) + +### ۲.۱. خانواده الف‌ها و همزه‌ها (Alef & Hamza Variants) +الف در رسم‌الخط عربی دارای حالت‌های متعددی است که در جستجو باید همگی یکپارچه شوند: + +| نویسه اصلی | نام یونیکد | کد یونیکد | هدف نرمال‌سازی | مثال جستجو | مثال در دیتابیس | +| :--- | :--- | :--- | :--- | :--- | :--- | +| **أ** | Arabic Letter Alef With Hamza Above | `U+0623` | **ا** (`U+0627`) | أنما | انما / إِنَّمَا | +| **إ** | Arabic Letter Alef With Hamza Below | `U+0625` | **ا** (`U+0627`) | إسناد | اسناد | +| **آ** | Arabic Letter Alef With Madda Above | `U+0622` | **ا** (`U+0627`) | آثار | اثار | +| **ٱ** | Arabic Letter Alef Wasla | `U+0671` | **ا** (`U+0627`) | ٱمرؤ | امرؤ | +| **ا** | Arabic Letter Alef (Bare) | `U+0627` | **ا** (`U+0627`) | اعمال | أعمال | +| **ٴ** | Arabic Letter High Hamza | `U+0674` | *حذف* یا **ا** | — | — | + +#### حالات دیگر همزه‌ها (Other Hamza Forms): +- **همزه روی واو (`ؤ` - `U+0624`):** در جستجوی آسان‌گیر (Lenient)، همزه روی واو به `و` (`U+0648`) نرمال می‌شود تا کاربر با سرچ «مومن» بتواند «مؤمن» را پیدا کند یا بالعکس («مسؤول» / «مسئول»). +- **همزه روی یاء / نبره (`ئ` - `U+0626`):** به `ی` / `ي` تبدیل می‌شود تا کلماتی چون «قائل»، «قایل»، «هیئة»، «هيئة» هم‌ارز شوند. +- **همزه تنها روی خط (`ء` - `U+0621`):** حذف یا تبدیل به فاصله در صورت نیاز. + +--- + +### ۲.۲. خانواده یاء و الف مقصوره (Yeh & Alef Maksura) +یکی از پرتکرارترین خطاها در جستجوی عربی و فارسی مربوط به حرف «ی» است: + +| نویسه اصلی | نام | کد یونیکد | رفتار نرمال‌سازی | مثال | +| :--- | :--- | :--- | :--- | :--- | +| **ي** | یاء عربی دو نقطه | `U+064A` | تبدیل به نویسه واحد مبنا (مثلاً `ی` یا `ي`) | علي / علی | +| **ى** | الف مقصوره عربی (بی‌نقطه) | `U+0649` | تبدیل به `ی` یا `ا` بر اساس سیاست پروژه | موسی / موسي / موسى | +| **ی** | یای فارسی بدون نقطه | `U+06CC` | تبدیل به نویسه واحد مبنا | حدیث / حديث | +| **ئ** | یاء با همزه | `U+0626` | تبدیل به نویسه واحد مبنا | بئر / بیر | + +> **نکته تخصصی در متون دینی:** الف مقصوره (`ى` در انتهای کلماتی چون «حتى»، «إلى»، «موسى») در کیبورد کاربران گاهی با `ی`، گاهی با `ي` و حتی گاهی به اشتباه با `ا` («حتا») نوشته می‌شود. نرمال‌سازی `[ي ى ی ئ]` به یک نویسه پایدار، پوشش جستجو را به ۱۰۰٪ می‌رساند. + +--- + +### ۲.۳. کاف عربی و فارسی (Kaf Normalization) +- **ك** (کاف عربی با نشان همزه/کاف کوچک: `U+0643`) +- **ک** (کاف فارسی سرکش‌دار: `U+06A9`) +- **قاعده:** تبدیل هر دو به یک فرم استاندارد (مثلاً `ك` برای متون عربی یا `ک`). + +--- + +### ۲.۴. تاء مربوطه و هاء (Teh Marbuta & Heh) +- **ة** (تاء مربوطه: `U+0629`) +- **ه** (هاء: `U+0647`) +- **ۀ** (هاء با همزه: `U+06C0`) +- **تحلیل رفتاری:** کاربران در تایپ سریع اسامی یا اصطلاحات اغلب تاء مربوطه را با هاء جابه‌جا می‌زنند: + - «معاویه» ↔ «معاوية» + - «فاطمه» ↔ «فاطمة» + - «صحابه» ↔ «صحابة» + - «رواة» ↔ «رواه» +- **قاعده سرچ نرمال:** برای جستجوی متنی، `ة` به `ه` تبدیل می‌شود تا تفاوت نگارشی کاربر باعث حذف رکورد نشود. + +--- + +### ۲.۵. حرکات، اعراب و تنوین‌ها (Tashkeel / Harakat / Diacritics) - حیاتی‌ترین بخش +تمام حرکات زیر باید از متن ورودی و از نسخه ایندکس‌شده جستجو **کاملاً حذف شوند**: + +| نام حرکت | علامت | کد یونیکد | اثر در جستجوی خام دیتابیس | +| :--- | :---: | :---: | :--- | +| **فتحه (Fatha)** | َ | `U+064E` | مانع تطابق متن ساده می‌شود | +| **ضمه (Damma)** | ُ | `U+064F` | مانع تطابق متن ساده می‌شود | +| **کسره (Kasra)** | ِ | `U+0650` | مانع تطابق متن ساده می‌شود | +| **تنوین نصب (Fathatan)** | ً | `U+064B` | مانع تطابق | +| **تنوین رفع (Dammatan)** | ٌ | `U+064C` | مانع تطابق | +| **تنوین جر (Kasratan)** | ٍ | `U+064D` | مانع تطابق | +| **سکون (Sukun)** | ْ | `U+0652` | مانع تطابق | +| **تشدید (Shadda)** | ّ | `U+0651` | مانع تطابق کلمات دارای تشدید | +| **الف خنجری (Dagger Alef)** | ٰ | `U+0670` | **بسیار خطرناک:** در «رَحْمٰنِ»، «إِلٰهَ»، «هٰذَا»، «إِسْمٰعِيل» اگر حذف نشود، سرچ «رحمن» هرگز «رحمٰن» را پیدا نمی‌کند! | +| **مده (Maddah)** | ٓ | `U+0653` | مانع تطابق | +| **همزه فوقانی اعرابی** | ٔ | `U+0654` | مانع تطابق | +| **همزه تحتانی اعرابی** | ٕ | `U+0655` | مانع تطابق | + +--- + +### ۲.۶. کشیدگی، تطویل و نشانه‌های نامرئی (Tatweel & Invisible Characters) +- **تطویل / کشیده (`ـ` - `U+0640`):** برای تنظیم طول خطوط در متون کهن یا زیبایی متنی به کار می‌رود (مثل `صـــــراط` یا `رســــول`). این کاراکتر باید به کلی حذف شود. +- **نیم‌فاصله (Zero-Width Non-Joiner - `U+200C`):** در متون فارسی و اسامی ترکیبی وجود دارد؛ باید به فاصله عادی یا حذف کامل تبدیل شود. +- **اتصال‌دهنده مجازی (ZWJ - `U+200D`):** باید حذف شود. +- **نشانگرهای جهت یونیکد (LRM `U+200E` و RLM `U+200F`):** باید کاملاً حذف شوند. + +--- + +### ۲.۷. نشانه‌ها و نمادهای وقوف قرآنی و مذهبی (Quranic Symbols & Waqf Marks) +در متون روایی و قرآنی نمادهای ویژه‌ای در یونیکد ذخیره می‌شوند که باید در لایه سرچ پالایش گردند: +- علائم وقف قرآنی: `ۖ` (`U+06D6`), `ۗ` (`U+06D7`), `ۘ` (`U+06D8`), `ۙ` (`U+06D9`), `ۚ` (`U+06DA`), `ۛ` (`U+06DB`), `ۜ` (`U+06DC`), `۝` (`U+06DD`), `۞` (`U+06DE`), `۟` (`U+06DF`), `۠` (`U+06E0`), `ۡ` (`U+06E1`), `ۢ` (`U+06E2`), `ۣ` (`U+06E3`), `ۤ` (`U+06E4`). +- نمادهای لیگچر مذهبی: `ﷺ` (`U+FDFA`), `ﷻ` (`U+FDFB`), `﷽` (`U+FDFD`), `ؑ` (`U+0611`). +- پرانتزها و براکت‌های قرآنی و نقل‌قول: `﴿`، `﴾`، `«`، `»`، `[`، `]`، `(`، `)`. + +--- + +### ۲.۸. ارقام و اعداد (Digits) +- ارقام عربی-مشرقی: `[٠, ١, ٢, ٣, ٤, ٥, ٦, ٧, ٨, ٩]` +- ارقام فارسی: `[۰, ۱, ۲, ۳, ۴, ۵, ۶, ۷, ۸, ۹]` +- ارقام استاندارد لاتین: `[0, 1, 2, 3, 4, 5, 6, 7, 8, 9]` +- **قاعده:** تبدیل تمام ارقام به ارقام استاندارد (0-9) تا سرچ شماره حدیث یا جلد و صفحه فارغ از نوع کیبورد عدد را پیدا کند. + +--- + +## ۳. ماتریس مقایسه‌ای سناریوهای سرچ (Search Scenario Matrix) + +| ورودی کاربر در سرچ | متن در دیتابیس | وضعیت جستجوی فعلی (خام) | وضعیت پس از نرمال‌سازی | +| :--- | :--- | :---: | :---: | +| `انما` | `أنما الأعمال بالنيات` | ❌ No result (تفاوت `ا` و `أ`) | ✅ منطبق و پیدا می‌شود | +| `انما` | `إِنَّمَا الأَعْمَالُ بِالنِّيَّاتِ` | ❌ No result (وجود کسره، تشدید، فتحه) | ✅ منطبق و پیدا می‌شود | +| `أبو هريرة` | `ابو هريره` | ❌ No result (تفاوت `أ/ا` و `ة/ه`) | ✅ منطبق و پیدا می‌شود | +| `صحیح بخاری` | `صَحِيحُ الْبُخَارِيِّ` | ❌ No result (تفاوت اعراب، `ی/ي`) | ✅ منطبق و پیدا می‌شود | +| `رحمن` | `الرَّحْمٰنِ الرَّحِيمِ` | ❌ No result (وجود الف خنجری `ٰ`) | ✅ منطبق و پیدا می‌شود | +| `مومن` | `إِنَّمَا الْمُؤْمِنُونَ إِخْوَةٌ` | ❌ No result (تفاوت `و` با `ؤ`) | ✅ منطبق و پیدا می‌شود | +| `صراط` | `صــــراط الذين` | ❌ No result (وجود کشیده `ـ`) | ✅ منطبق و پیدا می‌شود | +| `حدیث ۱۱۰` | `حديث 110` یا `حديث ١١٠` | ❌ No result (تفاوت ارقام) | ✅ منطبق و پیدا می‌شود | + +--- + +## ۴. گزینه‌ها و معماری پیاده‌سازی فنی در Django و PostgreSQL + +برای اعمال این نرمال‌سازی در سیستم، سه رویکرد معماری وجود دارد: + +### 🟢 گزینه اول: الگوی ستون جستجوی نرمال‌شده (Normalized Shadow Column / Search Column) — **[رویکرد پیشنهادی و استاندارد]** + +در این الگو، متن اصلی برای نمایش دست‌نخورده باقی می‌ماند (تا اعراب و زیبایی اصیل آن در UI حفظ شود)، اما یک ستون متنی نرمال‌شده در کنار آن ایجاد و ایندکس‌گذاری می‌شود: + +1. **در مدل‌ها (`Hadis`، `HadisCategory`، `Transmitter` و ...):** + - افزودن فیلد `normalized_text = models.TextField(blank=True, db_index=True)` یا استفاده از `django.contrib.postgres.search.SearchVector`. + - در متد `save()` مدل، متن اصلی از تابع نرمال‌ساز عبور کرده و فیلد نرمال‌شده به صورت خودکار پر می‌شود. + - ایجاد یک اسکریپت ساده migration برای پر کردن یک‌باره مقادیر رکوردهای موجود. +2. **در لایه Queryset / View:** + - عبارت سرچ کاربر (`search_query`) توسط همان تابع پایتون نرمال‌سازی می‌شود: `normalized_q = normalize_text(search_query)`. + - جستجو روی ستون `normalized_text__icontains=normalized_q` انجام می‌شود. +3. **مزایا:** + - **فوق‌العاده سریع (High Performance):** دیتابیس مستقیماً روی ستون ایندکس‌شده کوئری می‌زند بدون اینکه در هر ریکوئست تابع یا رجکس سنگین روی میلیون‌ها کاراکتر اجرا شود. + - **سادگی و پایداری:** سازگاری کامل با معماری فعلی Django بدون نیاز به نصب اکستنشن‌های پیچیده C در دیتابیس سرور. + - **دقت ۱۰۰٪:** تضمین می‌کند که منطق سمت پایتون در هر دو طرف ذخیره و جستجو دقیقاً یکی است. + +--- + +### 🟡 گزینه دوم: تابع پایگاه داده در سطح PostgreSQL (Database-Level Stored Function & Functional Index) + +1. ایجاد یک تابع PL/pgSQL در PostgreSQL (مثلاً `fn_normalize_arabic(text)`). +2. ساخت ایندکس تابعی: + ```sql + CREATE INDEX idx_hadis_normalized_text ON hadis_hadis (fn_normalize_arabic(text)); + ``` +3. در جنگو با استفاده از `Func` یا Raw SQL: + ```python + queryset.filter(Q(normalized_text_func__icontains=normalize_text(query))) + ``` +4. **مزایا:** عدم نیاز به ذخیره دیتای مضاعف در ستون جداگانه. +5. **معایب:** وابستگی شدید به دیتابیس، سختی مایگریشن در محیط‌های توسعه و تست SQLite/Docker، و پیچیدگی نگهداری لاجیک در SQL. + +--- + +### 🔴 گزینه سوم: استفاده از Regex در زمان کوئری (Query-time Regex) + +1. تبدیل هر حرف از کلمه سرچ به یک گروه رجکس؛ مثلاً تبدیل `انما` به: + `[اأإآٱ][ًٌٍَُِّْٰ]*ن[ًٌٍَُِّْٰ]*م[ًٌٍَُِّْٰ]*[اأإآٱ]` +2. ارسال به دیتابیس با `text__iregex=pattern`. +3. **معایب:** + - **بسیار کند:** دیتابیس نمی‌تواند از هیچ ایندکسی استفاده کند (Full Table Scan با Regex Engine). + - با افزایش تعداد احادیث و اسناد، پاسخ سرور از چند میلی‌ثانیه به چند ثانیه افزایش می‌یابد و بار سرور را به شدت بالا می‌برد. + +--- + +## ۵. کد مرجع پایتون برای تابع نرمال‌سازی (Python Reference Implementation) + +این تابع کامل‌ترین و بهینه‌ترین پیاده‌سازی منطبق با استاندارد Unicode Consortium برای متون عربی و فارسی است: + +```python +import re +import unicodedata + +# 1. حرکات، اعراب، تنوین‌ها، تشدید، سکون و الف خنجری +# شامل بازه U+064B تا U+065F و الف مقصوره بالایی U+0670 +ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]') + +# 2. کاراکتر کشیدگی / تطویل +TATWEEL_REGEX = re.compile(r'\u0640') + +# 3. جدول نگاشت الف‌ها و کاراکترهای چندشکلی +ARABIC_NORMALIZATION_MAP = str.maketrans({ + # انواع الف به الف ساده + 'أ': 'ا', + 'إ': 'ا', + 'آ': 'ا', + 'ٱ': 'ا', + + # انواع یاء و الف مقصوره به یای استاندارد + 'ي': 'ی', + 'ى': 'ی', + 'ئ': 'ی', + + # کاف عربی به کاف یکسان + 'ك': 'ک', + + # تاء مربوطه و هاء + 'ة': 'ه', + 'ۀ': 'ه', + + # واو همزه‌دار + 'ؤ': 'و', + + # ارقام عربی مشرقی و فارسی به ارقام استاندارد + '٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4', + '٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9', + '۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4', + '۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9', +}) + +# 4. کاراکترهای کنترلی و نامرئی (Zero-width spaces, LRM, RLM) +ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]') + +def normalize_for_search(text: str) -> str: + """ + متن ورودی را بر اساس قواعد استاندارد جستجوی متون عربی و اسلامی نرمال‌سازی می‌کند: + 1. حذف کاراکترهای کنترلی پنهان و نیم‌فاصله‌های نامتعارف + 2. حذف کامل تمام اعراب‌ها، حرکات، تشدید، تنوین‌ها و الف خنجری (Tashkeel) + 3. حذف علامت کشیدگی (تطویل / کشیده) + 4. یکسان‌سازی الف‌ها (أ، إ، آ، ٱ -> ا) + 5. یکسان‌سازی یاء و الف مقصوره (ي، ى، ئ -> ی) + 6. یکسان‌سازی کاف (ك -> ک) + 7. یکسان‌سازی تاء مربوطه (ة -> ه) + 8. یکسان‌سازی واو همزه‌دار (ؤ -> و) + 9. تبدیل ارقام عربی و فارسی به ارقام استاندارد + 10. یکپارچه‌سازی فاصله‌های خالی چندگانه + """ + if not text or not isinstance(text, str): + return "" + + # ۱. نرمال‌سازی فرم یونیکد (NFKC) + text = unicodedata.normalize('NFKC', text) + + # ۲. حذف کاراکترهای نامرئی + text = ZERO_WIDTH_REGEX.sub('', text) + + # ۳. حذف حرکات و اعراب + text = ARABIC_DIACRITICS_REGEX.sub('', text) + + # ۴. حذف تطویل + text = TATWEEL_REGEX.sub('', text) + + # ۵. نگاشت الف‌ها و کاراکترهای هم‌ارز + text = text.translate(ARABIC_NORMALIZATION_MAP) + + # ۶. حذف فاصله‌های اضافی مکرر + text = re.sub(r'\s+', ' ', text).strip() + + return text +``` + +--- + +## ۶. نقشه راه اجرایی پیشنهادی (Recommended Implementation Roadmap) + +1. **فاز ۱ — بررسی و تأیید نهایی:** + - تأیید قوانین نرمال‌سازی فوق توسط کارفرما و تیم فنی (به‌ویژه در خصوص تبدیل `ة` به `ه` و `ؤ` به `و`). +2. **فاز ۲ — اضافه کردن ماژول Utility:** + - افزودن فایل `backend/utils/text_normalizer.py` شامل تابع `normalize_for_search`. +3. **فاز ۳ — پیاده‌سازی پایگاه داده:** + - اضافه کردن فیلدهای `search_text` یا `normalized_text` در مدل‌های کلیدی (`Hadis`، `HadisCategory`، `Transmitter`، `HadisCorrection`، `ReferenceBook`). + - تنظیم پر شدن خودکار در `save()`. + - اجرای یک Management Command برای نرمال‌سازی داده‌های قبلی. +4. **فاز ۴ — به‌روزرسانی Queryset های جستجو:** + - اصلاح متدهای `apply_search_filter` در Viewها تا ورودی کاربر را پیش از جستجو نرمال کند. +5. **فاز ۵ — تست و اعتبارسنجی:** + - نوشتن تست‌های خودکار (Unit Tests) برای کلمات چالش‌برانگیز مثل «أنما»، «إِنَّمَا»، «الرَّحْمٰنِ»، «أبو هريرة»، «مسؤول» و اطمینان از نتیجه مثبت در تمامی حالات. diff --git a/utils/text_normalizer.py b/utils/text_normalizer.py new file mode 100644 index 0000000..d4e6b91 --- /dev/null +++ b/utils/text_normalizer.py @@ -0,0 +1,135 @@ +""" +Comprehensive Arabic & Persian Text Normalization Utility for Search. +Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification, +Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers. +""" + +import re +import unicodedata +from typing import Any, Iterable, List, Optional + +# 1. Arabic Harakat / Diacritics / Quranic annotations: +# \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc. +# \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن +# \u06D6 - \u06ED : Quranic pause marks, small high signs, etc. +ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]') + +# 2. Tatweel / Kashida (ـ) +TATWEEL_REGEX = re.compile(r'\u0640') + +# 3. HTML tags remover (e.g.

,
, , etc.) +HTML_TAGS_REGEX = re.compile(r'<[^>]+>') + +# 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.) +ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]') + +# 5. Normalization translation map +ARABIC_NORMALIZATION_MAP = str.maketrans({ + # Alef variations -> Plain Alef + 'أ': 'ا', + 'إ': 'ا', + 'آ': 'ا', + 'ٱ': 'ا', + + # Yeh & Alef Maksura -> Standard Yeh + 'ي': 'ی', + 'ى': 'ی', + 'ئ': 'ی', + + # Kaf -> Standard Kaf + 'ك': 'ک', + + # Teh Marbuta & Heh with Yeh -> Standard Heh + 'ة': 'ه', + 'ۀ': 'ه', + + # Waw with Hamza -> Standard Waw + 'ؤ': 'و', + + # Eastern Arabic and Persian digits -> Standard Latin digits + '٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4', + '٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9', + '۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4', + '۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9', +}) + + +def normalize_for_search(text: Optional[str]) -> str: + """ + Normalizes a text string for fuzzy / invariant Arabic and Persian search: + - Strips HTML tags + - Unicode NFKC normalization + - Removes zero-width and invisible control characters + - Strips all diacritics / tashkeel and dagger alef (\u0670) + - Strips tatweel / kashida (\u0640) + - Unifies all Alef forms (أ, إ, آ, ٱ -> ا) + - Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی) + - Unifies Kaf (ك -> ک) + - Unifies Teh Marbuta (ة, ۀ -> ه) + - Unifies Waw with Hamza (ؤ -> و) + - Converts Arabic & Persian numerals to ASCII (0-9) + - Collapses multiple whitespace characters and lowercases English text + """ + if not text or not isinstance(text, str): + return "" + + # 1. Remove HTML tags if present + cleaned = HTML_TAGS_REGEX.sub(' ', text) + + # 2. Unicode normalization (NFKC) + cleaned = unicodedata.normalize('NFKC', cleaned) + + # 3. Remove zero-width / control characters + cleaned = ZERO_WIDTH_REGEX.sub('', cleaned) + + # 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef + cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned) + + # 5. Remove Tatweel / Kashida + cleaned = TATWEEL_REGEX.sub('', cleaned) + + # 6. Apply character unification map + cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP) + + # 7. Lowercase English characters and collapse whitespace + cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower() + + return cleaned + + +def extract_searchable_strings(data: Any) -> List[str]: + """ + Recursively extracts all text values from nested lists, dicts, + or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]). + """ + results: List[str] = [] + + if data is None: + return results + + if isinstance(data, str): + val = data.strip() + if val: + results.append(val) + elif isinstance(data, dict): + for k, v in data.items(): + results.extend(extract_searchable_strings(v)) + elif isinstance(data, (list, tuple, set)): + for item in data: + results.extend(extract_searchable_strings(item)) + + return results + + +def build_search_blob(*components: Any) -> str: + """ + Extracts all text components, normalizes them, and joins them with spaces + into a single indexed search text blob. + """ + all_raw_strings: List[str] = [] + for c in components: + all_raw_strings.extend(extract_searchable_strings(c)) + + # Join and normalize in one pass + joined = " ".join(all_raw_strings) + return normalize_for_search(joined)