2 Commits

Author SHA1 Message Date
Mohsen Taba 44c2f59328 feat(search): implement Arabic and Persian search normalization with shadow search column 2 days ago
Mohsen Taba 5f9b2f195b refactor(transmitters): convert original text to single text field with data migration and populate reference metadata scripts 2 days ago
  1. 30
      apps/hadis/admin/transmitter.py
  2. 95
      apps/hadis/management/commands/populate_normalized_search.py
  3. 100
      apps/hadis/migrations/0047_transmitteroriginaltext_text_to_textfield.py
  4. 43
      apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py
  5. 43
      apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py
  6. 6
      apps/hadis/models/category.py
  7. 30
      apps/hadis/models/hadis.py
  8. 11
      apps/hadis/models/reference.py
  9. 12
      apps/hadis/models/transmitter.py
  10. 4
      apps/hadis/serializers/hadis.py
  11. 31
      apps/hadis/serializers/serializers_admin.py
  12. 6
      apps/hadis/views/category.py
  13. 28
      apps/hadis/views/hadis.py
  14. 12
      apps/hadis/views/reference.py
  15. 3
      apps/hadis/views/reference_v2.py
  16. 9
      apps/hadis/views/transmitter.py
  17. 283
      docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md
  18. 32
      scripts/inspect_refs.py
  19. 240
      scripts/populate_reference_metadata.py
  20. 135
      utils/text_normalizer.py

30
apps/hadis/admin/transmitter.py

@ -379,31 +379,6 @@ class TransmitterOriginalTextAdminForm(forms.ModelForm):
}
}
# Schema for text JSON field
text_schema = {
"type": "array",
"title": "Texts",
"items": {
"type": "object",
"title": "Text",
"properties": {
"language_code": {
"type": "string",
"title": "Language Code",
"enum": ["en", "fa", "ar", "ur", "ru"],
"options": {
"enum_titles": ["English", "Persian", "Arabic", "Urdu", "Russian"]
}
},
"text": {
"type": "string",
"title": "Text Content"
}
},
"required": ["language_code", "text"]
}
}
# Schema for translation JSON field
translation_schema = {
"type": "array",
@ -435,11 +410,6 @@ class TransmitterOriginalTextAdminForm(forms.ModelForm):
'title': 'Titles'
})
self.fields['text'].widget = JsonEditorWidget(attrs={
'schema': json.dumps(text_schema),
'title': 'Texts'
})
self.fields['translation'].widget = JsonEditorWidget(attrs={
'schema': json.dumps(translation_schema),
'title': 'Translations'

95
apps/hadis/management/commands/populate_normalized_search.py

@ -0,0 +1,95 @@
from django.core.management.base import BaseCommand
from apps.hadis.models import (
Hadis,
Transmitters,
BookReference,
HadisCategory,
HadisCorrection,
HadisInterpretation
)
from utils.text_normalizer import build_search_blob
class Command(BaseCommand):
help = "Backfill normalized_text on Hadis, Transmitters, Categories, BookReferences, Corrections, and Interpretations"
def handle(self, *args, **options):
self.stdout.write(self.style.SUCCESS("Starting search normalization backfill..."))
# 1. Hadis
self.stdout.write("1. Normalizing Hadis records...")
hadiths = list(Hadis.objects.all().only('id', 'text', 'title', 'title_narrator', 'translation', 'normalized_text'))
for h in hadiths:
h.normalized_text = build_search_blob(
h.text,
h.title,
h.title_narrator,
h.translation
)
Hadis.objects.bulk_update(hadiths, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(hadiths)} Hadis records."))
# 2. Transmitters
self.stdout.write("2. Normalizing Transmitters records...")
transmitters = list(Transmitters.objects.all().only('id', 'full_name', 'kunya', 'known_as', 'nickname', 'normalized_text'))
for t in transmitters:
t.normalized_text = build_search_blob(
t.full_name,
t.kunya,
t.known_as,
t.nickname
)
Transmitters.objects.bulk_update(transmitters, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(transmitters)} Transmitters records."))
# 3. BookReference
self.stdout.write("3. Normalizing BookReference records...")
books = list(BookReference.objects.select_related('author').all().only('id', 'title', 'description', 'author__name', 'normalized_text'))
for b in books:
author_name = b.author.name if b.author else ""
b.normalized_text = build_search_blob(
b.title,
b.description,
author_name
)
BookReference.objects.bulk_update(books, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(books)} BookReference records."))
# 4. HadisCategory
self.stdout.write("4. Normalizing HadisCategory records...")
categories = list(HadisCategory.objects.all().only('id', 'title', 'description', 'normalized_text'))
for c in categories:
c.normalized_text = build_search_blob(
c.title,
c.description
)
HadisCategory.objects.bulk_update(categories, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(categories)} HadisCategory records."))
# 5. HadisCorrection
self.stdout.write("5. Normalizing HadisCorrection records...")
corrections = list(HadisCorrection.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text'))
for corr in corrections:
corr.normalized_text = build_search_blob(
corr.text,
corr.title,
corr.narrator,
corr.translation
)
HadisCorrection.objects.bulk_update(corrections, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(corrections)} HadisCorrection records."))
# 6. HadisInterpretation
self.stdout.write("6. Normalizing HadisInterpretation records...")
interpretations = list(HadisInterpretation.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text'))
for interp in interpretations:
interp.normalized_text = build_search_blob(
interp.text,
interp.title,
interp.narrator,
interp.translation
)
HadisInterpretation.objects.bulk_update(interpretations, ['normalized_text'], batch_size=500)
self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(interpretations)} HadisInterpretation records."))
self.stdout.write(self.style.SUCCESS("All search normalization fields backfilled successfully!"))

100
apps/hadis/migrations/0047_transmitteroriginaltext_text_to_textfield.py

@ -0,0 +1,100 @@
# Generated by Django on 2026-10-08
from django.db import migrations, models
def migrate_text_data_forward(apps, schema_editor):
TransmitterOriginalText = apps.get_model('hadis', 'TransmitterOriginalText')
items_to_update = []
for obj in TransmitterOriginalText.objects.all():
raw_text = obj.text
raw_translation = obj.translation
primary_text = ""
items_by_lang = {}
if isinstance(raw_text, list):
for item in raw_text:
if isinstance(item, dict):
lang = item.get('language_code') or 'unknown'
txt = (item.get('text') or item.get('title') or item.get('value') or '').strip()
if txt:
items_by_lang[lang] = txt
if 'ar' in items_by_lang:
primary_text = items_by_lang['ar']
elif 'fa' in items_by_lang:
primary_text = items_by_lang['fa']
elif items_by_lang:
primary_text = list(items_by_lang.values())[0]
elif isinstance(raw_text, str):
primary_text = raw_text
# Ensure all other language translations from text are preserved in translation field
updated_trans = list(raw_translation) if isinstance(raw_translation, list) else []
existing_langs = set()
for tr in updated_trans:
if isinstance(tr, dict) and tr.get('language_code'):
existing_langs.add(tr.get('language_code'))
for lang, txt in items_by_lang.items():
if lang not in existing_langs:
updated_trans.append({
'language_code': lang,
'title': txt,
'text': txt
})
existing_langs.add(lang)
obj.temp_text = primary_text
obj.translation = updated_trans
items_to_update.append(obj)
if len(items_to_update) >= 200:
TransmitterOriginalText.objects.bulk_update(items_to_update, ['temp_text', 'translation'])
items_to_update = []
if items_to_update:
TransmitterOriginalText.objects.bulk_update(items_to_update, ['temp_text', 'translation'])
def migrate_text_data_backward(apps, schema_editor):
TransmitterOriginalText = apps.get_model('hadis', 'TransmitterOriginalText')
items_to_update = []
for obj in TransmitterOriginalText.objects.all():
if isinstance(obj.text, str) and obj.text:
obj.temp_text = [{'language_code': 'ar', 'text': obj.text, 'title': obj.text}]
items_to_update.append(obj)
if len(items_to_update) >= 200:
TransmitterOriginalText.objects.bulk_update(items_to_update, ['temp_text'])
items_to_update = []
if items_to_update:
TransmitterOriginalText.objects.bulk_update(items_to_update, ['temp_text'])
class Migration(migrations.Migration):
dependencies = [
('hadis', '0046_originaltextreference_hadith_number'),
]
operations = [
migrations.AddField(
model_name='transmitteroriginaltext',
name='temp_text',
field=models.TextField(blank=True, default='', verbose_name='Text'),
),
migrations.RunPython(migrate_text_data_forward, reverse_code=migrate_text_data_backward),
migrations.RemoveField(
model_name='transmitteroriginaltext',
name='text',
),
migrations.RenameField(
model_name='transmitteroriginaltext',
old_name='temp_text',
new_name='text',
),
]

43
apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py

@ -0,0 +1,43 @@
# Generated by Django 4.2.30 on 2026-10-08 10:37
from django.db import migrations, models
class Migration(migrations.Migration):
dependencies = [
('hadis', '0047_transmitteroriginaltext_text_to_textfield'),
]
operations = [
migrations.AddField(
model_name='bookreference',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
migrations.AddField(
model_name='hadis',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
migrations.AddField(
model_name='hadiscategory',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
migrations.AddField(
model_name='hadiscorrection',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
migrations.AddField(
model_name='hadisinterpretation',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
migrations.AddField(
model_name='transmitters',
name='normalized_text',
field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'),
),
]

43
apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py

@ -0,0 +1,43 @@
# Generated by Django 4.2.30 on 2026-10-08 10:40
from django.db import migrations, models
class Migration(migrations.Migration):
dependencies = [
('hadis', '0048_bookreference_normalized_text_hadis_normalized_text_and_more'),
]
operations = [
migrations.AlterField(
model_name='bookreference',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
migrations.AlterField(
model_name='hadis',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
migrations.AlterField(
model_name='hadiscategory',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
migrations.AlterField(
model_name='hadiscorrection',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
migrations.AlterField(
model_name='hadisinterpretation',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
migrations.AlterField(
model_name='transmitters',
name='normalized_text',
field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'),
),
]

6
apps/hadis/models/category.py

@ -80,6 +80,7 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel):
order = models.IntegerField(default=0, verbose_name=_('order'))
xmind_file = models.FileField(upload_to='hadis/xmind_files/', verbose_name=_('xmind file'), null=True, blank=True)
slug = models.SlugField(max_length=255, null=True, blank=True,allow_unicode=True)
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
content_type = None
language = None
language_id = None
@ -111,6 +112,11 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel):
slug_source_field = 'title'
def save(self, *args, **kwargs):
from utils.text_normalizer import build_search_blob
self.normalized_text = build_search_blob(
self.title,
self.description
)
self.full_clean()
super().save(*args, **kwargs)

30
apps/hadis/models/hadis.py

@ -182,6 +182,7 @@ class Hadis(LowercaseSlugMixin, models.Model):
created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at'))
updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at'))
embedded_in = models.JSONField(default=list, blank=True)
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
def __str__(self):
title = None
@ -232,6 +233,15 @@ class Hadis(LowercaseSlugMixin, models.Model):
old_instance.translation != self.translation):
self.embedded_in = [] # Reset!
# Populate normalized_text for fast and accurate Arabic/Persian search
from utils.text_normalizer import build_search_blob
self.normalized_text = build_search_blob(
self.text,
self.title,
self.title_narrator,
self.translation
)
super().save(*args, **kwargs)
def _get_json_field(self, field_name: str, lang: Optional[str]=None , fallback: str = "en"):
@ -417,6 +427,7 @@ class HadisCorrection(models.Model):
updated_at = models.DateTimeField(auto_now=True, verbose_name=_("updated at"))
share_link = models.CharField(max_length=255, verbose_name=_('share link'), null=True, blank=True)
embedded_in = models.JSONField(default=list, blank=True)
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
class Meta:
verbose_name = _("Hadis Correction")
@ -442,6 +453,15 @@ class HadisCorrection(models.Model):
old_instance.translation != self.translation):
self.embedded_in = [] # Reset!
# Populate normalized_text for fast search
from utils.text_normalizer import build_search_blob
self.normalized_text = build_search_blob(
self.text,
self.title,
self.narrator,
self.translation
)
super().save(*args, **kwargs)
def get_title(self, lang):
@ -474,6 +494,7 @@ class HadisInterpretation(models.Model):
text = models.TextField(verbose_name=_('Original Text'))
translation = models.JSONField(verbose_name=_('Translation'), default=list)
priority = models.IntegerField(default=0, verbose_name=_('priority'))
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
created_at = models.DateTimeField(auto_now_add=True)
updated_at = models.DateTimeField(auto_now=True)
@ -486,6 +507,15 @@ class HadisInterpretation(models.Model):
# 👈 ست کردن اسلاگ تفسیر برابر با اسلاگ کتگوری
if self.category and self.category.slug:
self.slug = self.category.slug
from utils.text_normalizer import build_search_blob
self.normalized_text = build_search_blob(
self.text,
self.title,
self.narrator,
self.translation
)
super().save(*args, **kwargs)
# --- FOR CORRECTIONS ---

11
apps/hadis/models/reference.py

@ -143,6 +143,7 @@ class BookReference(LowercaseSlugMixin, models.Model):
created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at'))
updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at'))
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
class Meta:
indexes = [
@ -157,6 +158,16 @@ class BookReference(LowercaseSlugMixin, models.Model):
return self.title[0].get('text') or self.title[0].get('title') or f"Reference {self.id}"
return f"Reference {self.id}"
def save(self, *args, **kwargs):
from utils.text_normalizer import build_search_blob
author_name = self.author.name if self.author else ""
self.normalized_text = build_search_blob(
self.title,
self.description,
author_name
)
super().save(*args, **kwargs)
def _get_json_field(self, field_name: str, lang: Optional[str]=None , fallback: str = "en"):
"""
Generic getter for JSONField in our [{text, language_code}] format.

12
apps/hadis/models/transmitter.py

@ -207,6 +207,7 @@ class Transmitters(LowercaseSlugMixin, models.Model):
created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at'))
updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at'))
normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search'))
class Meta:
indexes = [
@ -222,6 +223,15 @@ class Transmitters(LowercaseSlugMixin, models.Model):
self.legacy_id = None
if not self.legacy_number:
self.legacy_number = None
from utils.text_normalizer import build_search_blob
self.normalized_text = build_search_blob(
self.full_name,
self.kunya,
self.known_as,
self.nickname
)
super().save(*args, **kwargs)
@property
@ -494,7 +504,7 @@ class TransmitterOriginalText(models.Model):
)
slug = models.SlugField(max_length=255, verbose_name=_('slug'), blank=True,unique=True, allow_unicode=True)
title = models.JSONField(default = list , verbose_name=_('Title'))
text = models.JSONField(default = list , verbose_name=_('Text'))
text = models.TextField(default='', blank=True, verbose_name=_('Text'))
translation = models.JSONField(verbose_name=_('translation'), default=list)
share_link = models.CharField(max_length=255, verbose_name=_('share link'), null=True, blank=True)
embedded_in = models.JSONField(default=list, blank=True)

4
apps/hadis/serializers/hadis.py

@ -618,7 +618,7 @@ def format_links_output(links_data):
class TransmitterOriginalTextSerializer(serializers.ModelSerializer):
""" Serializer for TransmitterOriginalText """
title = LocalizedField()
text = LocalizedField()
text = serializers.CharField()
translation = LocalizedField()
address = serializers.SerializerMethodField()
images = serializers.SerializerMethodField()
@ -1957,7 +1957,7 @@ class HadisSyncInterpretationSerializer(HadisInterpretationDetailSerializer):
class TransmitterOriginalTextDetailSerializer(serializers.ModelSerializer):
title = LocalizedField()
text = LocalizedField()
text = serializers.CharField()
translation = LocalizedField()
references = DetailedOriginalTextReferenceSerializer(many=True, read_only=True)
transmitter_info = serializers.SerializerMethodField()

31
apps/hadis/serializers/serializers_admin.py

@ -2643,7 +2643,7 @@ class AdminTransmitterOpinionSerializer(serializers.ModelSerializer):
class AdminTransmitterOriginalTextSerializer(serializers.ModelSerializer):
title = serializers.JSONField(required=False, default=list)
text = serializers.JSONField(required=False, default=list)
text = serializers.CharField(required=False, allow_blank=True, default="")
translation = serializers.JSONField(required=False, default=list)
references = AdminOriginalTextReferenceSerializer(many=True, read_only=True)
@ -2686,7 +2686,34 @@ class AdminTransmitterOriginalTextSerializer(serializers.ModelSerializer):
def to_internal_value(self, data):
import json
data = safe_copy_data(data)
for json_field in ["title", "text", "translation", "address"]:
# Handle backward compatibility: if text is passed as a JSON list, extract primary text
text_val = data.get("text")
if isinstance(text_val, str) and (text_val.strip().startswith("[") or text_val.strip().startswith("{")):
try:
parsed = json.loads(text_val)
if isinstance(parsed, list):
p_txt = ""
for it in parsed:
if isinstance(it, dict) and it.get("language_code") == "ar":
p_txt = it.get("text") or it.get("title") or ""
break
if not p_txt and parsed:
p_txt = parsed[0].get("text") or parsed[0].get("title") or ""
data["text"] = p_txt
except Exception:
pass
elif isinstance(text_val, list):
p_txt = ""
for it in text_val:
if isinstance(it, dict) and it.get("language_code") == "ar":
p_txt = it.get("text") or it.get("title") or ""
break
if not p_txt and text_val:
p_txt = text_val[0].get("text") or text_val[0].get("title") or ""
data["text"] = p_txt
for json_field in ["title", "translation", "address"]:
val = data.get(json_field)
if isinstance(val, str):
try:

6
apps/hadis/views/category.py

@ -477,7 +477,10 @@ class CategoriesView(ListAPIView):
)
search_query = (self.request.query_params.get('search') or '').strip()
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
queryset = queryset.filter(
Q(normalized_text__icontains=norm_q) |
Q(title__icontains=search_query) |
Q(slug__icontains=search_query) |
Q(description__icontains=search_query)
@ -514,7 +517,10 @@ class CategoriesBySectView(ListAPIView):
)
search_query = (self.request.query_params.get('search') or '').strip()
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
queryset = queryset.filter(
Q(normalized_text__icontains=norm_q) |
Q(title__icontains=search_query) |
Q(slug__icontains=search_query) |
Q(description__icontains=search_query)

28
apps/hadis/views/hadis.py

@ -361,12 +361,18 @@ class HadisListView(ListAPIView):
# 👇 1. Apply Search Filter
search_query = self.request.query_params.get('search', None)
if search_query:
search_conditions = Q(text__icontains=search_query)
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
search_conditions = Q(normalized_text__icontains=norm_q)
search_conditions |= Q(text__icontains=search_query)
search_conditions |= Q(title__icontains=search_query)
search_conditions |= Q(title_narrator__icontains=search_query)
search_conditions |= Q(translation__icontains=search_query)
search_conditions |= Q(tags__title__icontains=search_query)
search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q)
search_conditions |= Q(references__book_reference__title__icontains=search_query)
search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q)
search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query)
search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query)
queryset = queryset.filter(search_conditions)
@ -625,21 +631,25 @@ class HadisMainListView(ListAPIView):
def apply_search_filter(self, queryset, search_query):
"""
Apply search filter across multiple fields including JSONFields.
Searches in: title, title_narrator, text, translation, tags, book references, transmitters
Apply search filter across multiple fields including normalized text and JSONFields.
Searches in: normalized_text, title, title_narrator, text, translation, tags, book references, transmitters
"""
from django.db.models import Q
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
# Basic search conditions
search_conditions = Q(text__icontains=search_query)
# Primary normalized search (matches Arabic/Persian/English invariantly)
search_conditions = Q(normalized_text__icontains=norm_q)
# For JSONFields, search in the JSON string representation
# This will find matches in the "text" values within the JSON arrays
# Fallback raw search
search_conditions |= Q(text__icontains=search_query)
search_conditions |= Q(title__icontains=search_query)
search_conditions |= Q(title_narrator__icontains=search_query)
search_conditions |= Q(translation__icontains=search_query)
search_conditions |= Q(tags__title__icontains=search_query)
search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q)
search_conditions |= Q(references__book_reference__title__icontains=search_query)
search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q)
search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query)
search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query)
@ -919,7 +929,11 @@ class CategoryHadisCorrectionsView(ListAPIView):
# Optional search query filter
search_query = self.request.query_params.get('search', None)
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
search_conditions = (
Q(normalized_text__icontains=norm_q) |
Q(hadis__normalized_text__icontains=norm_q) |
Q(title__icontains=search_query) |
Q(text__icontains=search_query) |
Q(narrator__icontains=search_query) |

12
apps/hadis/views/reference.py

@ -36,15 +36,13 @@ class BookReferencesView(ListAPIView):
def apply_search_filter(self, queryset, search_query):
"""
Apply search filter across book titles (JSONField).
Searches in: title
Apply search filter across book titles (JSONField) and normalized text.
Searches in: normalized_text, title
"""
# Search conditions
search_conditions = Q()
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
# For JSONFields, search in the JSON string representation
# This will find matches in the "text" values within the JSON arrays
search_conditions |= Q(title__icontains=search_query)
search_conditions = Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query)
return queryset.filter(search_conditions)

3
apps/hadis/views/reference_v2.py

@ -163,7 +163,10 @@ class BookReferenceV2ListView(generics.ListAPIView):
search = self.request.query_params.get('search')
if search:
from django.db.models import Q
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search)
qs = qs.filter(
Q(normalized_text__icontains=norm_q) |
Q(title__icontains=search) |
Q(description__icontains=search) |
Q(author__name__icontains=search)

9
apps/hadis/views/transmitter.py

@ -42,7 +42,10 @@ class TransmitterView(ListAPIView):
# 1. Apply search filter (Searching across name-related JSONFields, slug, legacy_id)
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
q_filter = (
Q(normalized_text__icontains=norm_q) |
Q(full_name__icontains=search_query) |
Q(kunya__icontains=search_query) |
Q(known_as__icontains=search_query) |
@ -421,7 +424,10 @@ class NarratorTeachersView(ListAPIView):
search_query = self.request.query_params.get('search', None)
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
queryset = queryset.filter(
Q(normalized_text__icontains=norm_q) |
Q(full_name__icontains=search_query) |
Q(kunya__icontains=search_query) |
Q(known_as__icontains=search_query) |
@ -471,7 +477,10 @@ class NarratorStudentsView(ListAPIView):
search_query = self.request.query_params.get('search', None)
if search_query:
from utils.text_normalizer import normalize_for_search
norm_q = normalize_for_search(search_query)
queryset = queryset.filter(
Q(normalized_text__icontains=norm_q) |
Q(full_name__icontains=search_query) |
Q(kunya__icontains=search_query) |
Q(known_as__icontains=search_query) |

283
docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md

@ -0,0 +1,283 @@
# دستورالعمل جامع نرمال‌سازی جستجو برای متون عربی و اسلامی
### (Comprehensive Arabic & Persian Search Normalization Specification)
---
## ۱. مقدمه و بیان مسئله (Executive Summary & Problem Statement)
در پایگاه‌های داده و سامانه‌های متنی اسلامی و حدیثی، جستجوی متنی با چالش‌های بنیادین زبان‌شناختی و فنی مواجه است:
1. **گوناگونی رسم‌الخط و کدگذاری یونیکد (Unicode Homoglyphs):** نویسه‌هایی مانند الف همزه‌دار (`أ`، `إ`)، الف ممدوده (`آ`)، الف وصل (`ٱ`) و الف ساده (`ا`) دارای کدهای یونیکد کاملاً متفاوتی هستند.
2. **اعراب و اعجام (Harakat / Tashkeel):** متون احادیث و اسناد دینی معمولاً با حرکت‌گذاری کامل (مانند `إِنَّمَا الأَعْمَالُ`) در پایگاه داده ذخیره شده‌اند، در حالی که کاربران جستجوهای خود را بدون اعراب (`انما الاعمال` یا `انما`) تایپ می‌کنند. در مقایسه‌های متنی پیش‌فرض دیتابیس (`LIKE` یا `ILIKE` و `icontains`)، وجود اعراب در میان حروف کلمه باعث می‌شود کلمه کاربر تطبیق داده نشود و نتیجه «هیچ رکوردی یافت نشد» برگردد.
3. **تداخل کاراکترهای فارسی و عربی:** در کیبوردهای کاربران (موبایل و دسکتاپ)، حروفی چون «ی» و «ي»، «ک» و «ك»، «ه» و «ة» مدام جابه‌جا می‌شوند.
4. **علائم قرآنی و وقوف:** نمادهایی مانند «الف خنجری» (`ٰ` مانند `رَحْمٰنِ`)، علائم وقف و نمادهای تعظیم اگر نرمال نشوند، جستجو را مسدود می‌کنند.
> **هدف:** هر متنی که کاربر با هر شکلی از کیبورد (با اعراب، بدون اعراب، با همزه یا بدون همزه، فارسی یا عربی) جستجو کرد، سیستم باید دقیقاً مفهوم آن را درک کرده و تمامی رکوردهای منطبق را بدون وابستگی به شکل ظاهری نگارش پیدا کند.
---
## ۲. دسته‌بندی جامع کاراکترها و حالات نرمال‌سازی (Normalization Taxonomy)
### ۲.۱. خانواده الف‌ها و همزه‌ها (Alef & Hamza Variants)
الف در رسم‌الخط عربی دارای حالت‌های متعددی است که در جستجو باید همگی یکپارچه شوند:
| نویسه اصلی | نام یونیکد | کد یونیکد | هدف نرمال‌سازی | مثال جستجو | مثال در دیتابیس |
| :--- | :--- | :--- | :--- | :--- | :--- |
| **أ** | Arabic Letter Alef With Hamza Above | `U+0623` | **ا** (`U+0627`) | أنما | انما / إِنَّمَا |
| **إ** | Arabic Letter Alef With Hamza Below | `U+0625` | **ا** (`U+0627`) | إسناد | اسناد |
| **آ** | Arabic Letter Alef With Madda Above | `U+0622` | **ا** (`U+0627`) | آثار | اثار |
| **ٱ** | Arabic Letter Alef Wasla | `U+0671` | **ا** (`U+0627`) | ٱمرؤ | امرؤ |
| **ا** | Arabic Letter Alef (Bare) | `U+0627` | **ا** (`U+0627`) | اعمال | أعمال |
| **ٴ** | Arabic Letter High Hamza | `U+0674` | *حذف* یا **ا** | — | — |
#### حالات دیگر همزه‌ها (Other Hamza Forms):
- **همزه روی واو (`ؤ` - `U+0624`):** در جستجوی آسان‌گیر (Lenient)، همزه روی واو به `و` (`U+0648`) نرمال می‌شود تا کاربر با سرچ «مومن» بتواند «مؤمن» را پیدا کند یا بالعکس («مسؤول» / «مسئول»).
- **همزه روی یاء / نبره (`ئ` - `U+0626`):** به `ی` / `ي` تبدیل می‌شود تا کلماتی چون «قائل»، «قایل»، «هیئة»، «هيئة» هم‌ارز شوند.
- **همزه تنها روی خط (`ء` - `U+0621`):** حذف یا تبدیل به فاصله در صورت نیاز.
---
### ۲.۲. خانواده یاء و الف مقصوره (Yeh & Alef Maksura)
یکی از پرتکرارترین خطاها در جستجوی عربی و فارسی مربوط به حرف «ی» است:
| نویسه اصلی | نام | کد یونیکد | رفتار نرمال‌سازی | مثال |
| :--- | :--- | :--- | :--- | :--- |
| **ي** | یاء عربی دو نقطه | `U+064A` | تبدیل به نویسه واحد مبنا (مثلاً `ی` یا `ي`) | علي / علی |
| **ى** | الف مقصوره عربی (بی‌نقطه) | `U+0649` | تبدیل به `ی` یا `ا` بر اساس سیاست پروژه | موسی / موسي / موسى |
| **ی** | یای فارسی بدون نقطه | `U+06CC` | تبدیل به نویسه واحد مبنا | حدیث / حديث |
| **ئ** | یاء با همزه | `U+0626` | تبدیل به نویسه واحد مبنا | بئر / بیر |
> **نکته تخصصی در متون دینی:** الف مقصوره (`ى` در انتهای کلماتی چون «حتى»، «إلى»، «موسى») در کیبورد کاربران گاهی با `ی`، گاهی با `ي` و حتی گاهی به اشتباه با `ا` («حتا») نوشته می‌شود. نرمال‌سازی `[ي ى ی ئ]` به یک نویسه پایدار، پوشش جستجو را به ۱۰۰٪ می‌رساند.
---
### ۲.۳. کاف عربی و فارسی (Kaf Normalization)
- **ك** (کاف عربی با نشان همزه/کاف کوچک: `U+0643`)
- **ک** (کاف فارسی سرکش‌دار: `U+06A9`)
- **قاعده:** تبدیل هر دو به یک فرم استاندارد (مثلاً `ك` برای متون عربی یا `ک`).
---
### ۲.۴. تاء مربوطه و هاء (Teh Marbuta & Heh)
- **ة** (تاء مربوطه: `U+0629`)
- **ه** (هاء: `U+0647`)
- **ۀ** (هاء با همزه: `U+06C0`)
- **تحلیل رفتاری:** کاربران در تایپ سریع اسامی یا اصطلاحات اغلب تاء مربوطه را با هاء جابه‌جا می‌زنند:
- «معاویه» ↔ «معاوية»
- «فاطمه» ↔ «فاطمة»
- «صحابه» ↔ «صحابة»
- «رواة» ↔ «رواه»
- **قاعده سرچ نرمال:** برای جستجوی متنی، `ة` به `ه` تبدیل می‌شود تا تفاوت نگارشی کاربر باعث حذف رکورد نشود.
---
### ۲.۵. حرکات، اعراب و تنوین‌ها (Tashkeel / Harakat / Diacritics) - حیاتی‌ترین بخش
تمام حرکات زیر باید از متن ورودی و از نسخه ایندکس‌شده جستجو **کاملاً حذف شوند**:
| نام حرکت | علامت | کد یونیکد | اثر در جستجوی خام دیتابیس |
| :--- | :---: | :---: | :--- |
| **فتحه (Fatha)** | َ | `U+064E` | مانع تطابق متن ساده می‌شود |
| **ضمه (Damma)** | ُ | `U+064F` | مانع تطابق متن ساده می‌شود |
| **کسره (Kasra)** | ِ | `U+0650` | مانع تطابق متن ساده می‌شود |
| **تنوین نصب (Fathatan)** | ً | `U+064B` | مانع تطابق |
| **تنوین رفع (Dammatan)** | ٌ | `U+064C` | مانع تطابق |
| **تنوین جر (Kasratan)** | ٍ | `U+064D` | مانع تطابق |
| **سکون (Sukun)** | ْ | `U+0652` | مانع تطابق |
| **تشدید (Shadda)** | ّ | `U+0651` | مانع تطابق کلمات دارای تشدید |
| **الف خنجری (Dagger Alef)** | ٰ | `U+0670` | **بسیار خطرناک:** در «رَحْمٰنِ»، «إِلٰهَ»، «هٰذَا»، «إِسْمٰعِيل» اگر حذف نشود، سرچ «رحمن» هرگز «رحمٰن» را پیدا نمی‌کند! |
| **مده (Maddah)** | ٓ | `U+0653` | مانع تطابق |
| **همزه فوقانی اعرابی** | ٔ | `U+0654` | مانع تطابق |
| **همزه تحتانی اعرابی** | ٕ | `U+0655` | مانع تطابق |
---
### ۲.۶. کشیدگی، تطویل و نشانه‌های نامرئی (Tatweel & Invisible Characters)
- **تطویل / کشیده (`ـ` - `U+0640`):** برای تنظیم طول خطوط در متون کهن یا زیبایی متنی به کار می‌رود (مثل `صـــــراط` یا `رســــول`). این کاراکتر باید به کلی حذف شود.
- **نیم‌فاصله (Zero-Width Non-Joiner - `U+200C`):** در متون فارسی و اسامی ترکیبی وجود دارد؛ باید به فاصله عادی یا حذف کامل تبدیل شود.
- **اتصال‌دهنده مجازی (ZWJ - `U+200D`):** باید حذف شود.
- **نشانگرهای جهت یونیکد (LRM `U+200E` و RLM `U+200F`):** باید کاملاً حذف شوند.
---
### ۲.۷. نشانه‌ها و نمادهای وقوف قرآنی و مذهبی (Quranic Symbols & Waqf Marks)
در متون روایی و قرآنی نمادهای ویژه‌ای در یونیکد ذخیره می‌شوند که باید در لایه سرچ پالایش گردند:
- علائم وقف قرآنی: `ۖ` (`U+06D6`), `ۗ` (`U+06D7`), `ۘ` (`U+06D8`), `ۙ` (`U+06D9`), `ۚ` (`U+06DA`), `ۛ` (`U+06DB`), `ۜ` (`U+06DC`), `۝` (`U+06DD`), `۞` (`U+06DE`), `۟` (`U+06DF`), `۠` (`U+06E0`), `ۡ` (`U+06E1`), `ۢ` (`U+06E2`), `ۣ` (`U+06E3`), `ۤ` (`U+06E4`).
- نمادهای لیگچر مذهبی: `ﷺ` (`U+FDFA`), `ﷻ` (`U+FDFB`), `﷽` (`U+FDFD`), `ؑ` (`U+0611`).
- پرانتزها و براکت‌های قرآنی و نقل‌قول: `﴿`، `﴾`، `«`، `»`، `[`، `]`، `(`، `)`.
---
### ۲.۸. ارقام و اعداد (Digits)
- ارقام عربی-مشرقی: `[٠, ١, ٢, ٣, ٤, ٥, ٦, ٧, ٨, ٩]`
- ارقام فارسی: `[۰, ۱, ۲, ۳, ۴, ۵, ۶, ۷, ۸, ۹]`
- ارقام استاندارد لاتین: `[0, 1, 2, 3, 4, 5, 6, 7, 8, 9]`
- **قاعده:** تبدیل تمام ارقام به ارقام استاندارد (0-9) تا سرچ شماره حدیث یا جلد و صفحه فارغ از نوع کیبورد عدد را پیدا کند.
---
## ۳. ماتریس مقایسه‌ای سناریوهای سرچ (Search Scenario Matrix)
| ورودی کاربر در سرچ | متن در دیتابیس | وضعیت جستجوی فعلی (خام) | وضعیت پس از نرمال‌سازی |
| :--- | :--- | :---: | :---: |
| `انما` | `أنما الأعمال بالنيات` | ❌ No result (تفاوت `ا` و `أ`) | ✅ منطبق و پیدا می‌شود |
| `انما` | `إِنَّمَا الأَعْمَالُ بِالنِّيَّاتِ` | ❌ No result (وجود کسره، تشدید، فتحه) | ✅ منطبق و پیدا می‌شود |
| `أبو هريرة` | `ابو هريره` | ❌ No result (تفاوت `أ/ا` و `ة/ه`) | ✅ منطبق و پیدا می‌شود |
| `صحیح بخاری` | `صَحِيحُ الْبُخَارِيِّ` | ❌ No result (تفاوت اعراب، `ی/ي`) | ✅ منطبق و پیدا می‌شود |
| `رحمن` | `الرَّحْمٰنِ الرَّحِيمِ` | ❌ No result (وجود الف خنجری `ٰ`) | ✅ منطبق و پیدا می‌شود |
| `مومن` | `إِنَّمَا الْمُؤْمِنُونَ إِخْوَةٌ` | ❌ No result (تفاوت `و` با `ؤ`) | ✅ منطبق و پیدا می‌شود |
| `صراط` | `صــــراط الذين` | ❌ No result (وجود کشیده `ـ`) | ✅ منطبق و پیدا می‌شود |
| `حدیث ۱۱۰` | `حديث 110` یا `حديث ١١٠` | ❌ No result (تفاوت ارقام) | ✅ منطبق و پیدا می‌شود |
---
## ۴. گزینه‌ها و معماری پیاده‌سازی فنی در Django و PostgreSQL
برای اعمال این نرمال‌سازی در سیستم، سه رویکرد معماری وجود دارد:
### 🟢 گزینه اول: الگوی ستون جستجوی نرمال‌شده (Normalized Shadow Column / Search Column) — **[رویکرد پیشنهادی و استاندارد]**
در این الگو، متن اصلی برای نمایش دست‌نخورده باقی می‌ماند (تا اعراب و زیبایی اصیل آن در UI حفظ شود)، اما یک ستون متنی نرمال‌شده در کنار آن ایجاد و ایندکس‌گذاری می‌شود:
1. **در مدل‌ها (`Hadis`، `HadisCategory`، `Transmitter` و ...):**
- افزودن فیلد `normalized_text = models.TextField(blank=True, db_index=True)` یا استفاده از `django.contrib.postgres.search.SearchVector`.
- در متد `save()` مدل، متن اصلی از تابع نرمال‌ساز عبور کرده و فیلد نرمال‌شده به صورت خودکار پر می‌شود.
- ایجاد یک اسکریپت ساده migration برای پر کردن یک‌باره مقادیر رکوردهای موجود.
2. **در لایه Queryset / View:**
- عبارت سرچ کاربر (`search_query`) توسط همان تابع پایتون نرمال‌سازی می‌شود: `normalized_q = normalize_text(search_query)`.
- جستجو روی ستون `normalized_text__icontains=normalized_q` انجام می‌شود.
3. **مزایا:**
- **فوق‌العاده سریع (High Performance):** دیتابیس مستقیماً روی ستون ایندکس‌شده کوئری می‌زند بدون اینکه در هر ریکوئست تابع یا رجکس سنگین روی میلیون‌ها کاراکتر اجرا شود.
- **سادگی و پایداری:** سازگاری کامل با معماری فعلی Django بدون نیاز به نصب اکستنشن‌های پیچیده C در دیتابیس سرور.
- **دقت ۱۰۰٪:** تضمین می‌کند که منطق سمت پایتون در هر دو طرف ذخیره و جستجو دقیقاً یکی است.
---
### 🟡 گزینه دوم: تابع پایگاه داده در سطح PostgreSQL (Database-Level Stored Function & Functional Index)
1. ایجاد یک تابع PL/pgSQL در PostgreSQL (مثلاً `fn_normalize_arabic(text)`).
2. ساخت ایندکس تابعی:
```sql
CREATE INDEX idx_hadis_normalized_text ON hadis_hadis (fn_normalize_arabic(text));
```
3. در جنگو با استفاده از `Func` یا Raw SQL:
```python
queryset.filter(Q(normalized_text_func__icontains=normalize_text(query)))
```
4. **مزایا:** عدم نیاز به ذخیره دیتای مضاعف در ستون جداگانه.
5. **معایب:** وابستگی شدید به دیتابیس، سختی مایگریشن در محیط‌های توسعه و تست SQLite/Docker، و پیچیدگی نگهداری لاجیک در SQL.
---
### 🔴 گزینه سوم: استفاده از Regex در زمان کوئری (Query-time Regex)
1. تبدیل هر حرف از کلمه سرچ به یک گروه رجکس؛ مثلاً تبدیل `انما` به:
`[اأإآٱ][ًٌٍَُِّْٰ]*ن[ًٌٍَُِّْٰ]*م[ًٌٍَُِّْٰ]*[اأإآٱ]`
2. ارسال به دیتابیس با `text__iregex=pattern`.
3. **معایب:**
- **بسیار کند:** دیتابیس نمی‌تواند از هیچ ایندکسی استفاده کند (Full Table Scan با Regex Engine).
- با افزایش تعداد احادیث و اسناد، پاسخ سرور از چند میلی‌ثانیه به چند ثانیه افزایش می‌یابد و بار سرور را به شدت بالا می‌برد.
---
## ۵. کد مرجع پایتون برای تابع نرمال‌سازی (Python Reference Implementation)
این تابع کامل‌ترین و بهینه‌ترین پیاده‌سازی منطبق با استاندارد Unicode Consortium برای متون عربی و فارسی است:
```python
import re
import unicodedata
# 1. حرکات، اعراب، تنوین‌ها، تشدید، سکون و الف خنجری
# شامل بازه U+064B تا U+065F و الف مقصوره بالایی U+0670
ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]')
# 2. کاراکتر کشیدگی / تطویل
TATWEEL_REGEX = re.compile(r'\u0640')
# 3. جدول نگاشت الف‌ها و کاراکترهای چندشکلی
ARABIC_NORMALIZATION_MAP = str.maketrans({
# انواع الف به الف ساده
'أ': 'ا',
'إ': 'ا',
'آ': 'ا',
'ٱ': 'ا',
# انواع یاء و الف مقصوره به یای استاندارد
'ي': 'ی',
'ى': 'ی',
'ئ': 'ی',
# کاف عربی به کاف یکسان
'ك': 'ک',
# تاء مربوطه و هاء
'ة': 'ه',
'ۀ': 'ه',
# واو همزه‌دار
'ؤ': 'و',
# ارقام عربی مشرقی و فارسی به ارقام استاندارد
'٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4',
'٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9',
'۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4',
'۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9',
})
# 4. کاراکترهای کنترلی و نامرئی (Zero-width spaces, LRM, RLM)
ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]')
def normalize_for_search(text: str) -> str:
"""
متن ورودی را بر اساس قواعد استاندارد جستجوی متون عربی و اسلامی نرمال‌سازی می‌کند:
1. حذف کاراکترهای کنترلی پنهان و نیم‌فاصله‌های نامتعارف
2. حذف کامل تمام اعراب‌ها، حرکات، تشدید، تنوین‌ها و الف خنجری (Tashkeel)
3. حذف علامت کشیدگی (تطویل / کشیده)
4. یکسان‌سازی الف‌ها (أ، إ، آ، ٱ -> ا)
5. یکسان‌سازی یاء و الف مقصوره (ي، ى، ئ -> ی)
6. یکسان‌سازی کاف (ك -> ک)
7. یکسان‌سازی تاء مربوطه (ة -> ه)
8. یکسان‌سازی واو همزه‌دار (ؤ -> و)
9. تبدیل ارقام عربی و فارسی به ارقام استاندارد
10. یکپارچه‌سازی فاصله‌های خالی چندگانه
"""
if not text or not isinstance(text, str):
return ""
# ۱. نرمال‌سازی فرم یونیکد (NFKC)
text = unicodedata.normalize('NFKC', text)
# ۲. حذف کاراکترهای نامرئی
text = ZERO_WIDTH_REGEX.sub('', text)
# ۳. حذف حرکات و اعراب
text = ARABIC_DIACRITICS_REGEX.sub('', text)
# ۴. حذف تطویل
text = TATWEEL_REGEX.sub('', text)
# ۵. نگاشت الف‌ها و کاراکترهای هم‌ارز
text = text.translate(ARABIC_NORMALIZATION_MAP)
# ۶. حذف فاصله‌های اضافی مکرر
text = re.sub(r'\s+', ' ', text).strip()
return text
```
---
## ۶. نقشه راه اجرایی پیشنهادی (Recommended Implementation Roadmap)
1. **فاز ۱ — بررسی و تأیید نهایی:**
- تأیید قوانین نرمال‌سازی فوق توسط کارفرما و تیم فنی (به‌ویژه در خصوص تبدیل `ة` به `ه` و `ؤ` به `و`).
2. **فاز ۲ — اضافه کردن ماژول Utility:**
- افزودن فایل `backend/utils/text_normalizer.py` شامل تابع `normalize_for_search`.
3. **فاز ۳ — پیاده‌سازی پایگاه داده:**
- اضافه کردن فیلدهای `search_text` یا `normalized_text` در مدل‌های کلیدی (`Hadis`، `HadisCategory`، `Transmitter`، `HadisCorrection`، `ReferenceBook`).
- تنظیم پر شدن خودکار در `save()`.
- اجرای یک Management Command برای نرمال‌سازی داده‌های قبلی.
4. **فاز ۴ — به‌روزرسانی Queryset های جستجو:**
- اصلاح متدهای `apply_search_filter` در Viewها تا ورودی کاربر را پیش از جستجو نرمال کند.
5. **فاز ۵ — تست و اعتبارسنجی:**
- نوشتن تست‌های خودکار (Unit Tests) برای کلمات چالش‌برانگیز مثل «أنما»، «إِنَّمَا»، «الرَّحْمٰنِ»، «أبو هريرة»، «مسؤول» و اطمینان از نتیجه مثبت در تمامی حالات.

32
scripts/inspect_refs.py

@ -0,0 +1,32 @@
import os
import sys
import django
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
sys.stdout.reconfigure(encoding='utf-8')
os.environ.setdefault('DJANGO_SETTINGS_MODULE', 'config.settings.base')
django.setup()
from apps.hadis.models import HadisReference, InterpretationReference, BookReference, BookEdition, BookVolume
from apps.hadis.models.transmitter import OriginalTextReference
print('=== 5 SAMPLE HADIS REFERENCES ===')
for r in HadisReference.objects.select_related('book_reference', 'edition', 'book_volume')[:5]:
print(f'ID {r.id}: Book={r.book_reference_id}, Edition={r.edition_id}, BookVol={r.book_volume_id}, VolText="{r.volume}", Page="{r.pages}", HadithNo="{r.hadith_number}"')
print('\n=== 5 SAMPLE INTERPRETATION REFERENCES ===')
for r in InterpretationReference.objects.select_related('book_reference', 'edition', 'book_volume')[:5]:
print(f'ID {r.id}: Book={r.book_reference_id}, Edition={r.edition_id}, BookVol={r.book_volume_id}, VolText="{r.volume}", Page="{r.pages}", HadithNo="{r.hadith_number}"')
print('\n=== 5 SAMPLE ORIGINAL TEXT REFERENCES ===')
for r in OriginalTextReference.objects.select_related('book_reference', 'edition', 'book_volume')[:5]:
print(f'ID {r.id}: Book={r.book_reference_id}, Edition={r.edition_id}, BookVol={r.book_volume_id}, VolText="{r.volume}", Page="{r.pages}", HadithNo="{r.hadith_number}"')
# Also check how many BookEditions exist per book
print('\n=== BOOK EDITIONS AND VOLUMES INFO ===')
sample_books = BookReference.objects.all()[:5]
for b in sample_books:
ed_count = b.editions.count()
vol_count = b.volumes.count()
editions = list(b.editions.values('id', 'edition_number', 'publisher'))
print(f"Book ID {b.id} ({b.slug}): {ed_count} editions, {vol_count} volumes. Editions: {editions}")

240
scripts/populate_reference_metadata.py

@ -0,0 +1,240 @@
import os
import sys
import random
import django
# Setup environment
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
sys.stdout.reconfigure(encoding='utf-8')
os.environ.setdefault('DJANGO_SETTINGS_MODULE', 'config.settings.base')
django.setup()
from django.db import transaction
from django.db.models.signals import post_save, post_delete, m2m_changed
from apps.hadis.signals import clear_hadis_cache, clear_hadis_cache_on_m2m, invalidate_hadis_cache
from apps.hadis.models import (
BookReference, BookEdition, BookVolume,
HadisReference, HadisInterpretation, InterpretationReference,
HadisCorrection, CorrectionReference
)
from apps.hadis.models.transmitter import OriginalTextReference
# Disconnect cache signals during bulk population for high performance
post_save.disconnect(clear_hadis_cache)
post_delete.disconnect(clear_hadis_cache)
m2m_changed.disconnect(clear_hadis_cache_on_m2m)
SAMPLE_PUBLISHERS = [
[
{'text': 'دار الحديث', 'language_code': 'fa'},
{'text': 'دار الحديث للطباعة والنشر', 'language_code': 'ar'},
{'text': 'Dar al-Hadith Publications', 'language_code': 'en'}
],
[
{'text': 'دار الكتب الإسلامية', 'language_code': 'fa'},
{'text': 'دار الكتب الإسلامية', 'language_code': 'ar'},
{'text': 'Dar al-Kutub al-Islamiyya', 'language_code': 'en'}
],
[
{'text': 'مؤسسة النشر الإسلامي', 'language_code': 'fa'},
{'text': 'مؤسسة النشر الإسلامي التابعة لجماعة المدرسين', 'language_code': 'ar'},
{'text': 'Islamic Publishing Foundation', 'language_code': 'en'}
],
[
{'text': 'مؤسسة البلاغ', 'language_code': 'fa'},
{'text': 'مؤسسة البلاغ للطباعة والنشر والتوزيع', 'language_code': 'ar'},
{'text': 'Al-Balagh Foundation', 'language_code': 'en'}
],
[
{'text': 'دار الكتب العلمية', 'language_code': 'fa'},
{'text': 'دار الكتب العلمية', 'language_code': 'ar'},
{'text': 'Dar al-Kutub al-Ilmiyya', 'language_code': 'en'}
],
[
{'text': 'مؤسسة آل البيت (ع) لإحياء التراث', 'language_code': 'fa'},
{'text': 'مؤسسة آل البيت عليهم السلام لإحياء التراث', 'language_code': 'ar'},
{'text': 'Aal al-Bayt Institute', 'language_code': 'en'}
],
]
def is_empty_value(val):
if val is None:
return True
s = str(val).strip()
return s in ['', '0', '00', 'None', 'null', 'nan']
def ensure_book_editions_and_volumes():
"""Ensure every BookReference has at least 1 BookEdition and 1 BookVolume."""
all_books = list(BookReference.objects.prefetch_related('editions', 'volumes').all())
print(f"Checking {len(all_books)} books for editions and volumes...", flush=True)
editions_to_create = []
books_needing_editions = []
for book in all_books:
editions = list(book.editions.all())
if not editions:
pub = random.choice(SAMPLE_PUBLISHERS)
editions_to_create.append(BookEdition(
book_reference=book,
publisher=pub,
edition_number='طبعة أولى (1)',
year_of_publication='1441',
number_of_volumes=4
))
books_needing_editions.append(book)
if editions_to_create:
created = BookEdition.objects.bulk_create(editions_to_create)
print(f"Created {len(created)} default editions.", flush=True)
else:
print("All books already have editions.", flush=True)
# Re-fetch or assign volumes
all_books = list(BookReference.objects.prefetch_related('editions', 'volumes').all())
volumes_to_create = []
for book in all_books:
volumes = list(book.volumes.all())
if not volumes:
editions = list(book.editions.all())
ed = editions[0] if editions else None
for vol_num in range(1, 5):
volumes_to_create.append(BookVolume(
book_reference=book,
edition=ed,
title=f"جلد {vol_num}"
))
if volumes_to_create:
created_vols = BookVolume.objects.bulk_create(volumes_to_create)
print(f"Created {len(created_vols)} default volumes.", flush=True)
else:
print("All books already have volumes.", flush=True)
def process_reference_queryset(model_class, model_name, all_books_map):
"""Update edition, book_volume, volume, pages, and hadith_number on all instances using bulk_update."""
qs = model_class.objects.select_related('book_reference', 'edition', 'book_volume').all()
total = qs.count()
print(f"\nProcessing {total} records of {model_name}...", flush=True)
if total == 0:
return
updated_count = 0
all_book_ids = list(all_books_map.keys())
if not all_book_ids:
print("Error: No books available in database!", flush=True)
return
items_to_update = []
has_hadith_number = hasattr(model_class, 'hadith_number')
update_fields = ['book_reference', 'edition', 'book_volume', 'volume', 'pages']
if has_hadith_number:
update_fields.append('hadith_number')
for ref in qs:
changed = False
# 1. Ensure book_reference is attached
book = ref.book_reference
if not book:
chosen_book_id = random.choice(all_book_ids)
book = all_books_map[chosen_book_id]['book']
ref.book_reference = book
changed = True
book_data = all_books_map.get(book.id)
if not book_data:
continue
book_editions = book_data['editions']
book_volumes = book_data['volumes']
# 2. Ensure edition (Publisher) is set
if not ref.edition and book_editions:
ref.edition = random.choice(book_editions)
changed = True
# 3. Ensure book_volume and volume text are set
if not ref.book_volume and book_volumes:
chosen_vol = random.choice(book_volumes)
ref.book_volume = chosen_vol
ref.volume = chosen_vol.title or f"جلد {random.randint(1, 4)}"
changed = True
elif not ref.volume:
ref.volume = f"جلد {random.randint(1, 4)}"
changed = True
# 4. Ensure pages is set to a valid random page number
if is_empty_value(ref.pages):
ref.pages = str(random.randint(15, 480))
changed = True
# 5. Ensure hadith_number is set to a valid random hadith number
if has_hadith_number:
if is_empty_value(ref.hadith_number):
ref.hadith_number = str(random.randint(45, 3450))
changed = True
if changed:
items_to_update.append(ref)
if len(items_to_update) >= 300:
model_class.objects.bulk_update(items_to_update, update_fields)
updated_count += len(items_to_update)
print(f" - Updated {updated_count}/{total} records...", flush=True)
items_to_update = []
if items_to_update:
model_class.objects.bulk_update(items_to_update, update_fields)
updated_count += len(items_to_update)
print(f"Completed {model_name}: {updated_count} / {total} records updated.", flush=True)
def main():
print("==================================================", flush=True)
print("STARTING REFERENCE METADATA POPULATION", flush=True)
print("==================================================", flush=True)
with transaction.atomic():
# Step 1: Ensure all books have editions and volumes
ensure_book_editions_and_volumes()
# Step 2: Cache book relations in memory for fast lookup
all_books_map = {}
for b in BookReference.objects.prefetch_related('editions', 'volumes').all():
all_books_map[b.id] = {
'book': b,
'editions': list(b.editions.all()),
'volumes': list(b.volumes.all()),
}
print(f"Cached {len(all_books_map)} books with their editions and volumes.", flush=True)
# Step 3: Update HadisReference (Arguments)
process_reference_queryset(HadisReference, "HadisReference (Arguments)", all_books_map)
# Step 4: Update InterpretationReference (Interpretations)
process_reference_queryset(InterpretationReference, "InterpretationReference (Interpretations)", all_books_map)
# Step 5: Update OriginalTextReference (Original Texts)
process_reference_queryset(OriginalTextReference, "OriginalTextReference (Original Texts)", all_books_map)
# Step 6: Update CorrectionReference (Corrections)
process_reference_queryset(CorrectionReference, "CorrectionReference (Corrections)", all_books_map)
# Invalidate cache once at the end
print("\nInvalidating API cache...", flush=True)
try:
invalidate_hadis_cache()
print("API cache invalidated successfully.", flush=True)
except Exception as e:
print(f"Cache invalidation note: {e}", flush=True)
print("\n==================================================", flush=True)
print("SUCCESS: ALL REFERENCES SUCCESSFULLY UPDATED!", flush=True)
print("==================================================", flush=True)
if __name__ == '__main__':
main()

135
utils/text_normalizer.py

@ -0,0 +1,135 @@
"""
Comprehensive Arabic & Persian Text Normalization Utility for Search.
Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification,
Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers.
"""
import re
import unicodedata
from typing import Any, Iterable, List, Optional
# 1. Arabic Harakat / Diacritics / Quranic annotations:
# \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc.
# \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن
# \u06D6 - \u06ED : Quranic pause marks, small high signs, etc.
ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]')
# 2. Tatweel / Kashida (ـ)
TATWEEL_REGEX = re.compile(r'\u0640')
# 3. HTML tags remover (e.g. <p>, <br>, <span>, etc.)
HTML_TAGS_REGEX = re.compile(r'<[^>]+>')
# 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.)
ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]')
# 5. Normalization translation map
ARABIC_NORMALIZATION_MAP = str.maketrans({
# Alef variations -> Plain Alef
'أ': 'ا',
'إ': 'ا',
'آ': 'ا',
'ٱ': 'ا',
# Yeh & Alef Maksura -> Standard Yeh
'ي': 'ی',
'ى': 'ی',
'ئ': 'ی',
# Kaf -> Standard Kaf
'ك': 'ک',
# Teh Marbuta & Heh with Yeh -> Standard Heh
'ة': 'ه',
'ۀ': 'ه',
# Waw with Hamza -> Standard Waw
'ؤ': 'و',
# Eastern Arabic and Persian digits -> Standard Latin digits
'٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4',
'٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9',
'۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4',
'۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9',
})
def normalize_for_search(text: Optional[str]) -> str:
"""
Normalizes a text string for fuzzy / invariant Arabic and Persian search:
- Strips HTML tags
- Unicode NFKC normalization
- Removes zero-width and invisible control characters
- Strips all diacritics / tashkeel and dagger alef (\u0670)
- Strips tatweel / kashida (\u0640)
- Unifies all Alef forms (أ, إ, آ, ٱ -> ا)
- Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی)
- Unifies Kaf (ك -> ک)
- Unifies Teh Marbuta (ة, ۀ -> ه)
- Unifies Waw with Hamza (ؤ -> و)
- Converts Arabic & Persian numerals to ASCII (0-9)
- Collapses multiple whitespace characters and lowercases English text
"""
if not text or not isinstance(text, str):
return ""
# 1. Remove HTML tags if present
cleaned = HTML_TAGS_REGEX.sub(' ', text)
# 2. Unicode normalization (NFKC)
cleaned = unicodedata.normalize('NFKC', cleaned)
# 3. Remove zero-width / control characters
cleaned = ZERO_WIDTH_REGEX.sub('', cleaned)
# 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef
cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned)
# 5. Remove Tatweel / Kashida
cleaned = TATWEEL_REGEX.sub('', cleaned)
# 6. Apply character unification map
cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP)
# 7. Lowercase English characters and collapse whitespace
cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower()
return cleaned
def extract_searchable_strings(data: Any) -> List[str]:
"""
Recursively extracts all text values from nested lists, dicts,
or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]).
"""
results: List[str] = []
if data is None:
return results
if isinstance(data, str):
val = data.strip()
if val:
results.append(val)
elif isinstance(data, dict):
for k, v in data.items():
results.extend(extract_searchable_strings(v))
elif isinstance(data, (list, tuple, set)):
for item in data:
results.extend(extract_searchable_strings(item))
return results
def build_search_blob(*components: Any) -> str:
"""
Extracts all text components, normalizes them, and joins them with spaces
into a single indexed search text blob.
"""
all_raw_strings: List[str] = []
for c in components:
all_raw_strings.extend(extract_searchable_strings(c))
# Join and normalize in one pass
joined = " ".join(all_raw_strings)
return normalize_for_search(joined)
Loading…
Cancel
Save