diff --git a/apps/hadis/management/commands/populate_normalized_search.py b/apps/hadis/management/commands/populate_normalized_search.py new file mode 100644 index 0000000..d8b48a4 --- /dev/null +++ b/apps/hadis/management/commands/populate_normalized_search.py @@ -0,0 +1,95 @@ +from django.core.management.base import BaseCommand +from apps.hadis.models import ( + Hadis, + Transmitters, + BookReference, + HadisCategory, + HadisCorrection, + HadisInterpretation +) +from utils.text_normalizer import build_search_blob + + +class Command(BaseCommand): + help = "Backfill normalized_text on Hadis, Transmitters, Categories, BookReferences, Corrections, and Interpretations" + + def handle(self, *args, **options): + self.stdout.write(self.style.SUCCESS("Starting search normalization backfill...")) + + # 1. Hadis + self.stdout.write("1. Normalizing Hadis records...") + hadiths = list(Hadis.objects.all().only('id', 'text', 'title', 'title_narrator', 'translation', 'normalized_text')) + for h in hadiths: + h.normalized_text = build_search_blob( + h.text, + h.title, + h.title_narrator, + h.translation + ) + Hadis.objects.bulk_update(hadiths, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(hadiths)} Hadis records.")) + + # 2. Transmitters + self.stdout.write("2. Normalizing Transmitters records...") + transmitters = list(Transmitters.objects.all().only('id', 'full_name', 'kunya', 'known_as', 'nickname', 'normalized_text')) + for t in transmitters: + t.normalized_text = build_search_blob( + t.full_name, + t.kunya, + t.known_as, + t.nickname + ) + Transmitters.objects.bulk_update(transmitters, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(transmitters)} Transmitters records.")) + + # 3. BookReference + self.stdout.write("3. Normalizing BookReference records...") + books = list(BookReference.objects.select_related('author').all().only('id', 'title', 'description', 'author__name', 'normalized_text')) + for b in books: + author_name = b.author.name if b.author else "" + b.normalized_text = build_search_blob( + b.title, + b.description, + author_name + ) + BookReference.objects.bulk_update(books, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(books)} BookReference records.")) + + # 4. HadisCategory + self.stdout.write("4. Normalizing HadisCategory records...") + categories = list(HadisCategory.objects.all().only('id', 'title', 'description', 'normalized_text')) + for c in categories: + c.normalized_text = build_search_blob( + c.title, + c.description + ) + HadisCategory.objects.bulk_update(categories, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(categories)} HadisCategory records.")) + + # 5. HadisCorrection + self.stdout.write("5. Normalizing HadisCorrection records...") + corrections = list(HadisCorrection.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text')) + for corr in corrections: + corr.normalized_text = build_search_blob( + corr.text, + corr.title, + corr.narrator, + corr.translation + ) + HadisCorrection.objects.bulk_update(corrections, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(corrections)} HadisCorrection records.")) + + # 6. HadisInterpretation + self.stdout.write("6. Normalizing HadisInterpretation records...") + interpretations = list(HadisInterpretation.objects.all().only('id', 'text', 'title', 'narrator', 'translation', 'normalized_text')) + for interp in interpretations: + interp.normalized_text = build_search_blob( + interp.text, + interp.title, + interp.narrator, + interp.translation + ) + HadisInterpretation.objects.bulk_update(interpretations, ['normalized_text'], batch_size=500) + self.stdout.write(self.style.SUCCESS(f" Successfully updated {len(interpretations)} HadisInterpretation records.")) + + self.stdout.write(self.style.SUCCESS("All search normalization fields backfilled successfully!")) diff --git a/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py b/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py new file mode 100644 index 0000000..b7792a0 --- /dev/null +++ b/apps/hadis/migrations/0048_bookreference_normalized_text_hadis_normalized_text_and_more.py @@ -0,0 +1,43 @@ +# Generated by Django 4.2.30 on 2026-10-08 10:37 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('hadis', '0047_transmitteroriginaltext_text_to_textfield'), + ] + + operations = [ + migrations.AddField( + model_name='bookreference', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadis', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadiscategory', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadiscorrection', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='hadisinterpretation', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AddField( + model_name='transmitters', + name='normalized_text', + field=models.TextField(blank=True, db_index=True, null=True, verbose_name='normalized text for search'), + ), + ] diff --git a/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py b/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py new file mode 100644 index 0000000..52ab625 --- /dev/null +++ b/apps/hadis/migrations/0049_alter_bookreference_normalized_text_and_more.py @@ -0,0 +1,43 @@ +# Generated by Django 4.2.30 on 2026-10-08 10:40 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('hadis', '0048_bookreference_normalized_text_hadis_normalized_text_and_more'), + ] + + operations = [ + migrations.AlterField( + model_name='bookreference', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadis', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadiscategory', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadiscorrection', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='hadisinterpretation', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + migrations.AlterField( + model_name='transmitters', + name='normalized_text', + field=models.TextField(blank=True, null=True, verbose_name='normalized text for search'), + ), + ] diff --git a/apps/hadis/models/category.py b/apps/hadis/models/category.py index 03c3739..cb84941 100644 --- a/apps/hadis/models/category.py +++ b/apps/hadis/models/category.py @@ -80,6 +80,7 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel): order = models.IntegerField(default=0, verbose_name=_('order')) xmind_file = models.FileField(upload_to='hadis/xmind_files/', verbose_name=_('xmind file'), null=True, blank=True) slug = models.SlugField(max_length=255, null=True, blank=True,allow_unicode=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) content_type = None language = None language_id = None @@ -111,6 +112,11 @@ class HadisCategory(LowercaseSlugMixin, MPTTModel): slug_source_field = 'title' def save(self, *args, **kwargs): + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.title, + self.description + ) self.full_clean() super().save(*args, **kwargs) diff --git a/apps/hadis/models/hadis.py b/apps/hadis/models/hadis.py index 7407bc2..f3f1e6e 100644 --- a/apps/hadis/models/hadis.py +++ b/apps/hadis/models/hadis.py @@ -182,6 +182,7 @@ class Hadis(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) embedded_in = models.JSONField(default=list, blank=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) def __str__(self): title = None @@ -231,6 +232,15 @@ class Hadis(LowercaseSlugMixin, models.Model): if (old_instance.text != self.text or old_instance.translation != self.translation): self.embedded_in = [] # Reset! + + # Populate normalized_text for fast and accurate Arabic/Persian search + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.title_narrator, + self.translation + ) super().save(*args, **kwargs) @@ -417,6 +427,7 @@ class HadisCorrection(models.Model): updated_at = models.DateTimeField(auto_now=True, verbose_name=_("updated at")) share_link = models.CharField(max_length=255, verbose_name=_('share link'), null=True, blank=True) embedded_in = models.JSONField(default=list, blank=True) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: verbose_name = _("Hadis Correction") @@ -442,6 +453,15 @@ class HadisCorrection(models.Model): old_instance.translation != self.translation): self.embedded_in = [] # Reset! + # Populate normalized_text for fast search + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.narrator, + self.translation + ) + super().save(*args, **kwargs) def get_title(self, lang): @@ -474,6 +494,7 @@ class HadisInterpretation(models.Model): text = models.TextField(verbose_name=_('Original Text')) translation = models.JSONField(verbose_name=_('Translation'), default=list) priority = models.IntegerField(default=0, verbose_name=_('priority')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) created_at = models.DateTimeField(auto_now_add=True) updated_at = models.DateTimeField(auto_now=True) @@ -486,6 +507,15 @@ class HadisInterpretation(models.Model): # 👈 ست کردن اسلاگ تفسیر برابر با اسلاگ کتگوری if self.category and self.category.slug: self.slug = self.category.slug + + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.text, + self.title, + self.narrator, + self.translation + ) + super().save(*args, **kwargs) # --- FOR CORRECTIONS --- diff --git a/apps/hadis/models/reference.py b/apps/hadis/models/reference.py index 55d5ec7..c475d76 100644 --- a/apps/hadis/models/reference.py +++ b/apps/hadis/models/reference.py @@ -143,6 +143,7 @@ class BookReference(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: indexes = [ @@ -157,6 +158,16 @@ class BookReference(LowercaseSlugMixin, models.Model): return self.title[0].get('text') or self.title[0].get('title') or f"Reference {self.id}" return f"Reference {self.id}" + def save(self, *args, **kwargs): + from utils.text_normalizer import build_search_blob + author_name = self.author.name if self.author else "" + self.normalized_text = build_search_blob( + self.title, + self.description, + author_name + ) + super().save(*args, **kwargs) + def _get_json_field(self, field_name: str, lang: Optional[str]=None , fallback: str = "en"): """ Generic getter for JSONField in our [{text, language_code}] format. diff --git a/apps/hadis/models/transmitter.py b/apps/hadis/models/transmitter.py index 49a9f75..9d2d6fa 100644 --- a/apps/hadis/models/transmitter.py +++ b/apps/hadis/models/transmitter.py @@ -207,6 +207,7 @@ class Transmitters(LowercaseSlugMixin, models.Model): created_at = models.DateTimeField(auto_now_add=True, verbose_name=_('created at')) updated_at = models.DateTimeField(auto_now=True, verbose_name=_('updated at')) + normalized_text = models.TextField(blank=True, null=True, verbose_name=_('normalized text for search')) class Meta: indexes = [ @@ -222,6 +223,15 @@ class Transmitters(LowercaseSlugMixin, models.Model): self.legacy_id = None if not self.legacy_number: self.legacy_number = None + + from utils.text_normalizer import build_search_blob + self.normalized_text = build_search_blob( + self.full_name, + self.kunya, + self.known_as, + self.nickname + ) + super().save(*args, **kwargs) @property diff --git a/apps/hadis/views/category.py b/apps/hadis/views/category.py index b72a168..0a0095a 100644 --- a/apps/hadis/views/category.py +++ b/apps/hadis/views/category.py @@ -477,7 +477,10 @@ class CategoriesView(ListAPIView): ) search_query = (self.request.query_params.get('search') or '').strip() if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(slug__icontains=search_query) | Q(description__icontains=search_query) @@ -514,7 +517,10 @@ class CategoriesBySectView(ListAPIView): ) search_query = (self.request.query_params.get('search') or '').strip() if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(slug__icontains=search_query) | Q(description__icontains=search_query) diff --git a/apps/hadis/views/hadis.py b/apps/hadis/views/hadis.py index f19d920..9ecaea5 100644 --- a/apps/hadis/views/hadis.py +++ b/apps/hadis/views/hadis.py @@ -361,12 +361,18 @@ class HadisListView(ListAPIView): # 👇 1. Apply Search Filter search_query = self.request.query_params.get('search', None) if search_query: - search_conditions = Q(text__icontains=search_query) + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) + + search_conditions = Q(normalized_text__icontains=norm_q) + search_conditions |= Q(text__icontains=search_query) search_conditions |= Q(title__icontains=search_query) search_conditions |= Q(title_narrator__icontains=search_query) search_conditions |= Q(translation__icontains=search_query) search_conditions |= Q(tags__title__icontains=search_query) + search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q) search_conditions |= Q(references__book_reference__title__icontains=search_query) + search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q) search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query) search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query) queryset = queryset.filter(search_conditions) @@ -625,21 +631,25 @@ class HadisMainListView(ListAPIView): def apply_search_filter(self, queryset, search_query): """ - Apply search filter across multiple fields including JSONFields. - Searches in: title, title_narrator, text, translation, tags, book references, transmitters + Apply search filter across multiple fields including normalized text and JSONFields. + Searches in: normalized_text, title, title_narrator, text, translation, tags, book references, transmitters """ from django.db.models import Q + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) - # Basic search conditions - search_conditions = Q(text__icontains=search_query) + # Primary normalized search (matches Arabic/Persian/English invariantly) + search_conditions = Q(normalized_text__icontains=norm_q) - # For JSONFields, search in the JSON string representation - # This will find matches in the "text" values within the JSON arrays + # Fallback raw search + search_conditions |= Q(text__icontains=search_query) search_conditions |= Q(title__icontains=search_query) search_conditions |= Q(title_narrator__icontains=search_query) search_conditions |= Q(translation__icontains=search_query) search_conditions |= Q(tags__title__icontains=search_query) + search_conditions |= Q(references__book_reference__normalized_text__icontains=norm_q) search_conditions |= Q(references__book_reference__title__icontains=search_query) + search_conditions |= Q(transmitters__transmitter__normalized_text__icontains=norm_q) search_conditions |= Q(transmitters__transmitter__full_name__icontains=search_query) search_conditions |= Q(transmitters__transmitter__known_as__icontains=search_query) @@ -919,7 +929,11 @@ class CategoryHadisCorrectionsView(ListAPIView): # Optional search query filter search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) search_conditions = ( + Q(normalized_text__icontains=norm_q) | + Q(hadis__normalized_text__icontains=norm_q) | Q(title__icontains=search_query) | Q(text__icontains=search_query) | Q(narrator__icontains=search_query) | diff --git a/apps/hadis/views/reference.py b/apps/hadis/views/reference.py index 11b4e1e..3a36c96 100644 --- a/apps/hadis/views/reference.py +++ b/apps/hadis/views/reference.py @@ -36,15 +36,13 @@ class BookReferencesView(ListAPIView): def apply_search_filter(self, queryset, search_query): """ - Apply search filter across book titles (JSONField). - Searches in: title + Apply search filter across book titles (JSONField) and normalized text. + Searches in: normalized_text, title """ - # Search conditions - search_conditions = Q() + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) - # For JSONFields, search in the JSON string representation - # This will find matches in the "text" values within the JSON arrays - search_conditions |= Q(title__icontains=search_query) + search_conditions = Q(normalized_text__icontains=norm_q) | Q(title__icontains=search_query) return queryset.filter(search_conditions) diff --git a/apps/hadis/views/reference_v2.py b/apps/hadis/views/reference_v2.py index d2abb24..d3b2157 100644 --- a/apps/hadis/views/reference_v2.py +++ b/apps/hadis/views/reference_v2.py @@ -163,7 +163,10 @@ class BookReferenceV2ListView(generics.ListAPIView): search = self.request.query_params.get('search') if search: from django.db.models import Q + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search) qs = qs.filter( + Q(normalized_text__icontains=norm_q) | Q(title__icontains=search) | Q(description__icontains=search) | Q(author__name__icontains=search) diff --git a/apps/hadis/views/transmitter.py b/apps/hadis/views/transmitter.py index e8a8109..93f848b 100644 --- a/apps/hadis/views/transmitter.py +++ b/apps/hadis/views/transmitter.py @@ -42,7 +42,10 @@ class TransmitterView(ListAPIView): # 1. Apply search filter (Searching across name-related JSONFields, slug, legacy_id) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) q_filter = ( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | @@ -421,7 +424,10 @@ class NarratorTeachersView(ListAPIView): search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | @@ -471,7 +477,10 @@ class NarratorStudentsView(ListAPIView): search_query = self.request.query_params.get('search', None) if search_query: + from utils.text_normalizer import normalize_for_search + norm_q = normalize_for_search(search_query) queryset = queryset.filter( + Q(normalized_text__icontains=norm_q) | Q(full_name__icontains=search_query) | Q(kunya__icontains=search_query) | Q(known_as__icontains=search_query) | diff --git a/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md b/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md new file mode 100644 index 0000000..540011b --- /dev/null +++ b/docs/ARABIC_SEARCH_NORMALIZATION_GUIDE.md @@ -0,0 +1,283 @@ +# دستورالعمل جامع نرمالسازی جستجو برای متون عربی و اسلامی +### (Comprehensive Arabic & Persian Search Normalization Specification) + +--- + +## ۱. مقدمه و بیان مسئله (Executive Summary & Problem Statement) + +در پایگاههای داده و سامانههای متنی اسلامی و حدیثی، جستجوی متنی با چالشهای بنیادین زبانشناختی و فنی مواجه است: + +1. **گوناگونی رسمالخط و کدگذاری یونیکد (Unicode Homoglyphs):** نویسههایی مانند الف همزهدار (`أ`، `إ`)، الف ممدوده (`آ`)، الف وصل (`ٱ`) و الف ساده (`ا`) دارای کدهای یونیکد کاملاً متفاوتی هستند. +2. **اعراب و اعجام (Harakat / Tashkeel):** متون احادیث و اسناد دینی معمولاً با حرکتگذاری کامل (مانند `إِنَّمَا الأَعْمَالُ`) در پایگاه داده ذخیره شدهاند، در حالی که کاربران جستجوهای خود را بدون اعراب (`انما الاعمال` یا `انما`) تایپ میکنند. در مقایسههای متنی پیشفرض دیتابیس (`LIKE` یا `ILIKE` و `icontains`)، وجود اعراب در میان حروف کلمه باعث میشود کلمه کاربر تطبیق داده نشود و نتیجه «هیچ رکوردی یافت نشد» برگردد. +3. **تداخل کاراکترهای فارسی و عربی:** در کیبوردهای کاربران (موبایل و دسکتاپ)، حروفی چون «ی» و «ي»، «ک» و «ك»، «ه» و «ة» مدام جابهجا میشوند. +4. **علائم قرآنی و وقوف:** نمادهایی مانند «الف خنجری» (`ٰ` مانند `رَحْمٰنِ`)، علائم وقف و نمادهای تعظیم اگر نرمال نشوند، جستجو را مسدود میکنند. + +> **هدف:** هر متنی که کاربر با هر شکلی از کیبورد (با اعراب، بدون اعراب، با همزه یا بدون همزه، فارسی یا عربی) جستجو کرد، سیستم باید دقیقاً مفهوم آن را درک کرده و تمامی رکوردهای منطبق را بدون وابستگی به شکل ظاهری نگارش پیدا کند. + +--- + +## ۲. دستهبندی جامع کاراکترها و حالات نرمالسازی (Normalization Taxonomy) + +### ۲.۱. خانواده الفها و همزهها (Alef & Hamza Variants) +الف در رسمالخط عربی دارای حالتهای متعددی است که در جستجو باید همگی یکپارچه شوند: + +| نویسه اصلی | نام یونیکد | کد یونیکد | هدف نرمالسازی | مثال جستجو | مثال در دیتابیس | +| :--- | :--- | :--- | :--- | :--- | :--- | +| **أ** | Arabic Letter Alef With Hamza Above | `U+0623` | **ا** (`U+0627`) | أنما | انما / إِنَّمَا | +| **إ** | Arabic Letter Alef With Hamza Below | `U+0625` | **ا** (`U+0627`) | إسناد | اسناد | +| **آ** | Arabic Letter Alef With Madda Above | `U+0622` | **ا** (`U+0627`) | آثار | اثار | +| **ٱ** | Arabic Letter Alef Wasla | `U+0671` | **ا** (`U+0627`) | ٱمرؤ | امرؤ | +| **ا** | Arabic Letter Alef (Bare) | `U+0627` | **ا** (`U+0627`) | اعمال | أعمال | +| **ٴ** | Arabic Letter High Hamza | `U+0674` | *حذف* یا **ا** | — | — | + +#### حالات دیگر همزهها (Other Hamza Forms): +- **همزه روی واو (`ؤ` - `U+0624`):** در جستجوی آسانگیر (Lenient)، همزه روی واو به `و` (`U+0648`) نرمال میشود تا کاربر با سرچ «مومن» بتواند «مؤمن» را پیدا کند یا بالعکس («مسؤول» / «مسئول»). +- **همزه روی یاء / نبره (`ئ` - `U+0626`):** به `ی` / `ي` تبدیل میشود تا کلماتی چون «قائل»، «قایل»، «هیئة»، «هيئة» همارز شوند. +- **همزه تنها روی خط (`ء` - `U+0621`):** حذف یا تبدیل به فاصله در صورت نیاز. + +--- + +### ۲.۲. خانواده یاء و الف مقصوره (Yeh & Alef Maksura) +یکی از پرتکرارترین خطاها در جستجوی عربی و فارسی مربوط به حرف «ی» است: + +| نویسه اصلی | نام | کد یونیکد | رفتار نرمالسازی | مثال | +| :--- | :--- | :--- | :--- | :--- | +| **ي** | یاء عربی دو نقطه | `U+064A` | تبدیل به نویسه واحد مبنا (مثلاً `ی` یا `ي`) | علي / علی | +| **ى** | الف مقصوره عربی (بینقطه) | `U+0649` | تبدیل به `ی` یا `ا` بر اساس سیاست پروژه | موسی / موسي / موسى | +| **ی** | یای فارسی بدون نقطه | `U+06CC` | تبدیل به نویسه واحد مبنا | حدیث / حديث | +| **ئ** | یاء با همزه | `U+0626` | تبدیل به نویسه واحد مبنا | بئر / بیر | + +> **نکته تخصصی در متون دینی:** الف مقصوره (`ى` در انتهای کلماتی چون «حتى»، «إلى»، «موسى») در کیبورد کاربران گاهی با `ی`، گاهی با `ي` و حتی گاهی به اشتباه با `ا` («حتا») نوشته میشود. نرمالسازی `[ي ى ی ئ]` به یک نویسه پایدار، پوشش جستجو را به ۱۰۰٪ میرساند. + +--- + +### ۲.۳. کاف عربی و فارسی (Kaf Normalization) +- **ك** (کاف عربی با نشان همزه/کاف کوچک: `U+0643`) +- **ک** (کاف فارسی سرکشدار: `U+06A9`) +- **قاعده:** تبدیل هر دو به یک فرم استاندارد (مثلاً `ك` برای متون عربی یا `ک`). + +--- + +### ۲.۴. تاء مربوطه و هاء (Teh Marbuta & Heh) +- **ة** (تاء مربوطه: `U+0629`) +- **ه** (هاء: `U+0647`) +- **ۀ** (هاء با همزه: `U+06C0`) +- **تحلیل رفتاری:** کاربران در تایپ سریع اسامی یا اصطلاحات اغلب تاء مربوطه را با هاء جابهجا میزنند: + - «معاویه» ↔ «معاوية» + - «فاطمه» ↔ «فاطمة» + - «صحابه» ↔ «صحابة» + - «رواة» ↔ «رواه» +- **قاعده سرچ نرمال:** برای جستجوی متنی، `ة` به `ه` تبدیل میشود تا تفاوت نگارشی کاربر باعث حذف رکورد نشود. + +--- + +### ۲.۵. حرکات، اعراب و تنوینها (Tashkeel / Harakat / Diacritics) - حیاتیترین بخش +تمام حرکات زیر باید از متن ورودی و از نسخه ایندکسشده جستجو **کاملاً حذف شوند**: + +| نام حرکت | علامت | کد یونیکد | اثر در جستجوی خام دیتابیس | +| :--- | :---: | :---: | :--- | +| **فتحه (Fatha)** | َ | `U+064E` | مانع تطابق متن ساده میشود | +| **ضمه (Damma)** | ُ | `U+064F` | مانع تطابق متن ساده میشود | +| **کسره (Kasra)** | ِ | `U+0650` | مانع تطابق متن ساده میشود | +| **تنوین نصب (Fathatan)** | ً | `U+064B` | مانع تطابق | +| **تنوین رفع (Dammatan)** | ٌ | `U+064C` | مانع تطابق | +| **تنوین جر (Kasratan)** | ٍ | `U+064D` | مانع تطابق | +| **سکون (Sukun)** | ْ | `U+0652` | مانع تطابق | +| **تشدید (Shadda)** | ّ | `U+0651` | مانع تطابق کلمات دارای تشدید | +| **الف خنجری (Dagger Alef)** | ٰ | `U+0670` | **بسیار خطرناک:** در «رَحْمٰنِ»، «إِلٰهَ»، «هٰذَا»، «إِسْمٰعِيل» اگر حذف نشود، سرچ «رحمن» هرگز «رحمٰن» را پیدا نمیکند! | +| **مده (Maddah)** | ٓ | `U+0653` | مانع تطابق | +| **همزه فوقانی اعرابی** | ٔ | `U+0654` | مانع تطابق | +| **همزه تحتانی اعرابی** | ٕ | `U+0655` | مانع تطابق | + +--- + +### ۲.۶. کشیدگی، تطویل و نشانههای نامرئی (Tatweel & Invisible Characters) +- **تطویل / کشیده (`ـ` - `U+0640`):** برای تنظیم طول خطوط در متون کهن یا زیبایی متنی به کار میرود (مثل `صـــــراط` یا `رســــول`). این کاراکتر باید به کلی حذف شود. +- **نیمفاصله (Zero-Width Non-Joiner - `U+200C`):** در متون فارسی و اسامی ترکیبی وجود دارد؛ باید به فاصله عادی یا حذف کامل تبدیل شود. +- **اتصالدهنده مجازی (ZWJ - `U+200D`):** باید حذف شود. +- **نشانگرهای جهت یونیکد (LRM `U+200E` و RLM `U+200F`):** باید کاملاً حذف شوند. + +--- + +### ۲.۷. نشانهها و نمادهای وقوف قرآنی و مذهبی (Quranic Symbols & Waqf Marks) +در متون روایی و قرآنی نمادهای ویژهای در یونیکد ذخیره میشوند که باید در لایه سرچ پالایش گردند: +- علائم وقف قرآنی: `ۖ` (`U+06D6`), `ۗ` (`U+06D7`), `ۘ` (`U+06D8`), `ۙ` (`U+06D9`), `ۚ` (`U+06DA`), `ۛ` (`U+06DB`), `ۜ` (`U+06DC`), `` (`U+06DD`), `۞` (`U+06DE`), `۟` (`U+06DF`), `۠` (`U+06E0`), `ۡ` (`U+06E1`), `ۢ` (`U+06E2`), `ۣ` (`U+06E3`), `ۤ` (`U+06E4`). +- نمادهای لیگچر مذهبی: `ﷺ` (`U+FDFA`), `ﷻ` (`U+FDFB`), `﷽` (`U+FDFD`), `ؑ` (`U+0611`). +- پرانتزها و براکتهای قرآنی و نقلقول: `﴿`، `﴾`، `«`، `»`، `[`، `]`، `(`، `)`. + +--- + +### ۲.۸. ارقام و اعداد (Digits) +- ارقام عربی-مشرقی: `[٠, ١, ٢, ٣, ٤, ٥, ٦, ٧, ٨, ٩]` +- ارقام فارسی: `[۰, ۱, ۲, ۳, ۴, ۵, ۶, ۷, ۸, ۹]` +- ارقام استاندارد لاتین: `[0, 1, 2, 3, 4, 5, 6, 7, 8, 9]` +- **قاعده:** تبدیل تمام ارقام به ارقام استاندارد (0-9) تا سرچ شماره حدیث یا جلد و صفحه فارغ از نوع کیبورد عدد را پیدا کند. + +--- + +## ۳. ماتریس مقایسهای سناریوهای سرچ (Search Scenario Matrix) + +| ورودی کاربر در سرچ | متن در دیتابیس | وضعیت جستجوی فعلی (خام) | وضعیت پس از نرمالسازی | +| :--- | :--- | :---: | :---: | +| `انما` | `أنما الأعمال بالنيات` | ❌ No result (تفاوت `ا` و `أ`) | ✅ منطبق و پیدا میشود | +| `انما` | `إِنَّمَا الأَعْمَالُ بِالنِّيَّاتِ` | ❌ No result (وجود کسره، تشدید، فتحه) | ✅ منطبق و پیدا میشود | +| `أبو هريرة` | `ابو هريره` | ❌ No result (تفاوت `أ/ا` و `ة/ه`) | ✅ منطبق و پیدا میشود | +| `صحیح بخاری` | `صَحِيحُ الْبُخَارِيِّ` | ❌ No result (تفاوت اعراب، `ی/ي`) | ✅ منطبق و پیدا میشود | +| `رحمن` | `الرَّحْمٰنِ الرَّحِيمِ` | ❌ No result (وجود الف خنجری `ٰ`) | ✅ منطبق و پیدا میشود | +| `مومن` | `إِنَّمَا الْمُؤْمِنُونَ إِخْوَةٌ` | ❌ No result (تفاوت `و` با `ؤ`) | ✅ منطبق و پیدا میشود | +| `صراط` | `صــــراط الذين` | ❌ No result (وجود کشیده `ـ`) | ✅ منطبق و پیدا میشود | +| `حدیث ۱۱۰` | `حديث 110` یا `حديث ١١٠` | ❌ No result (تفاوت ارقام) | ✅ منطبق و پیدا میشود | + +--- + +## ۴. گزینهها و معماری پیادهسازی فنی در Django و PostgreSQL + +برای اعمال این نرمالسازی در سیستم، سه رویکرد معماری وجود دارد: + +### 🟢 گزینه اول: الگوی ستون جستجوی نرمالشده (Normalized Shadow Column / Search Column) — **[رویکرد پیشنهادی و استاندارد]** + +در این الگو، متن اصلی برای نمایش دستنخورده باقی میماند (تا اعراب و زیبایی اصیل آن در UI حفظ شود)، اما یک ستون متنی نرمالشده در کنار آن ایجاد و ایندکسگذاری میشود: + +1. **در مدلها (`Hadis`، `HadisCategory`، `Transmitter` و ...):** + - افزودن فیلد `normalized_text = models.TextField(blank=True, db_index=True)` یا استفاده از `django.contrib.postgres.search.SearchVector`. + - در متد `save()` مدل، متن اصلی از تابع نرمالساز عبور کرده و فیلد نرمالشده به صورت خودکار پر میشود. + - ایجاد یک اسکریپت ساده migration برای پر کردن یکباره مقادیر رکوردهای موجود. +2. **در لایه Queryset / View:** + - عبارت سرچ کاربر (`search_query`) توسط همان تابع پایتون نرمالسازی میشود: `normalized_q = normalize_text(search_query)`. + - جستجو روی ستون `normalized_text__icontains=normalized_q` انجام میشود. +3. **مزایا:** + - **فوقالعاده سریع (High Performance):** دیتابیس مستقیماً روی ستون ایندکسشده کوئری میزند بدون اینکه در هر ریکوئست تابع یا رجکس سنگین روی میلیونها کاراکتر اجرا شود. + - **سادگی و پایداری:** سازگاری کامل با معماری فعلی Django بدون نیاز به نصب اکستنشنهای پیچیده C در دیتابیس سرور. + - **دقت ۱۰۰٪:** تضمین میکند که منطق سمت پایتون در هر دو طرف ذخیره و جستجو دقیقاً یکی است. + +--- + +### 🟡 گزینه دوم: تابع پایگاه داده در سطح PostgreSQL (Database-Level Stored Function & Functional Index) + +1. ایجاد یک تابع PL/pgSQL در PostgreSQL (مثلاً `fn_normalize_arabic(text)`). +2. ساخت ایندکس تابعی: + ```sql + CREATE INDEX idx_hadis_normalized_text ON hadis_hadis (fn_normalize_arabic(text)); + ``` +3. در جنگو با استفاده از `Func` یا Raw SQL: + ```python + queryset.filter(Q(normalized_text_func__icontains=normalize_text(query))) + ``` +4. **مزایا:** عدم نیاز به ذخیره دیتای مضاعف در ستون جداگانه. +5. **معایب:** وابستگی شدید به دیتابیس، سختی مایگریشن در محیطهای توسعه و تست SQLite/Docker، و پیچیدگی نگهداری لاجیک در SQL. + +--- + +### 🔴 گزینه سوم: استفاده از Regex در زمان کوئری (Query-time Regex) + +1. تبدیل هر حرف از کلمه سرچ به یک گروه رجکس؛ مثلاً تبدیل `انما` به: + `[اأإآٱ][ًٌٍَُِّْٰ]*ن[ًٌٍَُِّْٰ]*م[ًٌٍَُِّْٰ]*[اأإآٱ]` +2. ارسال به دیتابیس با `text__iregex=pattern`. +3. **معایب:** + - **بسیار کند:** دیتابیس نمیتواند از هیچ ایندکسی استفاده کند (Full Table Scan با Regex Engine). + - با افزایش تعداد احادیث و اسناد، پاسخ سرور از چند میلیثانیه به چند ثانیه افزایش مییابد و بار سرور را به شدت بالا میبرد. + +--- + +## ۵. کد مرجع پایتون برای تابع نرمالسازی (Python Reference Implementation) + +این تابع کاملترین و بهینهترین پیادهسازی منطبق با استاندارد Unicode Consortium برای متون عربی و فارسی است: + +```python +import re +import unicodedata + +# 1. حرکات، اعراب، تنوینها، تشدید، سکون و الف خنجری +# شامل بازه U+064B تا U+065F و الف مقصوره بالایی U+0670 +ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]') + +# 2. کاراکتر کشیدگی / تطویل +TATWEEL_REGEX = re.compile(r'\u0640') + +# 3. جدول نگاشت الفها و کاراکترهای چندشکلی +ARABIC_NORMALIZATION_MAP = str.maketrans({ + # انواع الف به الف ساده + 'أ': 'ا', + 'إ': 'ا', + 'آ': 'ا', + 'ٱ': 'ا', + + # انواع یاء و الف مقصوره به یای استاندارد + 'ي': 'ی', + 'ى': 'ی', + 'ئ': 'ی', + + # کاف عربی به کاف یکسان + 'ك': 'ک', + + # تاء مربوطه و هاء + 'ة': 'ه', + 'ۀ': 'ه', + + # واو همزهدار + 'ؤ': 'و', + + # ارقام عربی مشرقی و فارسی به ارقام استاندارد + '٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4', + '٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9', + '۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4', + '۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9', +}) + +# 4. کاراکترهای کنترلی و نامرئی (Zero-width spaces, LRM, RLM) +ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]') + +def normalize_for_search(text: str) -> str: + """ + متن ورودی را بر اساس قواعد استاندارد جستجوی متون عربی و اسلامی نرمالسازی میکند: + 1. حذف کاراکترهای کنترلی پنهان و نیمفاصلههای نامتعارف + 2. حذف کامل تمام اعرابها، حرکات، تشدید، تنوینها و الف خنجری (Tashkeel) + 3. حذف علامت کشیدگی (تطویل / کشیده) + 4. یکسانسازی الفها (أ، إ، آ، ٱ -> ا) + 5. یکسانسازی یاء و الف مقصوره (ي، ى، ئ -> ی) + 6. یکسانسازی کاف (ك -> ک) + 7. یکسانسازی تاء مربوطه (ة -> ه) + 8. یکسانسازی واو همزهدار (ؤ -> و) + 9. تبدیل ارقام عربی و فارسی به ارقام استاندارد + 10. یکپارچهسازی فاصلههای خالی چندگانه + """ + if not text or not isinstance(text, str): + return "" + + # ۱. نرمالسازی فرم یونیکد (NFKC) + text = unicodedata.normalize('NFKC', text) + + # ۲. حذف کاراکترهای نامرئی + text = ZERO_WIDTH_REGEX.sub('', text) + + # ۳. حذف حرکات و اعراب + text = ARABIC_DIACRITICS_REGEX.sub('', text) + + # ۴. حذف تطویل + text = TATWEEL_REGEX.sub('', text) + + # ۵. نگاشت الفها و کاراکترهای همارز + text = text.translate(ARABIC_NORMALIZATION_MAP) + + # ۶. حذف فاصلههای اضافی مکرر + text = re.sub(r'\s+', ' ', text).strip() + + return text +``` + +--- + +## ۶. نقشه راه اجرایی پیشنهادی (Recommended Implementation Roadmap) + +1. **فاز ۱ — بررسی و تأیید نهایی:** + - تأیید قوانین نرمالسازی فوق توسط کارفرما و تیم فنی (بهویژه در خصوص تبدیل `ة` به `ه` و `ؤ` به `و`). +2. **فاز ۲ — اضافه کردن ماژول Utility:** + - افزودن فایل `backend/utils/text_normalizer.py` شامل تابع `normalize_for_search`. +3. **فاز ۳ — پیادهسازی پایگاه داده:** + - اضافه کردن فیلدهای `search_text` یا `normalized_text` در مدلهای کلیدی (`Hadis`، `HadisCategory`، `Transmitter`، `HadisCorrection`، `ReferenceBook`). + - تنظیم پر شدن خودکار در `save()`. + - اجرای یک Management Command برای نرمالسازی دادههای قبلی. +4. **فاز ۴ — بهروزرسانی Queryset های جستجو:** + - اصلاح متدهای `apply_search_filter` در Viewها تا ورودی کاربر را پیش از جستجو نرمال کند. +5. **فاز ۵ — تست و اعتبارسنجی:** + - نوشتن تستهای خودکار (Unit Tests) برای کلمات چالشبرانگیز مثل «أنما»، «إِنَّمَا»، «الرَّحْمٰنِ»، «أبو هريرة»، «مسؤول» و اطمینان از نتیجه مثبت در تمامی حالات. diff --git a/utils/text_normalizer.py b/utils/text_normalizer.py new file mode 100644 index 0000000..d4e6b91 --- /dev/null +++ b/utils/text_normalizer.py @@ -0,0 +1,135 @@ +""" +Comprehensive Arabic & Persian Text Normalization Utility for Search. +Handles Tashkeel (diacritics), Alef/Hamza unification, Yeh/Kaf unification, +Teh Marbuta, Tatweel, Dagger Alef, Quranic annotations, and numbers. +""" + +import re +import unicodedata +from typing import Any, Iterable, List, Optional + +# 1. Arabic Harakat / Diacritics / Quranic annotations: +# \u064B - \u065F : Fathatan, Dammatan, Kasratan, Fatha, Damma, Kasra, Shadda, Sukun, etc. +# \u0670 : Dagger Alef (Superscript Alef) e.g. رحمن vs رَحْمٰن +# \u06D6 - \u06ED : Quranic pause marks, small high signs, etc. +ARABIC_DIACRITICS_REGEX = re.compile(r'[\u064B-\u065F\u0670\u06D6-\u06ED]') + +# 2. Tatweel / Kashida (ـ) +TATWEEL_REGEX = re.compile(r'\u0640') + +# 3. HTML tags remover (e.g.
,
, , etc.)
+HTML_TAGS_REGEX = re.compile(r'<[^>]+>')
+
+# 4. Invisible / Control / Zero-width characters (ZWNJ, ZWJ, LRM, RLM, etc.)
+ZERO_WIDTH_REGEX = re.compile(r'[\u200B-\u200F\u202A-\u202E\uFEFF]')
+
+# 5. Normalization translation map
+ARABIC_NORMALIZATION_MAP = str.maketrans({
+ # Alef variations -> Plain Alef
+ 'أ': 'ا',
+ 'إ': 'ا',
+ 'آ': 'ا',
+ 'ٱ': 'ا',
+
+ # Yeh & Alef Maksura -> Standard Yeh
+ 'ي': 'ی',
+ 'ى': 'ی',
+ 'ئ': 'ی',
+
+ # Kaf -> Standard Kaf
+ 'ك': 'ک',
+
+ # Teh Marbuta & Heh with Yeh -> Standard Heh
+ 'ة': 'ه',
+ 'ۀ': 'ه',
+
+ # Waw with Hamza -> Standard Waw
+ 'ؤ': 'و',
+
+ # Eastern Arabic and Persian digits -> Standard Latin digits
+ '٠': '0', '١': '1', '٢': '2', '٣': '3', '٤': '4',
+ '٥': '5', '٦': '6', '٧': '7', '٨': '8', '٩': '9',
+ '۰': '0', '۱': '1', '۲': '2', '۳': '3', '۴': '4',
+ '۵': '5', '۶': '6', '۷': '7', '۸': '8', '۹': '9',
+})
+
+
+def normalize_for_search(text: Optional[str]) -> str:
+ """
+ Normalizes a text string for fuzzy / invariant Arabic and Persian search:
+ - Strips HTML tags
+ - Unicode NFKC normalization
+ - Removes zero-width and invisible control characters
+ - Strips all diacritics / tashkeel and dagger alef (\u0670)
+ - Strips tatweel / kashida (\u0640)
+ - Unifies all Alef forms (أ, إ, آ, ٱ -> ا)
+ - Unifies Yeh and Alef Maksura (ي, ى, ئ -> ی)
+ - Unifies Kaf (ك -> ک)
+ - Unifies Teh Marbuta (ة, ۀ -> ه)
+ - Unifies Waw with Hamza (ؤ -> و)
+ - Converts Arabic & Persian numerals to ASCII (0-9)
+ - Collapses multiple whitespace characters and lowercases English text
+ """
+ if not text or not isinstance(text, str):
+ return ""
+
+ # 1. Remove HTML tags if present
+ cleaned = HTML_TAGS_REGEX.sub(' ', text)
+
+ # 2. Unicode normalization (NFKC)
+ cleaned = unicodedata.normalize('NFKC', cleaned)
+
+ # 3. Remove zero-width / control characters
+ cleaned = ZERO_WIDTH_REGEX.sub('', cleaned)
+
+ # 4. Remove all Arabic diacritics / Tashkeel / Dagger Alef
+ cleaned = ARABIC_DIACRITICS_REGEX.sub('', cleaned)
+
+ # 5. Remove Tatweel / Kashida
+ cleaned = TATWEEL_REGEX.sub('', cleaned)
+
+ # 6. Apply character unification map
+ cleaned = cleaned.translate(ARABIC_NORMALIZATION_MAP)
+
+ # 7. Lowercase English characters and collapse whitespace
+ cleaned = re.sub(r'\s+', ' ', cleaned).strip().lower()
+
+ return cleaned
+
+
+def extract_searchable_strings(data: Any) -> List[str]:
+ """
+ Recursively extracts all text values from nested lists, dicts,
+ or strings (e.g., multilingual JSONFields like [{'text': '...', 'language_code': 'ar'}]).
+ """
+ results: List[str] = []
+
+ if data is None:
+ return results
+
+ if isinstance(data, str):
+ val = data.strip()
+ if val:
+ results.append(val)
+ elif isinstance(data, dict):
+ for k, v in data.items():
+ results.extend(extract_searchable_strings(v))
+ elif isinstance(data, (list, tuple, set)):
+ for item in data:
+ results.extend(extract_searchable_strings(item))
+
+ return results
+
+
+def build_search_blob(*components: Any) -> str:
+ """
+ Extracts all text components, normalizes them, and joins them with spaces
+ into a single indexed search text blob.
+ """
+ all_raw_strings: List[str] = []
+ for c in components:
+ all_raw_strings.extend(extract_searchable_strings(c))
+
+ # Join and normalize in one pass
+ joined = " ".join(all_raw_strings)
+ return normalize_for_search(joined)