versed-pdf 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. versed_pdf-1.1.0/LICENSE +21 -0
  2. versed_pdf-1.1.0/PKG-INFO +73 -0
  3. versed_pdf-1.1.0/README.md +45 -0
  4. versed_pdf-1.1.0/pyproject.toml +47 -0
  5. versed_pdf-1.1.0/setup.cfg +4 -0
  6. versed_pdf-1.1.0/src/versed/__init__.py +144 -0
  7. versed_pdf-1.1.0/src/versed/_arabic.py +213 -0
  8. versed_pdf-1.1.0/src/versed/_data/qcf_mapping.db +0 -0
  9. versed_pdf-1.1.0/src/versed/_data/transliterations.json +45 -0
  10. versed_pdf-1.1.0/src/versed/arabic.py +110 -0
  11. versed_pdf-1.1.0/src/versed/classify.py +211 -0
  12. versed_pdf-1.1.0/src/versed/cli.py +296 -0
  13. versed_pdf-1.1.0/src/versed/detect.py +118 -0
  14. versed_pdf-1.1.0/src/versed/extract.py +296 -0
  15. versed_pdf-1.1.0/src/versed/filtering.py +74 -0
  16. versed_pdf-1.1.0/src/versed/health.py +174 -0
  17. versed_pdf-1.1.0/src/versed/honorifics.py +396 -0
  18. versed_pdf-1.1.0/src/versed/layout.py +457 -0
  19. versed_pdf-1.1.0/src/versed/markdown.py +166 -0
  20. versed_pdf-1.1.0/src/versed/py.typed +0 -0
  21. versed_pdf-1.1.0/src/versed/qcf.py +465 -0
  22. versed_pdf-1.1.0/src/versed/repair.py +134 -0
  23. versed_pdf-1.1.0/src/versed/routing.py +230 -0
  24. versed_pdf-1.1.0/src/versed/types.py +126 -0
  25. versed_pdf-1.1.0/src/versed_pdf.egg-info/PKG-INFO +73 -0
  26. versed_pdf-1.1.0/src/versed_pdf.egg-info/SOURCES.txt +38 -0
  27. versed_pdf-1.1.0/src/versed_pdf.egg-info/dependency_links.txt +1 -0
  28. versed_pdf-1.1.0/src/versed_pdf.egg-info/entry_points.txt +2 -0
  29. versed_pdf-1.1.0/src/versed_pdf.egg-info/requires.txt +10 -0
  30. versed_pdf-1.1.0/src/versed_pdf.egg-info/top_level.txt +1 -0
  31. versed_pdf-1.1.0/tests/test_arabic.py +35 -0
  32. versed_pdf-1.1.0/tests/test_classify.py +41 -0
  33. versed_pdf-1.1.0/tests/test_cli.py +59 -0
  34. versed_pdf-1.1.0/tests/test_detect.py +72 -0
  35. versed_pdf-1.1.0/tests/test_extract.py +84 -0
  36. versed_pdf-1.1.0/tests/test_honorifics.py +211 -0
  37. versed_pdf-1.1.0/tests/test_layout_markdown.py +88 -0
  38. versed_pdf-1.1.0/tests/test_qcf.py +219 -0
  39. versed_pdf-1.1.0/tests/test_repair.py +182 -0
  40. versed_pdf-1.1.0/tests/test_routing.py +32 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Sama Team
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,73 @@
1
+ Metadata-Version: 2.4
2
+ Name: versed-pdf
3
+ Version: 1.1.0
4
+ Summary: Semantic PDF-to-Markdown engine for Arabic and bilingual texts with local routing and repair
5
+ Author: Versed Team
6
+ License: MIT
7
+ Keywords: arabic,pdf,markdown,text-extraction,quran,ocr,nlp
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.9
13
+ Classifier: Programming Language :: Python :: 3.10
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Topic :: Text Processing :: Linguistic
17
+ Requires-Python: >=3.9
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ Provides-Extra: pdf
21
+ Requires-Dist: pymupdf>=1.24.0; extra == "pdf"
22
+ Provides-Extra: ocr
23
+ Requires-Dist: pytesseract>=0.3.10; extra == "ocr"
24
+ Requires-Dist: Pillow>=10.0.0; extra == "ocr"
25
+ Provides-Extra: dev
26
+ Requires-Dist: pytest>=7.0; extra == "dev"
27
+ Dynamic: license-file
28
+
29
+ # versed
30
+
31
+ Local PDF-to-Markdown tooling for Arabic and bilingual texts.
32
+
33
+ It repairs broken extraction, decodes QCF Quran fonts, classifies pages, and renders semantic Markdown from local PDFs.
34
+
35
+ ## Install
36
+
37
+ ```bash
38
+ pip install versed-pdf
39
+ pip install versed-pdf[pdf]
40
+ pip install versed-pdf[pdf,ocr]
41
+ ```
42
+
43
+ ## Quick start
44
+
45
+ ```python
46
+ from versed import extract_document
47
+
48
+ result = extract_document("book.pdf", title="Book")
49
+ print(result.markdown)
50
+ ```
51
+
52
+ ## CLI
53
+
54
+ ```bash
55
+ versed repair-text "taf߬l"
56
+ versed detect book.pdf
57
+ versed classify book.pdf
58
+ versed extract book.pdf -o book.md
59
+ ```
60
+
61
+ ## Public modules
62
+
63
+ - `versed.repair`: Sabon mojibake repair helpers
64
+ - `versed.qcf`: QCF Quran font decoding
65
+ - `versed.classify`: local page classification and backend selection
66
+ - `versed.routing`: cost-aware routing heuristics
67
+ - `versed.layout`: aligned words to semantic blocks
68
+ - `versed.markdown`: semantic blocks to Markdown/plain text
69
+ - `versed.extract`: end-to-end local extraction
70
+
71
+ ## License
72
+
73
+ MIT
@@ -0,0 +1,45 @@
1
+ # versed
2
+
3
+ Local PDF-to-Markdown tooling for Arabic and bilingual texts.
4
+
5
+ It repairs broken extraction, decodes QCF Quran fonts, classifies pages, and renders semantic Markdown from local PDFs.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pip install versed-pdf
11
+ pip install versed-pdf[pdf]
12
+ pip install versed-pdf[pdf,ocr]
13
+ ```
14
+
15
+ ## Quick start
16
+
17
+ ```python
18
+ from versed import extract_document
19
+
20
+ result = extract_document("book.pdf", title="Book")
21
+ print(result.markdown)
22
+ ```
23
+
24
+ ## CLI
25
+
26
+ ```bash
27
+ versed repair-text "taf߬l"
28
+ versed detect book.pdf
29
+ versed classify book.pdf
30
+ versed extract book.pdf -o book.md
31
+ ```
32
+
33
+ ## Public modules
34
+
35
+ - `versed.repair`: Sabon mojibake repair helpers
36
+ - `versed.qcf`: QCF Quran font decoding
37
+ - `versed.classify`: local page classification and backend selection
38
+ - `versed.routing`: cost-aware routing heuristics
39
+ - `versed.layout`: aligned words to semantic blocks
40
+ - `versed.markdown`: semantic blocks to Markdown/plain text
41
+ - `versed.extract`: end-to-end local extraction
42
+
43
+ ## License
44
+
45
+ MIT
@@ -0,0 +1,47 @@
1
+ [project]
2
+ name = "versed-pdf"
3
+ version = "1.1.0"
4
+ description = "Semantic PDF-to-Markdown engine for Arabic and bilingual texts with local routing and repair"
5
+ readme = "README.md"
6
+ license = {text = "MIT"}
7
+ requires-python = ">=3.9"
8
+ authors = [
9
+ {name = "Versed Team"}
10
+ ]
11
+ classifiers = [
12
+ "Development Status :: 4 - Beta",
13
+ "Intended Audience :: Developers",
14
+ "License :: OSI Approved :: MIT License",
15
+ "Programming Language :: Python :: 3",
16
+ "Programming Language :: Python :: 3.9",
17
+ "Programming Language :: Python :: 3.10",
18
+ "Programming Language :: Python :: 3.11",
19
+ "Programming Language :: Python :: 3.12",
20
+ "Topic :: Text Processing :: Linguistic",
21
+ ]
22
+ keywords = ["arabic", "pdf", "markdown", "text-extraction", "quran", "ocr", "nlp"]
23
+
24
+ dependencies = [] # Zero required deps! Pure Python for string repair.
25
+
26
+ [project.optional-dependencies]
27
+ pdf = ["pymupdf>=1.24.0"]
28
+ ocr = ["pytesseract>=0.3.10", "Pillow>=10.0.0"]
29
+ dev = ["pytest>=7.0"]
30
+
31
+ [project.scripts]
32
+ versed = "versed.cli:main"
33
+
34
+ [build-system]
35
+ requires = ["setuptools>=65.0", "wheel"]
36
+ build-backend = "setuptools.build_meta"
37
+
38
+ [tool.setuptools.packages.find]
39
+ where = ["src"]
40
+
41
+ [tool.setuptools.package-data]
42
+ versed = ["_data/*.json", "_data/*.db", "py.typed"]
43
+
44
+ [tool.pytest.ini_options]
45
+ testpaths = ["tests"]
46
+ python_files = ["test_*.py"]
47
+ addopts = "-v --tb=short"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,144 @@
1
+ """versed — local PDF-to-Markdown tooling for Arabic and bilingual texts."""
2
+
3
+ __version__ = "1.1.0"
4
+
5
+ from .arabic import (
6
+ detect_batch_reversal,
7
+ is_arabic,
8
+ is_mostly_arabic,
9
+ orphan_diacritic_rate,
10
+ strip_diacritics,
11
+ )
12
+ from .classify import (
13
+ BackendConfig,
14
+ PageProbe,
15
+ PageType,
16
+ classify_and_select,
17
+ classify_page,
18
+ select_backend,
19
+ )
20
+ from .detect import (
21
+ KNOWN_MOJIBAKE_CHARS,
22
+ MojibakeReport,
23
+ detect_mojibake,
24
+ detect_mojibake_in_pdf,
25
+ )
26
+ from .extract import ExtractResult, extract_document
27
+ from .health import summarize_text_health
28
+ from .honorifics import (
29
+ HONORIFIC_SYMBOLS,
30
+ NormalizedWord,
31
+ annotate_transliterations,
32
+ decode_honorific,
33
+ expand_honorifics,
34
+ find_transliteration,
35
+ get_spoken_text,
36
+ has_leading_honorific,
37
+ normalize_text,
38
+ normalize_words,
39
+ )
40
+ from .layout import (
41
+ document_from_aligned_words,
42
+ document_from_markdown,
43
+ document_from_structured,
44
+ )
45
+ from .markdown import (
46
+ EnhancedMarkdownResult,
47
+ build_enhanced_markdown,
48
+ compute_aligned_words_checksum,
49
+ compute_cache_key,
50
+ )
51
+ from .qcf import (
52
+ QCFDecoder,
53
+ QCFVerse,
54
+ QCFWord,
55
+ build_qcf_mapping_from_quran_data,
56
+ detect_qcf_regions,
57
+ extract_qcf_page_number,
58
+ is_qcf_glyph,
59
+ is_qcf_text,
60
+ )
61
+ from .repair import (
62
+ SABON_CHAR_REPAIR,
63
+ SABON_FONT_PREFIXES,
64
+ extract_repairable_font_spans,
65
+ find_font_for_word,
66
+ is_repairable_font,
67
+ repair_text,
68
+ repair_text_for_font,
69
+ repair_words_with_font_info,
70
+ )
71
+ from .routing import (
72
+ EnrichmentDecision,
73
+ PageObservations,
74
+ TaskNeeds,
75
+ observe_from_extraction,
76
+ observe_page,
77
+ route_enrichment,
78
+ )
79
+ from .types import AlignedWord, BlockType, Document, TextBlock, WordBox
80
+
81
+ __all__ = [
82
+ "AlignedWord",
83
+ "BackendConfig",
84
+ "BlockType",
85
+ "Document",
86
+ "EnhancedMarkdownResult",
87
+ "EnrichmentDecision",
88
+ "ExtractResult",
89
+ "HONORIFIC_SYMBOLS",
90
+ "KNOWN_MOJIBAKE_CHARS",
91
+ "MojibakeReport",
92
+ "NormalizedWord",
93
+ "PageObservations",
94
+ "PageProbe",
95
+ "PageType",
96
+ "QCFDecoder",
97
+ "QCFVerse",
98
+ "QCFWord",
99
+ "SABON_CHAR_REPAIR",
100
+ "SABON_FONT_PREFIXES",
101
+ "TaskNeeds",
102
+ "TextBlock",
103
+ "WordBox",
104
+ "annotate_transliterations",
105
+ "build_enhanced_markdown",
106
+ "build_qcf_mapping_from_quran_data",
107
+ "classify_and_select",
108
+ "classify_page",
109
+ "compute_aligned_words_checksum",
110
+ "compute_cache_key",
111
+ "decode_honorific",
112
+ "detect_batch_reversal",
113
+ "detect_mojibake",
114
+ "detect_mojibake_in_pdf",
115
+ "detect_qcf_regions",
116
+ "document_from_aligned_words",
117
+ "document_from_markdown",
118
+ "document_from_structured",
119
+ "expand_honorifics",
120
+ "extract_document",
121
+ "extract_qcf_page_number",
122
+ "extract_repairable_font_spans",
123
+ "find_font_for_word",
124
+ "find_transliteration",
125
+ "get_spoken_text",
126
+ "has_leading_honorific",
127
+ "is_arabic",
128
+ "is_mostly_arabic",
129
+ "is_qcf_glyph",
130
+ "is_qcf_text",
131
+ "is_repairable_font",
132
+ "normalize_text",
133
+ "normalize_words",
134
+ "observe_from_extraction",
135
+ "observe_page",
136
+ "orphan_diacritic_rate",
137
+ "repair_text",
138
+ "repair_text_for_font",
139
+ "repair_words_with_font_info",
140
+ "route_enrichment",
141
+ "select_backend",
142
+ "strip_diacritics",
143
+ "summarize_text_health",
144
+ ]
@@ -0,0 +1,213 @@
1
+ """
2
+ Arabic text utilities — normalization, detection, similarity.
3
+
4
+ Internal module used by honorifics.py and detect.py.
5
+ Not part of the public API.
6
+ """
7
+
8
+ import re
9
+ import unicodedata
10
+ from typing import Dict, Optional
11
+
12
+
13
+ class TextUtils:
14
+ """Text processing utilities."""
15
+
16
+ # Arabic character ranges
17
+ ARABIC_PATTERN = re.compile(r'[\u0600-\u06FF\u0750-\u077F\uFB50-\uFDFF\uFE70-\uFEFF]')
18
+
19
+ # Punctuation and diacritics to strip
20
+ PUNCTUATION_PATTERN = re.compile(
21
+ r"[\'\"`´′″‹›«»\u2018\u2019\u201C\u201D\u02BC\u2032\u2033.,;:!?()[\]{}،؛؟–—-]"
22
+ )
23
+ DIACRITICS_PATTERN = re.compile(r'[\u064B-\u065F\u0670]')
24
+
25
+ # Alef variants (أإآٱ) → bare alef (ا)
26
+ ALEF_VARIANTS = re.compile(r'[أإآٱ]')
27
+ # Zero-width and directional markers to drop
28
+ ZERO_WIDTH_TRANSLATION = str.maketrans({
29
+ "\u200B": "", # zero width space
30
+ "\u200C": "", # zero width non-joiner
31
+ "\u200D": "", # zero width joiner
32
+ "\u2060": "", # word joiner
33
+ "\uFEFF": "", # zero width no-break space
34
+ "\u200E": "", # LRM
35
+ "\u200F": "", # RLM
36
+ })
37
+ # Common Latin ligatures (extra safety; NFKC should also handle these)
38
+ LIGATURE_TRANSLATION = str.maketrans({
39
+ "\uFB00": "ff",
40
+ "\uFB01": "fi",
41
+ "\uFB02": "fl",
42
+ "\uFB03": "ffi",
43
+ "\uFB04": "ffl",
44
+ "\uFB05": "st",
45
+ "\uFB06": "st",
46
+ })
47
+
48
+ @classmethod
49
+ def normalize_arabic(cls, text: str) -> str:
50
+ """
51
+ Normalize Arabic text for comparison/matching.
52
+
53
+ - Removes diacritics (tashkeel)
54
+ - Normalizes alef variants to bare alef
55
+ - Normalizes teh marbuta to heh
56
+ - Normalizes alef maksura to yeh
57
+ - Removes tatweel (kashida)
58
+ """
59
+ if not text:
60
+ return ""
61
+
62
+ # Remove tashkeel (Arabic diacritics)
63
+ text = cls.DIACRITICS_PATTERN.sub('', text)
64
+
65
+ # Normalize alef variants to bare alef
66
+ text = cls.ALEF_VARIANTS.sub('ا', text)
67
+
68
+ # Normalize teh marbuta to heh
69
+ text = text.replace('ة', 'ه')
70
+
71
+ # Normalize alef maksura to yeh
72
+ text = text.replace('ى', 'ي')
73
+
74
+ # Remove tatweel (kashida)
75
+ text = text.replace('\u0640', '')
76
+
77
+ return text.strip()
78
+
79
+ @classmethod
80
+ def normalize(cls, text: str) -> str:
81
+ """
82
+ Normalize text for matching.
83
+
84
+ - Lowercases (for non-Arabic)
85
+ - Strips whitespace
86
+ - Removes punctuation
87
+ - Applies full Arabic normalization (diacritics, alef variants, etc.)
88
+ """
89
+ if not text:
90
+ return ""
91
+ text = unicodedata.normalize("NFKC", text).strip()
92
+ text = text.translate(cls.ZERO_WIDTH_TRANSLATION)
93
+ text = text.translate(cls.LIGATURE_TRANSLATION)
94
+ text = cls.PUNCTUATION_PATTERN.sub("", text)
95
+
96
+ # Apply full Arabic normalization
97
+ text = cls.normalize_arabic(text)
98
+
99
+ # Lowercase for non-Arabic matching (safe after Arabic normalization)
100
+ text = text.lower()
101
+
102
+ return text.strip()
103
+
104
+ @classmethod
105
+ def is_arabic(cls, text: str) -> bool:
106
+ """Check if text contains Arabic characters."""
107
+ return bool(cls.ARABIC_PATTERN.search(text)) if text else False
108
+
109
+ @classmethod
110
+ def levenshtein_similarity(cls, s1: str, s2: str) -> float:
111
+ """
112
+ Calculate similarity score between two strings.
113
+
114
+ Returns:
115
+ Similarity score 0.0-1.0 (1.0 = identical)
116
+ """
117
+ s1, s2 = cls.normalize(s1), cls.normalize(s2)
118
+ if not s1 or not s2:
119
+ return 0.0
120
+ if s1 == s2:
121
+ return 1.0
122
+
123
+ len1, len2 = len(s1), len(s2)
124
+ if len1 > len2:
125
+ s1, s2 = s2, s1
126
+ len1, len2 = len2, len1
127
+
128
+ current_row = list(range(len1 + 1))
129
+ for i in range(1, len2 + 1):
130
+ previous_row, current_row = current_row, [i] + [0] * len1
131
+ for j in range(1, len1 + 1):
132
+ add = previous_row[j] + 1
133
+ delete = current_row[j - 1] + 1
134
+ change = previous_row[j - 1] + (0 if s1[j - 1] == s2[i - 1] else 1)
135
+ current_row[j] = min(add, delete, change)
136
+
137
+ distance = current_row[len1]
138
+ return 1.0 - (distance / max(len1, len2))
139
+
140
+
141
+ class HonorificsHandler:
142
+ """
143
+ Handler for Islamic honorific symbols.
144
+
145
+ Maps display symbols (ﷺ, ﷻ, etc.) to spoken Arabic phrases.
146
+ """
147
+
148
+ # Honorific expansions WITH HARAKAT for proper TTS pronunciation
149
+ HONORIFICS: Dict[str, str] = {
150
+ # Prophet Muhammad ﷺ - صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ
151
+ '\uFDFA': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
152
+ '\uFD46': 'صَلَّى اللهُ عَلَيْهِ وَآلِهِ وَسَلَّمَ',
153
+ 'ﷺ': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
154
+ '\uF067': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
155
+ '\uF030': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
156
+ '\uF031': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
157
+ 'PBUH': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
158
+ 'pbuh': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
159
+ '(s)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
160
+ '(S)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
161
+ '(saw)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
162
+ '(SAW)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
163
+ '(PBUH)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
164
+ '(pbuh)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
165
+
166
+ # Allah ﷻ - عَزَّ وَجَلَّ
167
+ '\uFDFB': 'عَزَّ وَجَلَّ',
168
+ 'ﷻ': 'عَزَّ وَجَلَّ',
169
+ '\uFBF2': 'عَزَّ وَجَلَّ',
170
+ '\uFBF1': 'عَزَّ وَجَلَّ',
171
+ '\uF063': 'سُبْحَانَهُ وَتَعَالَى',
172
+ '(swt)': 'سُبْحَانَهُ وَتَعَالَى',
173
+ '(SWT)': 'سُبْحَانَهُ وَتَعَالَى',
174
+
175
+ # Other prophets - عَلَيْهِ السَّلَامُ
176
+ '\uF064': 'عَلَيْهِ السَّلَامُ',
177
+ '(as)': 'عَلَيْهِ السَّلَامُ',
178
+ '(AS)': 'عَلَيْهِ السَّلَامُ',
179
+
180
+ # Companions - رَضِيَ اللهُ عَنْهُ
181
+ '\uF065': 'رَضِيَ اللهُ عَنْهُ',
182
+ '(ra)': 'رَضِيَ اللهُ عَنْهُ',
183
+ '(RA)': 'رَضِيَ اللهُ عَنْهُ',
184
+ }
185
+
186
+ SYMBOLS = set(HONORIFICS.keys())
187
+
188
+ @classmethod
189
+ def is_honorific(cls, text: str) -> bool:
190
+ """Check if text is an honorific symbol."""
191
+ return text in cls.SYMBOLS
192
+
193
+ @classmethod
194
+ def expand(cls, symbol: str) -> Optional[str]:
195
+ """Expand honorific symbol to spoken Arabic."""
196
+ return cls.HONORIFICS.get(symbol)
197
+
198
+ @classmethod
199
+ def process_text(cls, text: str) -> tuple[str, str]:
200
+ """
201
+ Process text that may contain honorifics.
202
+
203
+ Returns:
204
+ Tuple of (display_text, spoken_text)
205
+ """
206
+ display = text
207
+ spoken = text
208
+
209
+ for symbol, expansion in cls.HONORIFICS.items():
210
+ if symbol in text:
211
+ spoken = spoken.replace(symbol, expansion)
212
+
213
+ return display, spoken
@@ -0,0 +1,45 @@
1
+ {
2
+ "words": {
3
+ "prophet": "النبي",
4
+ "allah": "الله",
5
+ "god": "الله",
6
+ "quran": "القرآن",
7
+ "qur'an": "القرآن",
8
+ "hadith": "حديث",
9
+ "sunnah": "سنة",
10
+ "surah": "سورة",
11
+ "sura": "سورة",
12
+ "ayah": "آية",
13
+ "ayat": "آيات",
14
+ "hammazan": "هَمَّاز",
15
+ "lammazan": "لَمَّاز",
16
+ "ayyaban": "عَيَّاب",
17
+ "al-fatiha": "الفاتحة",
18
+ "al-baqarah": "البقرة",
19
+ "al-imran": "آل عمران",
20
+ "al-nisa": "النساء",
21
+ "al-ma'idah": "المائدة",
22
+ "al-an'am": "الأنعام",
23
+ "al-a'raf": "الأعراف",
24
+ "al-anfal": "الأنفال",
25
+ "al-tawbah": "التوبة",
26
+ "al-jathiyah": "الجاثية",
27
+ "al-kahf": "الكهف",
28
+ "al-isra": "الإسراء",
29
+ "al-maryam": "مريم",
30
+ "al-haj": "الحج",
31
+ "al-nur": "النور",
32
+ "al-furqan": "الفرقان",
33
+ "ya-sin": "يس",
34
+ "al-rahman": "الرحمن",
35
+ "al-mulk": "الملك",
36
+ "al-qalam": "القلم"
37
+ },
38
+ "prefixes": {
39
+ "abu": "أبو",
40
+ "ibn": "ابن",
41
+ "umm": "أم",
42
+ "bint": "بنت",
43
+ "al-": "الـ"
44
+ }
45
+ }