versed-pdf 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- versed_pdf-1.1.0/LICENSE +21 -0
- versed_pdf-1.1.0/PKG-INFO +73 -0
- versed_pdf-1.1.0/README.md +45 -0
- versed_pdf-1.1.0/pyproject.toml +47 -0
- versed_pdf-1.1.0/setup.cfg +4 -0
- versed_pdf-1.1.0/src/versed/__init__.py +144 -0
- versed_pdf-1.1.0/src/versed/_arabic.py +213 -0
- versed_pdf-1.1.0/src/versed/_data/qcf_mapping.db +0 -0
- versed_pdf-1.1.0/src/versed/_data/transliterations.json +45 -0
- versed_pdf-1.1.0/src/versed/arabic.py +110 -0
- versed_pdf-1.1.0/src/versed/classify.py +211 -0
- versed_pdf-1.1.0/src/versed/cli.py +296 -0
- versed_pdf-1.1.0/src/versed/detect.py +118 -0
- versed_pdf-1.1.0/src/versed/extract.py +296 -0
- versed_pdf-1.1.0/src/versed/filtering.py +74 -0
- versed_pdf-1.1.0/src/versed/health.py +174 -0
- versed_pdf-1.1.0/src/versed/honorifics.py +396 -0
- versed_pdf-1.1.0/src/versed/layout.py +457 -0
- versed_pdf-1.1.0/src/versed/markdown.py +166 -0
- versed_pdf-1.1.0/src/versed/py.typed +0 -0
- versed_pdf-1.1.0/src/versed/qcf.py +465 -0
- versed_pdf-1.1.0/src/versed/repair.py +134 -0
- versed_pdf-1.1.0/src/versed/routing.py +230 -0
- versed_pdf-1.1.0/src/versed/types.py +126 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/PKG-INFO +73 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/SOURCES.txt +38 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/dependency_links.txt +1 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/entry_points.txt +2 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/requires.txt +10 -0
- versed_pdf-1.1.0/src/versed_pdf.egg-info/top_level.txt +1 -0
- versed_pdf-1.1.0/tests/test_arabic.py +35 -0
- versed_pdf-1.1.0/tests/test_classify.py +41 -0
- versed_pdf-1.1.0/tests/test_cli.py +59 -0
- versed_pdf-1.1.0/tests/test_detect.py +72 -0
- versed_pdf-1.1.0/tests/test_extract.py +84 -0
- versed_pdf-1.1.0/tests/test_honorifics.py +211 -0
- versed_pdf-1.1.0/tests/test_layout_markdown.py +88 -0
- versed_pdf-1.1.0/tests/test_qcf.py +219 -0
- versed_pdf-1.1.0/tests/test_repair.py +182 -0
- versed_pdf-1.1.0/tests/test_routing.py +32 -0
versed_pdf-1.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Sama Team
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: versed-pdf
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Semantic PDF-to-Markdown engine for Arabic and bilingual texts with local routing and repair
|
|
5
|
+
Author: Versed Team
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: arabic,pdf,markdown,text-extraction,quran,ocr,nlp
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
17
|
+
Requires-Python: >=3.9
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Provides-Extra: pdf
|
|
21
|
+
Requires-Dist: pymupdf>=1.24.0; extra == "pdf"
|
|
22
|
+
Provides-Extra: ocr
|
|
23
|
+
Requires-Dist: pytesseract>=0.3.10; extra == "ocr"
|
|
24
|
+
Requires-Dist: Pillow>=10.0.0; extra == "ocr"
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# versed
|
|
30
|
+
|
|
31
|
+
Local PDF-to-Markdown tooling for Arabic and bilingual texts.
|
|
32
|
+
|
|
33
|
+
It repairs broken extraction, decodes QCF Quran fonts, classifies pages, and renders semantic Markdown from local PDFs.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install versed-pdf
|
|
39
|
+
pip install versed-pdf[pdf]
|
|
40
|
+
pip install versed-pdf[pdf,ocr]
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Quick start
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from versed import extract_document
|
|
47
|
+
|
|
48
|
+
result = extract_document("book.pdf", title="Book")
|
|
49
|
+
print(result.markdown)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## CLI
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
versed repair-text "taf߬l"
|
|
56
|
+
versed detect book.pdf
|
|
57
|
+
versed classify book.pdf
|
|
58
|
+
versed extract book.pdf -o book.md
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Public modules
|
|
62
|
+
|
|
63
|
+
- `versed.repair`: Sabon mojibake repair helpers
|
|
64
|
+
- `versed.qcf`: QCF Quran font decoding
|
|
65
|
+
- `versed.classify`: local page classification and backend selection
|
|
66
|
+
- `versed.routing`: cost-aware routing heuristics
|
|
67
|
+
- `versed.layout`: aligned words to semantic blocks
|
|
68
|
+
- `versed.markdown`: semantic blocks to Markdown/plain text
|
|
69
|
+
- `versed.extract`: end-to-end local extraction
|
|
70
|
+
|
|
71
|
+
## License
|
|
72
|
+
|
|
73
|
+
MIT
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# versed
|
|
2
|
+
|
|
3
|
+
Local PDF-to-Markdown tooling for Arabic and bilingual texts.
|
|
4
|
+
|
|
5
|
+
It repairs broken extraction, decodes QCF Quran fonts, classifies pages, and renders semantic Markdown from local PDFs.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install versed-pdf
|
|
11
|
+
pip install versed-pdf[pdf]
|
|
12
|
+
pip install versed-pdf[pdf,ocr]
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Quick start
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from versed import extract_document
|
|
19
|
+
|
|
20
|
+
result = extract_document("book.pdf", title="Book")
|
|
21
|
+
print(result.markdown)
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## CLI
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
versed repair-text "taf߬l"
|
|
28
|
+
versed detect book.pdf
|
|
29
|
+
versed classify book.pdf
|
|
30
|
+
versed extract book.pdf -o book.md
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Public modules
|
|
34
|
+
|
|
35
|
+
- `versed.repair`: Sabon mojibake repair helpers
|
|
36
|
+
- `versed.qcf`: QCF Quran font decoding
|
|
37
|
+
- `versed.classify`: local page classification and backend selection
|
|
38
|
+
- `versed.routing`: cost-aware routing heuristics
|
|
39
|
+
- `versed.layout`: aligned words to semantic blocks
|
|
40
|
+
- `versed.markdown`: semantic blocks to Markdown/plain text
|
|
41
|
+
- `versed.extract`: end-to-end local extraction
|
|
42
|
+
|
|
43
|
+
## License
|
|
44
|
+
|
|
45
|
+
MIT
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "versed-pdf"
|
|
3
|
+
version = "1.1.0"
|
|
4
|
+
description = "Semantic PDF-to-Markdown engine for Arabic and bilingual texts with local routing and repair"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = {text = "MIT"}
|
|
7
|
+
requires-python = ">=3.9"
|
|
8
|
+
authors = [
|
|
9
|
+
{name = "Versed Team"}
|
|
10
|
+
]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 4 - Beta",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"License :: OSI Approved :: MIT License",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.9",
|
|
17
|
+
"Programming Language :: Python :: 3.10",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Topic :: Text Processing :: Linguistic",
|
|
21
|
+
]
|
|
22
|
+
keywords = ["arabic", "pdf", "markdown", "text-extraction", "quran", "ocr", "nlp"]
|
|
23
|
+
|
|
24
|
+
dependencies = [] # Zero required deps! Pure Python for string repair.
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
pdf = ["pymupdf>=1.24.0"]
|
|
28
|
+
ocr = ["pytesseract>=0.3.10", "Pillow>=10.0.0"]
|
|
29
|
+
dev = ["pytest>=7.0"]
|
|
30
|
+
|
|
31
|
+
[project.scripts]
|
|
32
|
+
versed = "versed.cli:main"
|
|
33
|
+
|
|
34
|
+
[build-system]
|
|
35
|
+
requires = ["setuptools>=65.0", "wheel"]
|
|
36
|
+
build-backend = "setuptools.build_meta"
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.packages.find]
|
|
39
|
+
where = ["src"]
|
|
40
|
+
|
|
41
|
+
[tool.setuptools.package-data]
|
|
42
|
+
versed = ["_data/*.json", "_data/*.db", "py.typed"]
|
|
43
|
+
|
|
44
|
+
[tool.pytest.ini_options]
|
|
45
|
+
testpaths = ["tests"]
|
|
46
|
+
python_files = ["test_*.py"]
|
|
47
|
+
addopts = "-v --tb=short"
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""versed — local PDF-to-Markdown tooling for Arabic and bilingual texts."""
|
|
2
|
+
|
|
3
|
+
__version__ = "1.1.0"
|
|
4
|
+
|
|
5
|
+
from .arabic import (
|
|
6
|
+
detect_batch_reversal,
|
|
7
|
+
is_arabic,
|
|
8
|
+
is_mostly_arabic,
|
|
9
|
+
orphan_diacritic_rate,
|
|
10
|
+
strip_diacritics,
|
|
11
|
+
)
|
|
12
|
+
from .classify import (
|
|
13
|
+
BackendConfig,
|
|
14
|
+
PageProbe,
|
|
15
|
+
PageType,
|
|
16
|
+
classify_and_select,
|
|
17
|
+
classify_page,
|
|
18
|
+
select_backend,
|
|
19
|
+
)
|
|
20
|
+
from .detect import (
|
|
21
|
+
KNOWN_MOJIBAKE_CHARS,
|
|
22
|
+
MojibakeReport,
|
|
23
|
+
detect_mojibake,
|
|
24
|
+
detect_mojibake_in_pdf,
|
|
25
|
+
)
|
|
26
|
+
from .extract import ExtractResult, extract_document
|
|
27
|
+
from .health import summarize_text_health
|
|
28
|
+
from .honorifics import (
|
|
29
|
+
HONORIFIC_SYMBOLS,
|
|
30
|
+
NormalizedWord,
|
|
31
|
+
annotate_transliterations,
|
|
32
|
+
decode_honorific,
|
|
33
|
+
expand_honorifics,
|
|
34
|
+
find_transliteration,
|
|
35
|
+
get_spoken_text,
|
|
36
|
+
has_leading_honorific,
|
|
37
|
+
normalize_text,
|
|
38
|
+
normalize_words,
|
|
39
|
+
)
|
|
40
|
+
from .layout import (
|
|
41
|
+
document_from_aligned_words,
|
|
42
|
+
document_from_markdown,
|
|
43
|
+
document_from_structured,
|
|
44
|
+
)
|
|
45
|
+
from .markdown import (
|
|
46
|
+
EnhancedMarkdownResult,
|
|
47
|
+
build_enhanced_markdown,
|
|
48
|
+
compute_aligned_words_checksum,
|
|
49
|
+
compute_cache_key,
|
|
50
|
+
)
|
|
51
|
+
from .qcf import (
|
|
52
|
+
QCFDecoder,
|
|
53
|
+
QCFVerse,
|
|
54
|
+
QCFWord,
|
|
55
|
+
build_qcf_mapping_from_quran_data,
|
|
56
|
+
detect_qcf_regions,
|
|
57
|
+
extract_qcf_page_number,
|
|
58
|
+
is_qcf_glyph,
|
|
59
|
+
is_qcf_text,
|
|
60
|
+
)
|
|
61
|
+
from .repair import (
|
|
62
|
+
SABON_CHAR_REPAIR,
|
|
63
|
+
SABON_FONT_PREFIXES,
|
|
64
|
+
extract_repairable_font_spans,
|
|
65
|
+
find_font_for_word,
|
|
66
|
+
is_repairable_font,
|
|
67
|
+
repair_text,
|
|
68
|
+
repair_text_for_font,
|
|
69
|
+
repair_words_with_font_info,
|
|
70
|
+
)
|
|
71
|
+
from .routing import (
|
|
72
|
+
EnrichmentDecision,
|
|
73
|
+
PageObservations,
|
|
74
|
+
TaskNeeds,
|
|
75
|
+
observe_from_extraction,
|
|
76
|
+
observe_page,
|
|
77
|
+
route_enrichment,
|
|
78
|
+
)
|
|
79
|
+
from .types import AlignedWord, BlockType, Document, TextBlock, WordBox
|
|
80
|
+
|
|
81
|
+
__all__ = [
|
|
82
|
+
"AlignedWord",
|
|
83
|
+
"BackendConfig",
|
|
84
|
+
"BlockType",
|
|
85
|
+
"Document",
|
|
86
|
+
"EnhancedMarkdownResult",
|
|
87
|
+
"EnrichmentDecision",
|
|
88
|
+
"ExtractResult",
|
|
89
|
+
"HONORIFIC_SYMBOLS",
|
|
90
|
+
"KNOWN_MOJIBAKE_CHARS",
|
|
91
|
+
"MojibakeReport",
|
|
92
|
+
"NormalizedWord",
|
|
93
|
+
"PageObservations",
|
|
94
|
+
"PageProbe",
|
|
95
|
+
"PageType",
|
|
96
|
+
"QCFDecoder",
|
|
97
|
+
"QCFVerse",
|
|
98
|
+
"QCFWord",
|
|
99
|
+
"SABON_CHAR_REPAIR",
|
|
100
|
+
"SABON_FONT_PREFIXES",
|
|
101
|
+
"TaskNeeds",
|
|
102
|
+
"TextBlock",
|
|
103
|
+
"WordBox",
|
|
104
|
+
"annotate_transliterations",
|
|
105
|
+
"build_enhanced_markdown",
|
|
106
|
+
"build_qcf_mapping_from_quran_data",
|
|
107
|
+
"classify_and_select",
|
|
108
|
+
"classify_page",
|
|
109
|
+
"compute_aligned_words_checksum",
|
|
110
|
+
"compute_cache_key",
|
|
111
|
+
"decode_honorific",
|
|
112
|
+
"detect_batch_reversal",
|
|
113
|
+
"detect_mojibake",
|
|
114
|
+
"detect_mojibake_in_pdf",
|
|
115
|
+
"detect_qcf_regions",
|
|
116
|
+
"document_from_aligned_words",
|
|
117
|
+
"document_from_markdown",
|
|
118
|
+
"document_from_structured",
|
|
119
|
+
"expand_honorifics",
|
|
120
|
+
"extract_document",
|
|
121
|
+
"extract_qcf_page_number",
|
|
122
|
+
"extract_repairable_font_spans",
|
|
123
|
+
"find_font_for_word",
|
|
124
|
+
"find_transliteration",
|
|
125
|
+
"get_spoken_text",
|
|
126
|
+
"has_leading_honorific",
|
|
127
|
+
"is_arabic",
|
|
128
|
+
"is_mostly_arabic",
|
|
129
|
+
"is_qcf_glyph",
|
|
130
|
+
"is_qcf_text",
|
|
131
|
+
"is_repairable_font",
|
|
132
|
+
"normalize_text",
|
|
133
|
+
"normalize_words",
|
|
134
|
+
"observe_from_extraction",
|
|
135
|
+
"observe_page",
|
|
136
|
+
"orphan_diacritic_rate",
|
|
137
|
+
"repair_text",
|
|
138
|
+
"repair_text_for_font",
|
|
139
|
+
"repair_words_with_font_info",
|
|
140
|
+
"route_enrichment",
|
|
141
|
+
"select_backend",
|
|
142
|
+
"strip_diacritics",
|
|
143
|
+
"summarize_text_health",
|
|
144
|
+
]
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Arabic text utilities — normalization, detection, similarity.
|
|
3
|
+
|
|
4
|
+
Internal module used by honorifics.py and detect.py.
|
|
5
|
+
Not part of the public API.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import unicodedata
|
|
10
|
+
from typing import Dict, Optional
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class TextUtils:
|
|
14
|
+
"""Text processing utilities."""
|
|
15
|
+
|
|
16
|
+
# Arabic character ranges
|
|
17
|
+
ARABIC_PATTERN = re.compile(r'[\u0600-\u06FF\u0750-\u077F\uFB50-\uFDFF\uFE70-\uFEFF]')
|
|
18
|
+
|
|
19
|
+
# Punctuation and diacritics to strip
|
|
20
|
+
PUNCTUATION_PATTERN = re.compile(
|
|
21
|
+
r"[\'\"`´′″‹›«»\u2018\u2019\u201C\u201D\u02BC\u2032\u2033.,;:!?()[\]{}،؛؟–—-]"
|
|
22
|
+
)
|
|
23
|
+
DIACRITICS_PATTERN = re.compile(r'[\u064B-\u065F\u0670]')
|
|
24
|
+
|
|
25
|
+
# Alef variants (أإآٱ) → bare alef (ا)
|
|
26
|
+
ALEF_VARIANTS = re.compile(r'[أإآٱ]')
|
|
27
|
+
# Zero-width and directional markers to drop
|
|
28
|
+
ZERO_WIDTH_TRANSLATION = str.maketrans({
|
|
29
|
+
"\u200B": "", # zero width space
|
|
30
|
+
"\u200C": "", # zero width non-joiner
|
|
31
|
+
"\u200D": "", # zero width joiner
|
|
32
|
+
"\u2060": "", # word joiner
|
|
33
|
+
"\uFEFF": "", # zero width no-break space
|
|
34
|
+
"\u200E": "", # LRM
|
|
35
|
+
"\u200F": "", # RLM
|
|
36
|
+
})
|
|
37
|
+
# Common Latin ligatures (extra safety; NFKC should also handle these)
|
|
38
|
+
LIGATURE_TRANSLATION = str.maketrans({
|
|
39
|
+
"\uFB00": "ff",
|
|
40
|
+
"\uFB01": "fi",
|
|
41
|
+
"\uFB02": "fl",
|
|
42
|
+
"\uFB03": "ffi",
|
|
43
|
+
"\uFB04": "ffl",
|
|
44
|
+
"\uFB05": "st",
|
|
45
|
+
"\uFB06": "st",
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def normalize_arabic(cls, text: str) -> str:
|
|
50
|
+
"""
|
|
51
|
+
Normalize Arabic text for comparison/matching.
|
|
52
|
+
|
|
53
|
+
- Removes diacritics (tashkeel)
|
|
54
|
+
- Normalizes alef variants to bare alef
|
|
55
|
+
- Normalizes teh marbuta to heh
|
|
56
|
+
- Normalizes alef maksura to yeh
|
|
57
|
+
- Removes tatweel (kashida)
|
|
58
|
+
"""
|
|
59
|
+
if not text:
|
|
60
|
+
return ""
|
|
61
|
+
|
|
62
|
+
# Remove tashkeel (Arabic diacritics)
|
|
63
|
+
text = cls.DIACRITICS_PATTERN.sub('', text)
|
|
64
|
+
|
|
65
|
+
# Normalize alef variants to bare alef
|
|
66
|
+
text = cls.ALEF_VARIANTS.sub('ا', text)
|
|
67
|
+
|
|
68
|
+
# Normalize teh marbuta to heh
|
|
69
|
+
text = text.replace('ة', 'ه')
|
|
70
|
+
|
|
71
|
+
# Normalize alef maksura to yeh
|
|
72
|
+
text = text.replace('ى', 'ي')
|
|
73
|
+
|
|
74
|
+
# Remove tatweel (kashida)
|
|
75
|
+
text = text.replace('\u0640', '')
|
|
76
|
+
|
|
77
|
+
return text.strip()
|
|
78
|
+
|
|
79
|
+
@classmethod
|
|
80
|
+
def normalize(cls, text: str) -> str:
|
|
81
|
+
"""
|
|
82
|
+
Normalize text for matching.
|
|
83
|
+
|
|
84
|
+
- Lowercases (for non-Arabic)
|
|
85
|
+
- Strips whitespace
|
|
86
|
+
- Removes punctuation
|
|
87
|
+
- Applies full Arabic normalization (diacritics, alef variants, etc.)
|
|
88
|
+
"""
|
|
89
|
+
if not text:
|
|
90
|
+
return ""
|
|
91
|
+
text = unicodedata.normalize("NFKC", text).strip()
|
|
92
|
+
text = text.translate(cls.ZERO_WIDTH_TRANSLATION)
|
|
93
|
+
text = text.translate(cls.LIGATURE_TRANSLATION)
|
|
94
|
+
text = cls.PUNCTUATION_PATTERN.sub("", text)
|
|
95
|
+
|
|
96
|
+
# Apply full Arabic normalization
|
|
97
|
+
text = cls.normalize_arabic(text)
|
|
98
|
+
|
|
99
|
+
# Lowercase for non-Arabic matching (safe after Arabic normalization)
|
|
100
|
+
text = text.lower()
|
|
101
|
+
|
|
102
|
+
return text.strip()
|
|
103
|
+
|
|
104
|
+
@classmethod
|
|
105
|
+
def is_arabic(cls, text: str) -> bool:
|
|
106
|
+
"""Check if text contains Arabic characters."""
|
|
107
|
+
return bool(cls.ARABIC_PATTERN.search(text)) if text else False
|
|
108
|
+
|
|
109
|
+
@classmethod
|
|
110
|
+
def levenshtein_similarity(cls, s1: str, s2: str) -> float:
|
|
111
|
+
"""
|
|
112
|
+
Calculate similarity score between two strings.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
Similarity score 0.0-1.0 (1.0 = identical)
|
|
116
|
+
"""
|
|
117
|
+
s1, s2 = cls.normalize(s1), cls.normalize(s2)
|
|
118
|
+
if not s1 or not s2:
|
|
119
|
+
return 0.0
|
|
120
|
+
if s1 == s2:
|
|
121
|
+
return 1.0
|
|
122
|
+
|
|
123
|
+
len1, len2 = len(s1), len(s2)
|
|
124
|
+
if len1 > len2:
|
|
125
|
+
s1, s2 = s2, s1
|
|
126
|
+
len1, len2 = len2, len1
|
|
127
|
+
|
|
128
|
+
current_row = list(range(len1 + 1))
|
|
129
|
+
for i in range(1, len2 + 1):
|
|
130
|
+
previous_row, current_row = current_row, [i] + [0] * len1
|
|
131
|
+
for j in range(1, len1 + 1):
|
|
132
|
+
add = previous_row[j] + 1
|
|
133
|
+
delete = current_row[j - 1] + 1
|
|
134
|
+
change = previous_row[j - 1] + (0 if s1[j - 1] == s2[i - 1] else 1)
|
|
135
|
+
current_row[j] = min(add, delete, change)
|
|
136
|
+
|
|
137
|
+
distance = current_row[len1]
|
|
138
|
+
return 1.0 - (distance / max(len1, len2))
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class HonorificsHandler:
|
|
142
|
+
"""
|
|
143
|
+
Handler for Islamic honorific symbols.
|
|
144
|
+
|
|
145
|
+
Maps display symbols (ﷺ, ﷻ, etc.) to spoken Arabic phrases.
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
# Honorific expansions WITH HARAKAT for proper TTS pronunciation
|
|
149
|
+
HONORIFICS: Dict[str, str] = {
|
|
150
|
+
# Prophet Muhammad ﷺ - صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ
|
|
151
|
+
'\uFDFA': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
152
|
+
'\uFD46': 'صَلَّى اللهُ عَلَيْهِ وَآلِهِ وَسَلَّمَ',
|
|
153
|
+
'ﷺ': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
154
|
+
'\uF067': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
155
|
+
'\uF030': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
156
|
+
'\uF031': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
157
|
+
'PBUH': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
158
|
+
'pbuh': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
159
|
+
'(s)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
160
|
+
'(S)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
161
|
+
'(saw)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
162
|
+
'(SAW)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
163
|
+
'(PBUH)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
164
|
+
'(pbuh)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
165
|
+
|
|
166
|
+
# Allah ﷻ - عَزَّ وَجَلَّ
|
|
167
|
+
'\uFDFB': 'عَزَّ وَجَلَّ',
|
|
168
|
+
'ﷻ': 'عَزَّ وَجَلَّ',
|
|
169
|
+
'\uFBF2': 'عَزَّ وَجَلَّ',
|
|
170
|
+
'\uFBF1': 'عَزَّ وَجَلَّ',
|
|
171
|
+
'\uF063': 'سُبْحَانَهُ وَتَعَالَى',
|
|
172
|
+
'(swt)': 'سُبْحَانَهُ وَتَعَالَى',
|
|
173
|
+
'(SWT)': 'سُبْحَانَهُ وَتَعَالَى',
|
|
174
|
+
|
|
175
|
+
# Other prophets - عَلَيْهِ السَّلَامُ
|
|
176
|
+
'\uF064': 'عَلَيْهِ السَّلَامُ',
|
|
177
|
+
'(as)': 'عَلَيْهِ السَّلَامُ',
|
|
178
|
+
'(AS)': 'عَلَيْهِ السَّلَامُ',
|
|
179
|
+
|
|
180
|
+
# Companions - رَضِيَ اللهُ عَنْهُ
|
|
181
|
+
'\uF065': 'رَضِيَ اللهُ عَنْهُ',
|
|
182
|
+
'(ra)': 'رَضِيَ اللهُ عَنْهُ',
|
|
183
|
+
'(RA)': 'رَضِيَ اللهُ عَنْهُ',
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
SYMBOLS = set(HONORIFICS.keys())
|
|
187
|
+
|
|
188
|
+
@classmethod
|
|
189
|
+
def is_honorific(cls, text: str) -> bool:
|
|
190
|
+
"""Check if text is an honorific symbol."""
|
|
191
|
+
return text in cls.SYMBOLS
|
|
192
|
+
|
|
193
|
+
@classmethod
|
|
194
|
+
def expand(cls, symbol: str) -> Optional[str]:
|
|
195
|
+
"""Expand honorific symbol to spoken Arabic."""
|
|
196
|
+
return cls.HONORIFICS.get(symbol)
|
|
197
|
+
|
|
198
|
+
@classmethod
|
|
199
|
+
def process_text(cls, text: str) -> tuple[str, str]:
|
|
200
|
+
"""
|
|
201
|
+
Process text that may contain honorifics.
|
|
202
|
+
|
|
203
|
+
Returns:
|
|
204
|
+
Tuple of (display_text, spoken_text)
|
|
205
|
+
"""
|
|
206
|
+
display = text
|
|
207
|
+
spoken = text
|
|
208
|
+
|
|
209
|
+
for symbol, expansion in cls.HONORIFICS.items():
|
|
210
|
+
if symbol in text:
|
|
211
|
+
spoken = spoken.replace(symbol, expansion)
|
|
212
|
+
|
|
213
|
+
return display, spoken
|
|
Binary file
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"words": {
|
|
3
|
+
"prophet": "النبي",
|
|
4
|
+
"allah": "الله",
|
|
5
|
+
"god": "الله",
|
|
6
|
+
"quran": "القرآن",
|
|
7
|
+
"qur'an": "القرآن",
|
|
8
|
+
"hadith": "حديث",
|
|
9
|
+
"sunnah": "سنة",
|
|
10
|
+
"surah": "سورة",
|
|
11
|
+
"sura": "سورة",
|
|
12
|
+
"ayah": "آية",
|
|
13
|
+
"ayat": "آيات",
|
|
14
|
+
"hammazan": "هَمَّاز",
|
|
15
|
+
"lammazan": "لَمَّاز",
|
|
16
|
+
"ayyaban": "عَيَّاب",
|
|
17
|
+
"al-fatiha": "الفاتحة",
|
|
18
|
+
"al-baqarah": "البقرة",
|
|
19
|
+
"al-imran": "آل عمران",
|
|
20
|
+
"al-nisa": "النساء",
|
|
21
|
+
"al-ma'idah": "المائدة",
|
|
22
|
+
"al-an'am": "الأنعام",
|
|
23
|
+
"al-a'raf": "الأعراف",
|
|
24
|
+
"al-anfal": "الأنفال",
|
|
25
|
+
"al-tawbah": "التوبة",
|
|
26
|
+
"al-jathiyah": "الجاثية",
|
|
27
|
+
"al-kahf": "الكهف",
|
|
28
|
+
"al-isra": "الإسراء",
|
|
29
|
+
"al-maryam": "مريم",
|
|
30
|
+
"al-haj": "الحج",
|
|
31
|
+
"al-nur": "النور",
|
|
32
|
+
"al-furqan": "الفرقان",
|
|
33
|
+
"ya-sin": "يس",
|
|
34
|
+
"al-rahman": "الرحمن",
|
|
35
|
+
"al-mulk": "الملك",
|
|
36
|
+
"al-qalam": "القلم"
|
|
37
|
+
},
|
|
38
|
+
"prefixes": {
|
|
39
|
+
"abu": "أبو",
|
|
40
|
+
"ibn": "ابن",
|
|
41
|
+
"umm": "أم",
|
|
42
|
+
"bint": "بنت",
|
|
43
|
+
"al-": "الـ"
|
|
44
|
+
}
|
|
45
|
+
}
|