versed-pdf 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- versed/__init__.py +144 -0
- versed/_arabic.py +213 -0
- versed/_data/qcf_mapping.db +0 -0
- versed/_data/transliterations.json +45 -0
- versed/arabic.py +110 -0
- versed/classify.py +211 -0
- versed/cli.py +296 -0
- versed/detect.py +118 -0
- versed/extract.py +296 -0
- versed/filtering.py +74 -0
- versed/health.py +174 -0
- versed/honorifics.py +396 -0
- versed/layout.py +457 -0
- versed/markdown.py +166 -0
- versed/py.typed +0 -0
- versed/qcf.py +465 -0
- versed/repair.py +134 -0
- versed/routing.py +230 -0
- versed/types.py +126 -0
- versed_pdf-1.1.0.dist-info/METADATA +73 -0
- versed_pdf-1.1.0.dist-info/RECORD +25 -0
- versed_pdf-1.1.0.dist-info/WHEEL +5 -0
- versed_pdf-1.1.0.dist-info/entry_points.txt +2 -0
- versed_pdf-1.1.0.dist-info/licenses/LICENSE +21 -0
- versed_pdf-1.1.0.dist-info/top_level.txt +1 -0
versed/__init__.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""versed — local PDF-to-Markdown tooling for Arabic and bilingual texts."""
|
|
2
|
+
|
|
3
|
+
__version__ = "1.1.0"
|
|
4
|
+
|
|
5
|
+
from .arabic import (
|
|
6
|
+
detect_batch_reversal,
|
|
7
|
+
is_arabic,
|
|
8
|
+
is_mostly_arabic,
|
|
9
|
+
orphan_diacritic_rate,
|
|
10
|
+
strip_diacritics,
|
|
11
|
+
)
|
|
12
|
+
from .classify import (
|
|
13
|
+
BackendConfig,
|
|
14
|
+
PageProbe,
|
|
15
|
+
PageType,
|
|
16
|
+
classify_and_select,
|
|
17
|
+
classify_page,
|
|
18
|
+
select_backend,
|
|
19
|
+
)
|
|
20
|
+
from .detect import (
|
|
21
|
+
KNOWN_MOJIBAKE_CHARS,
|
|
22
|
+
MojibakeReport,
|
|
23
|
+
detect_mojibake,
|
|
24
|
+
detect_mojibake_in_pdf,
|
|
25
|
+
)
|
|
26
|
+
from .extract import ExtractResult, extract_document
|
|
27
|
+
from .health import summarize_text_health
|
|
28
|
+
from .honorifics import (
|
|
29
|
+
HONORIFIC_SYMBOLS,
|
|
30
|
+
NormalizedWord,
|
|
31
|
+
annotate_transliterations,
|
|
32
|
+
decode_honorific,
|
|
33
|
+
expand_honorifics,
|
|
34
|
+
find_transliteration,
|
|
35
|
+
get_spoken_text,
|
|
36
|
+
has_leading_honorific,
|
|
37
|
+
normalize_text,
|
|
38
|
+
normalize_words,
|
|
39
|
+
)
|
|
40
|
+
from .layout import (
|
|
41
|
+
document_from_aligned_words,
|
|
42
|
+
document_from_markdown,
|
|
43
|
+
document_from_structured,
|
|
44
|
+
)
|
|
45
|
+
from .markdown import (
|
|
46
|
+
EnhancedMarkdownResult,
|
|
47
|
+
build_enhanced_markdown,
|
|
48
|
+
compute_aligned_words_checksum,
|
|
49
|
+
compute_cache_key,
|
|
50
|
+
)
|
|
51
|
+
from .qcf import (
|
|
52
|
+
QCFDecoder,
|
|
53
|
+
QCFVerse,
|
|
54
|
+
QCFWord,
|
|
55
|
+
build_qcf_mapping_from_quran_data,
|
|
56
|
+
detect_qcf_regions,
|
|
57
|
+
extract_qcf_page_number,
|
|
58
|
+
is_qcf_glyph,
|
|
59
|
+
is_qcf_text,
|
|
60
|
+
)
|
|
61
|
+
from .repair import (
|
|
62
|
+
SABON_CHAR_REPAIR,
|
|
63
|
+
SABON_FONT_PREFIXES,
|
|
64
|
+
extract_repairable_font_spans,
|
|
65
|
+
find_font_for_word,
|
|
66
|
+
is_repairable_font,
|
|
67
|
+
repair_text,
|
|
68
|
+
repair_text_for_font,
|
|
69
|
+
repair_words_with_font_info,
|
|
70
|
+
)
|
|
71
|
+
from .routing import (
|
|
72
|
+
EnrichmentDecision,
|
|
73
|
+
PageObservations,
|
|
74
|
+
TaskNeeds,
|
|
75
|
+
observe_from_extraction,
|
|
76
|
+
observe_page,
|
|
77
|
+
route_enrichment,
|
|
78
|
+
)
|
|
79
|
+
from .types import AlignedWord, BlockType, Document, TextBlock, WordBox
|
|
80
|
+
|
|
81
|
+
__all__ = [
|
|
82
|
+
"AlignedWord",
|
|
83
|
+
"BackendConfig",
|
|
84
|
+
"BlockType",
|
|
85
|
+
"Document",
|
|
86
|
+
"EnhancedMarkdownResult",
|
|
87
|
+
"EnrichmentDecision",
|
|
88
|
+
"ExtractResult",
|
|
89
|
+
"HONORIFIC_SYMBOLS",
|
|
90
|
+
"KNOWN_MOJIBAKE_CHARS",
|
|
91
|
+
"MojibakeReport",
|
|
92
|
+
"NormalizedWord",
|
|
93
|
+
"PageObservations",
|
|
94
|
+
"PageProbe",
|
|
95
|
+
"PageType",
|
|
96
|
+
"QCFDecoder",
|
|
97
|
+
"QCFVerse",
|
|
98
|
+
"QCFWord",
|
|
99
|
+
"SABON_CHAR_REPAIR",
|
|
100
|
+
"SABON_FONT_PREFIXES",
|
|
101
|
+
"TaskNeeds",
|
|
102
|
+
"TextBlock",
|
|
103
|
+
"WordBox",
|
|
104
|
+
"annotate_transliterations",
|
|
105
|
+
"build_enhanced_markdown",
|
|
106
|
+
"build_qcf_mapping_from_quran_data",
|
|
107
|
+
"classify_and_select",
|
|
108
|
+
"classify_page",
|
|
109
|
+
"compute_aligned_words_checksum",
|
|
110
|
+
"compute_cache_key",
|
|
111
|
+
"decode_honorific",
|
|
112
|
+
"detect_batch_reversal",
|
|
113
|
+
"detect_mojibake",
|
|
114
|
+
"detect_mojibake_in_pdf",
|
|
115
|
+
"detect_qcf_regions",
|
|
116
|
+
"document_from_aligned_words",
|
|
117
|
+
"document_from_markdown",
|
|
118
|
+
"document_from_structured",
|
|
119
|
+
"expand_honorifics",
|
|
120
|
+
"extract_document",
|
|
121
|
+
"extract_qcf_page_number",
|
|
122
|
+
"extract_repairable_font_spans",
|
|
123
|
+
"find_font_for_word",
|
|
124
|
+
"find_transliteration",
|
|
125
|
+
"get_spoken_text",
|
|
126
|
+
"has_leading_honorific",
|
|
127
|
+
"is_arabic",
|
|
128
|
+
"is_mostly_arabic",
|
|
129
|
+
"is_qcf_glyph",
|
|
130
|
+
"is_qcf_text",
|
|
131
|
+
"is_repairable_font",
|
|
132
|
+
"normalize_text",
|
|
133
|
+
"normalize_words",
|
|
134
|
+
"observe_from_extraction",
|
|
135
|
+
"observe_page",
|
|
136
|
+
"orphan_diacritic_rate",
|
|
137
|
+
"repair_text",
|
|
138
|
+
"repair_text_for_font",
|
|
139
|
+
"repair_words_with_font_info",
|
|
140
|
+
"route_enrichment",
|
|
141
|
+
"select_backend",
|
|
142
|
+
"strip_diacritics",
|
|
143
|
+
"summarize_text_health",
|
|
144
|
+
]
|
versed/_arabic.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Arabic text utilities — normalization, detection, similarity.
|
|
3
|
+
|
|
4
|
+
Internal module used by honorifics.py and detect.py.
|
|
5
|
+
Not part of the public API.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
import unicodedata
|
|
10
|
+
from typing import Dict, Optional
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class TextUtils:
|
|
14
|
+
"""Text processing utilities."""
|
|
15
|
+
|
|
16
|
+
# Arabic character ranges
|
|
17
|
+
ARABIC_PATTERN = re.compile(r'[\u0600-\u06FF\u0750-\u077F\uFB50-\uFDFF\uFE70-\uFEFF]')
|
|
18
|
+
|
|
19
|
+
# Punctuation and diacritics to strip
|
|
20
|
+
PUNCTUATION_PATTERN = re.compile(
|
|
21
|
+
r"[\'\"`´′″‹›«»\u2018\u2019\u201C\u201D\u02BC\u2032\u2033.,;:!?()[\]{}،؛؟–—-]"
|
|
22
|
+
)
|
|
23
|
+
DIACRITICS_PATTERN = re.compile(r'[\u064B-\u065F\u0670]')
|
|
24
|
+
|
|
25
|
+
# Alef variants (أإآٱ) → bare alef (ا)
|
|
26
|
+
ALEF_VARIANTS = re.compile(r'[أإآٱ]')
|
|
27
|
+
# Zero-width and directional markers to drop
|
|
28
|
+
ZERO_WIDTH_TRANSLATION = str.maketrans({
|
|
29
|
+
"\u200B": "", # zero width space
|
|
30
|
+
"\u200C": "", # zero width non-joiner
|
|
31
|
+
"\u200D": "", # zero width joiner
|
|
32
|
+
"\u2060": "", # word joiner
|
|
33
|
+
"\uFEFF": "", # zero width no-break space
|
|
34
|
+
"\u200E": "", # LRM
|
|
35
|
+
"\u200F": "", # RLM
|
|
36
|
+
})
|
|
37
|
+
# Common Latin ligatures (extra safety; NFKC should also handle these)
|
|
38
|
+
LIGATURE_TRANSLATION = str.maketrans({
|
|
39
|
+
"\uFB00": "ff",
|
|
40
|
+
"\uFB01": "fi",
|
|
41
|
+
"\uFB02": "fl",
|
|
42
|
+
"\uFB03": "ffi",
|
|
43
|
+
"\uFB04": "ffl",
|
|
44
|
+
"\uFB05": "st",
|
|
45
|
+
"\uFB06": "st",
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def normalize_arabic(cls, text: str) -> str:
|
|
50
|
+
"""
|
|
51
|
+
Normalize Arabic text for comparison/matching.
|
|
52
|
+
|
|
53
|
+
- Removes diacritics (tashkeel)
|
|
54
|
+
- Normalizes alef variants to bare alef
|
|
55
|
+
- Normalizes teh marbuta to heh
|
|
56
|
+
- Normalizes alef maksura to yeh
|
|
57
|
+
- Removes tatweel (kashida)
|
|
58
|
+
"""
|
|
59
|
+
if not text:
|
|
60
|
+
return ""
|
|
61
|
+
|
|
62
|
+
# Remove tashkeel (Arabic diacritics)
|
|
63
|
+
text = cls.DIACRITICS_PATTERN.sub('', text)
|
|
64
|
+
|
|
65
|
+
# Normalize alef variants to bare alef
|
|
66
|
+
text = cls.ALEF_VARIANTS.sub('ا', text)
|
|
67
|
+
|
|
68
|
+
# Normalize teh marbuta to heh
|
|
69
|
+
text = text.replace('ة', 'ه')
|
|
70
|
+
|
|
71
|
+
# Normalize alef maksura to yeh
|
|
72
|
+
text = text.replace('ى', 'ي')
|
|
73
|
+
|
|
74
|
+
# Remove tatweel (kashida)
|
|
75
|
+
text = text.replace('\u0640', '')
|
|
76
|
+
|
|
77
|
+
return text.strip()
|
|
78
|
+
|
|
79
|
+
@classmethod
|
|
80
|
+
def normalize(cls, text: str) -> str:
|
|
81
|
+
"""
|
|
82
|
+
Normalize text for matching.
|
|
83
|
+
|
|
84
|
+
- Lowercases (for non-Arabic)
|
|
85
|
+
- Strips whitespace
|
|
86
|
+
- Removes punctuation
|
|
87
|
+
- Applies full Arabic normalization (diacritics, alef variants, etc.)
|
|
88
|
+
"""
|
|
89
|
+
if not text:
|
|
90
|
+
return ""
|
|
91
|
+
text = unicodedata.normalize("NFKC", text).strip()
|
|
92
|
+
text = text.translate(cls.ZERO_WIDTH_TRANSLATION)
|
|
93
|
+
text = text.translate(cls.LIGATURE_TRANSLATION)
|
|
94
|
+
text = cls.PUNCTUATION_PATTERN.sub("", text)
|
|
95
|
+
|
|
96
|
+
# Apply full Arabic normalization
|
|
97
|
+
text = cls.normalize_arabic(text)
|
|
98
|
+
|
|
99
|
+
# Lowercase for non-Arabic matching (safe after Arabic normalization)
|
|
100
|
+
text = text.lower()
|
|
101
|
+
|
|
102
|
+
return text.strip()
|
|
103
|
+
|
|
104
|
+
@classmethod
|
|
105
|
+
def is_arabic(cls, text: str) -> bool:
|
|
106
|
+
"""Check if text contains Arabic characters."""
|
|
107
|
+
return bool(cls.ARABIC_PATTERN.search(text)) if text else False
|
|
108
|
+
|
|
109
|
+
@classmethod
|
|
110
|
+
def levenshtein_similarity(cls, s1: str, s2: str) -> float:
|
|
111
|
+
"""
|
|
112
|
+
Calculate similarity score between two strings.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
Similarity score 0.0-1.0 (1.0 = identical)
|
|
116
|
+
"""
|
|
117
|
+
s1, s2 = cls.normalize(s1), cls.normalize(s2)
|
|
118
|
+
if not s1 or not s2:
|
|
119
|
+
return 0.0
|
|
120
|
+
if s1 == s2:
|
|
121
|
+
return 1.0
|
|
122
|
+
|
|
123
|
+
len1, len2 = len(s1), len(s2)
|
|
124
|
+
if len1 > len2:
|
|
125
|
+
s1, s2 = s2, s1
|
|
126
|
+
len1, len2 = len2, len1
|
|
127
|
+
|
|
128
|
+
current_row = list(range(len1 + 1))
|
|
129
|
+
for i in range(1, len2 + 1):
|
|
130
|
+
previous_row, current_row = current_row, [i] + [0] * len1
|
|
131
|
+
for j in range(1, len1 + 1):
|
|
132
|
+
add = previous_row[j] + 1
|
|
133
|
+
delete = current_row[j - 1] + 1
|
|
134
|
+
change = previous_row[j - 1] + (0 if s1[j - 1] == s2[i - 1] else 1)
|
|
135
|
+
current_row[j] = min(add, delete, change)
|
|
136
|
+
|
|
137
|
+
distance = current_row[len1]
|
|
138
|
+
return 1.0 - (distance / max(len1, len2))
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class HonorificsHandler:
|
|
142
|
+
"""
|
|
143
|
+
Handler for Islamic honorific symbols.
|
|
144
|
+
|
|
145
|
+
Maps display symbols (ﷺ, ﷻ, etc.) to spoken Arabic phrases.
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
# Honorific expansions WITH HARAKAT for proper TTS pronunciation
|
|
149
|
+
HONORIFICS: Dict[str, str] = {
|
|
150
|
+
# Prophet Muhammad ﷺ - صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ
|
|
151
|
+
'\uFDFA': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
152
|
+
'\uFD46': 'صَلَّى اللهُ عَلَيْهِ وَآلِهِ وَسَلَّمَ',
|
|
153
|
+
'ﷺ': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
154
|
+
'\uF067': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
155
|
+
'\uF030': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
156
|
+
'\uF031': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
157
|
+
'PBUH': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
158
|
+
'pbuh': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
159
|
+
'(s)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
160
|
+
'(S)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
161
|
+
'(saw)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
162
|
+
'(SAW)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
163
|
+
'(PBUH)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
164
|
+
'(pbuh)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
|
|
165
|
+
|
|
166
|
+
# Allah ﷻ - عَزَّ وَجَلَّ
|
|
167
|
+
'\uFDFB': 'عَزَّ وَجَلَّ',
|
|
168
|
+
'ﷻ': 'عَزَّ وَجَلَّ',
|
|
169
|
+
'\uFBF2': 'عَزَّ وَجَلَّ',
|
|
170
|
+
'\uFBF1': 'عَزَّ وَجَلَّ',
|
|
171
|
+
'\uF063': 'سُبْحَانَهُ وَتَعَالَى',
|
|
172
|
+
'(swt)': 'سُبْحَانَهُ وَتَعَالَى',
|
|
173
|
+
'(SWT)': 'سُبْحَانَهُ وَتَعَالَى',
|
|
174
|
+
|
|
175
|
+
# Other prophets - عَلَيْهِ السَّلَامُ
|
|
176
|
+
'\uF064': 'عَلَيْهِ السَّلَامُ',
|
|
177
|
+
'(as)': 'عَلَيْهِ السَّلَامُ',
|
|
178
|
+
'(AS)': 'عَلَيْهِ السَّلَامُ',
|
|
179
|
+
|
|
180
|
+
# Companions - رَضِيَ اللهُ عَنْهُ
|
|
181
|
+
'\uF065': 'رَضِيَ اللهُ عَنْهُ',
|
|
182
|
+
'(ra)': 'رَضِيَ اللهُ عَنْهُ',
|
|
183
|
+
'(RA)': 'رَضِيَ اللهُ عَنْهُ',
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
SYMBOLS = set(HONORIFICS.keys())
|
|
187
|
+
|
|
188
|
+
@classmethod
|
|
189
|
+
def is_honorific(cls, text: str) -> bool:
|
|
190
|
+
"""Check if text is an honorific symbol."""
|
|
191
|
+
return text in cls.SYMBOLS
|
|
192
|
+
|
|
193
|
+
@classmethod
|
|
194
|
+
def expand(cls, symbol: str) -> Optional[str]:
|
|
195
|
+
"""Expand honorific symbol to spoken Arabic."""
|
|
196
|
+
return cls.HONORIFICS.get(symbol)
|
|
197
|
+
|
|
198
|
+
@classmethod
|
|
199
|
+
def process_text(cls, text: str) -> tuple[str, str]:
|
|
200
|
+
"""
|
|
201
|
+
Process text that may contain honorifics.
|
|
202
|
+
|
|
203
|
+
Returns:
|
|
204
|
+
Tuple of (display_text, spoken_text)
|
|
205
|
+
"""
|
|
206
|
+
display = text
|
|
207
|
+
spoken = text
|
|
208
|
+
|
|
209
|
+
for symbol, expansion in cls.HONORIFICS.items():
|
|
210
|
+
if symbol in text:
|
|
211
|
+
spoken = spoken.replace(symbol, expansion)
|
|
212
|
+
|
|
213
|
+
return display, spoken
|
|
Binary file
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
{
|
|
2
|
+
"words": {
|
|
3
|
+
"prophet": "النبي",
|
|
4
|
+
"allah": "الله",
|
|
5
|
+
"god": "الله",
|
|
6
|
+
"quran": "القرآن",
|
|
7
|
+
"qur'an": "القرآن",
|
|
8
|
+
"hadith": "حديث",
|
|
9
|
+
"sunnah": "سنة",
|
|
10
|
+
"surah": "سورة",
|
|
11
|
+
"sura": "سورة",
|
|
12
|
+
"ayah": "آية",
|
|
13
|
+
"ayat": "آيات",
|
|
14
|
+
"hammazan": "هَمَّاز",
|
|
15
|
+
"lammazan": "لَمَّاز",
|
|
16
|
+
"ayyaban": "عَيَّاب",
|
|
17
|
+
"al-fatiha": "الفاتحة",
|
|
18
|
+
"al-baqarah": "البقرة",
|
|
19
|
+
"al-imran": "آل عمران",
|
|
20
|
+
"al-nisa": "النساء",
|
|
21
|
+
"al-ma'idah": "المائدة",
|
|
22
|
+
"al-an'am": "الأنعام",
|
|
23
|
+
"al-a'raf": "الأعراف",
|
|
24
|
+
"al-anfal": "الأنفال",
|
|
25
|
+
"al-tawbah": "التوبة",
|
|
26
|
+
"al-jathiyah": "الجاثية",
|
|
27
|
+
"al-kahf": "الكهف",
|
|
28
|
+
"al-isra": "الإسراء",
|
|
29
|
+
"al-maryam": "مريم",
|
|
30
|
+
"al-haj": "الحج",
|
|
31
|
+
"al-nur": "النور",
|
|
32
|
+
"al-furqan": "الفرقان",
|
|
33
|
+
"ya-sin": "يس",
|
|
34
|
+
"al-rahman": "الرحمن",
|
|
35
|
+
"al-mulk": "الملك",
|
|
36
|
+
"al-qalam": "القلم"
|
|
37
|
+
},
|
|
38
|
+
"prefixes": {
|
|
39
|
+
"abu": "أبو",
|
|
40
|
+
"ibn": "ابن",
|
|
41
|
+
"umm": "أم",
|
|
42
|
+
"bint": "بنت",
|
|
43
|
+
"al-": "الـ"
|
|
44
|
+
}
|
|
45
|
+
}
|
versed/arabic.py
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"""Portable Arabic text helpers for the public engine."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Iterable
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
ARABIC_DIACRITICS = frozenset(
|
|
9
|
+
{
|
|
10
|
+
"\u064B",
|
|
11
|
+
"\u064C",
|
|
12
|
+
"\u064D",
|
|
13
|
+
"\u064E",
|
|
14
|
+
"\u064F",
|
|
15
|
+
"\u0650",
|
|
16
|
+
"\u0651",
|
|
17
|
+
"\u0652",
|
|
18
|
+
"\u0653",
|
|
19
|
+
"\u0654",
|
|
20
|
+
"\u0655",
|
|
21
|
+
"\u0656",
|
|
22
|
+
"\u0657",
|
|
23
|
+
"\u0658",
|
|
24
|
+
"\u0670",
|
|
25
|
+
}
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
ARABIC_BASE_LETTERS = frozenset(chr(codepoint) for codepoint in range(0x0621, 0x064B))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _iter_arabic_chars(text: str) -> Iterable[str]:
|
|
32
|
+
return (
|
|
33
|
+
char
|
|
34
|
+
for char in text
|
|
35
|
+
if "\u0600" <= char <= "\u06FF" or "\u0750" <= char <= "\u077F"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def is_arabic(text: str) -> bool:
|
|
40
|
+
"""Return True when the string contains Arabic characters."""
|
|
41
|
+
return any(True for _ in _iter_arabic_chars(text))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def is_mostly_arabic(text: str, threshold: float = 0.5) -> bool:
|
|
45
|
+
"""Return True when at least `threshold` of visible chars are Arabic."""
|
|
46
|
+
if not text:
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
visible_chars = [char for char in text if not char.isspace()]
|
|
50
|
+
if not visible_chars:
|
|
51
|
+
return False
|
|
52
|
+
|
|
53
|
+
arabic_chars = [char for char in visible_chars if is_arabic(char)]
|
|
54
|
+
return (len(arabic_chars) / len(visible_chars)) >= threshold
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def strip_diacritics(text: str) -> str:
|
|
58
|
+
"""Remove Arabic diacritics for fuzzy matching."""
|
|
59
|
+
return "".join(char for char in text if char not in ARABIC_DIACRITICS)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def orphan_diacritic_rate(text: str) -> float:
|
|
63
|
+
"""Measure how often diacritics appear without a valid Arabic base letter."""
|
|
64
|
+
if not text:
|
|
65
|
+
return 0.0
|
|
66
|
+
|
|
67
|
+
diacritic_count = 0
|
|
68
|
+
orphan_count = 0
|
|
69
|
+
|
|
70
|
+
for index, char in enumerate(text):
|
|
71
|
+
if char not in ARABIC_DIACRITICS:
|
|
72
|
+
continue
|
|
73
|
+
|
|
74
|
+
diacritic_count += 1
|
|
75
|
+
|
|
76
|
+
cursor = index - 1
|
|
77
|
+
while cursor >= 0 and text[cursor] in ARABIC_DIACRITICS:
|
|
78
|
+
cursor -= 1
|
|
79
|
+
|
|
80
|
+
if cursor < 0:
|
|
81
|
+
orphan_count += 1
|
|
82
|
+
elif text[cursor] not in ARABIC_BASE_LETTERS and not is_arabic(text[cursor]):
|
|
83
|
+
orphan_count += 1
|
|
84
|
+
|
|
85
|
+
return orphan_count / diacritic_count if diacritic_count else 0.0
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def detect_batch_reversal(words_raw: list) -> bool:
|
|
89
|
+
"""Detect likely RTL reversal in a PyMuPDF-style `get_text(\"words\")` batch."""
|
|
90
|
+
arabic_texts = []
|
|
91
|
+
for word in words_raw:
|
|
92
|
+
text = word[4] if len(word) > 4 else ""
|
|
93
|
+
if is_arabic(text) and any(char in ARABIC_DIACRITICS for char in text):
|
|
94
|
+
arabic_texts.append(text)
|
|
95
|
+
|
|
96
|
+
if len(arabic_texts) < 3:
|
|
97
|
+
return False
|
|
98
|
+
|
|
99
|
+
orphan_rates = [orphan_diacritic_rate(text) for text in arabic_texts[:30]]
|
|
100
|
+
|
|
101
|
+
if max(orphan_rates) >= 0.4:
|
|
102
|
+
return True
|
|
103
|
+
|
|
104
|
+
words_with_orphans = sum(1 for rate in orphan_rates if rate > 0.1)
|
|
105
|
+
if words_with_orphans / len(orphan_rates) >= 0.3:
|
|
106
|
+
return True
|
|
107
|
+
|
|
108
|
+
average_rate = sum(orphan_rates) / len(orphan_rates)
|
|
109
|
+
return average_rate > 0.15
|
|
110
|
+
|