versed-pdf 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
versed/__init__.py ADDED
@@ -0,0 +1,144 @@
1
+ """versed — local PDF-to-Markdown tooling for Arabic and bilingual texts."""
2
+
3
+ __version__ = "1.1.0"
4
+
5
+ from .arabic import (
6
+ detect_batch_reversal,
7
+ is_arabic,
8
+ is_mostly_arabic,
9
+ orphan_diacritic_rate,
10
+ strip_diacritics,
11
+ )
12
+ from .classify import (
13
+ BackendConfig,
14
+ PageProbe,
15
+ PageType,
16
+ classify_and_select,
17
+ classify_page,
18
+ select_backend,
19
+ )
20
+ from .detect import (
21
+ KNOWN_MOJIBAKE_CHARS,
22
+ MojibakeReport,
23
+ detect_mojibake,
24
+ detect_mojibake_in_pdf,
25
+ )
26
+ from .extract import ExtractResult, extract_document
27
+ from .health import summarize_text_health
28
+ from .honorifics import (
29
+ HONORIFIC_SYMBOLS,
30
+ NormalizedWord,
31
+ annotate_transliterations,
32
+ decode_honorific,
33
+ expand_honorifics,
34
+ find_transliteration,
35
+ get_spoken_text,
36
+ has_leading_honorific,
37
+ normalize_text,
38
+ normalize_words,
39
+ )
40
+ from .layout import (
41
+ document_from_aligned_words,
42
+ document_from_markdown,
43
+ document_from_structured,
44
+ )
45
+ from .markdown import (
46
+ EnhancedMarkdownResult,
47
+ build_enhanced_markdown,
48
+ compute_aligned_words_checksum,
49
+ compute_cache_key,
50
+ )
51
+ from .qcf import (
52
+ QCFDecoder,
53
+ QCFVerse,
54
+ QCFWord,
55
+ build_qcf_mapping_from_quran_data,
56
+ detect_qcf_regions,
57
+ extract_qcf_page_number,
58
+ is_qcf_glyph,
59
+ is_qcf_text,
60
+ )
61
+ from .repair import (
62
+ SABON_CHAR_REPAIR,
63
+ SABON_FONT_PREFIXES,
64
+ extract_repairable_font_spans,
65
+ find_font_for_word,
66
+ is_repairable_font,
67
+ repair_text,
68
+ repair_text_for_font,
69
+ repair_words_with_font_info,
70
+ )
71
+ from .routing import (
72
+ EnrichmentDecision,
73
+ PageObservations,
74
+ TaskNeeds,
75
+ observe_from_extraction,
76
+ observe_page,
77
+ route_enrichment,
78
+ )
79
+ from .types import AlignedWord, BlockType, Document, TextBlock, WordBox
80
+
81
+ __all__ = [
82
+ "AlignedWord",
83
+ "BackendConfig",
84
+ "BlockType",
85
+ "Document",
86
+ "EnhancedMarkdownResult",
87
+ "EnrichmentDecision",
88
+ "ExtractResult",
89
+ "HONORIFIC_SYMBOLS",
90
+ "KNOWN_MOJIBAKE_CHARS",
91
+ "MojibakeReport",
92
+ "NormalizedWord",
93
+ "PageObservations",
94
+ "PageProbe",
95
+ "PageType",
96
+ "QCFDecoder",
97
+ "QCFVerse",
98
+ "QCFWord",
99
+ "SABON_CHAR_REPAIR",
100
+ "SABON_FONT_PREFIXES",
101
+ "TaskNeeds",
102
+ "TextBlock",
103
+ "WordBox",
104
+ "annotate_transliterations",
105
+ "build_enhanced_markdown",
106
+ "build_qcf_mapping_from_quran_data",
107
+ "classify_and_select",
108
+ "classify_page",
109
+ "compute_aligned_words_checksum",
110
+ "compute_cache_key",
111
+ "decode_honorific",
112
+ "detect_batch_reversal",
113
+ "detect_mojibake",
114
+ "detect_mojibake_in_pdf",
115
+ "detect_qcf_regions",
116
+ "document_from_aligned_words",
117
+ "document_from_markdown",
118
+ "document_from_structured",
119
+ "expand_honorifics",
120
+ "extract_document",
121
+ "extract_qcf_page_number",
122
+ "extract_repairable_font_spans",
123
+ "find_font_for_word",
124
+ "find_transliteration",
125
+ "get_spoken_text",
126
+ "has_leading_honorific",
127
+ "is_arabic",
128
+ "is_mostly_arabic",
129
+ "is_qcf_glyph",
130
+ "is_qcf_text",
131
+ "is_repairable_font",
132
+ "normalize_text",
133
+ "normalize_words",
134
+ "observe_from_extraction",
135
+ "observe_page",
136
+ "orphan_diacritic_rate",
137
+ "repair_text",
138
+ "repair_text_for_font",
139
+ "repair_words_with_font_info",
140
+ "route_enrichment",
141
+ "select_backend",
142
+ "strip_diacritics",
143
+ "summarize_text_health",
144
+ ]
versed/_arabic.py ADDED
@@ -0,0 +1,213 @@
1
+ """
2
+ Arabic text utilities — normalization, detection, similarity.
3
+
4
+ Internal module used by honorifics.py and detect.py.
5
+ Not part of the public API.
6
+ """
7
+
8
+ import re
9
+ import unicodedata
10
+ from typing import Dict, Optional
11
+
12
+
13
+ class TextUtils:
14
+ """Text processing utilities."""
15
+
16
+ # Arabic character ranges
17
+ ARABIC_PATTERN = re.compile(r'[\u0600-\u06FF\u0750-\u077F\uFB50-\uFDFF\uFE70-\uFEFF]')
18
+
19
+ # Punctuation and diacritics to strip
20
+ PUNCTUATION_PATTERN = re.compile(
21
+ r"[\'\"`´′″‹›«»\u2018\u2019\u201C\u201D\u02BC\u2032\u2033.,;:!?()[\]{}،؛؟–—-]"
22
+ )
23
+ DIACRITICS_PATTERN = re.compile(r'[\u064B-\u065F\u0670]')
24
+
25
+ # Alef variants (أإآٱ) → bare alef (ا)
26
+ ALEF_VARIANTS = re.compile(r'[أإآٱ]')
27
+ # Zero-width and directional markers to drop
28
+ ZERO_WIDTH_TRANSLATION = str.maketrans({
29
+ "\u200B": "", # zero width space
30
+ "\u200C": "", # zero width non-joiner
31
+ "\u200D": "", # zero width joiner
32
+ "\u2060": "", # word joiner
33
+ "\uFEFF": "", # zero width no-break space
34
+ "\u200E": "", # LRM
35
+ "\u200F": "", # RLM
36
+ })
37
+ # Common Latin ligatures (extra safety; NFKC should also handle these)
38
+ LIGATURE_TRANSLATION = str.maketrans({
39
+ "\uFB00": "ff",
40
+ "\uFB01": "fi",
41
+ "\uFB02": "fl",
42
+ "\uFB03": "ffi",
43
+ "\uFB04": "ffl",
44
+ "\uFB05": "st",
45
+ "\uFB06": "st",
46
+ })
47
+
48
+ @classmethod
49
+ def normalize_arabic(cls, text: str) -> str:
50
+ """
51
+ Normalize Arabic text for comparison/matching.
52
+
53
+ - Removes diacritics (tashkeel)
54
+ - Normalizes alef variants to bare alef
55
+ - Normalizes teh marbuta to heh
56
+ - Normalizes alef maksura to yeh
57
+ - Removes tatweel (kashida)
58
+ """
59
+ if not text:
60
+ return ""
61
+
62
+ # Remove tashkeel (Arabic diacritics)
63
+ text = cls.DIACRITICS_PATTERN.sub('', text)
64
+
65
+ # Normalize alef variants to bare alef
66
+ text = cls.ALEF_VARIANTS.sub('ا', text)
67
+
68
+ # Normalize teh marbuta to heh
69
+ text = text.replace('ة', 'ه')
70
+
71
+ # Normalize alef maksura to yeh
72
+ text = text.replace('ى', 'ي')
73
+
74
+ # Remove tatweel (kashida)
75
+ text = text.replace('\u0640', '')
76
+
77
+ return text.strip()
78
+
79
+ @classmethod
80
+ def normalize(cls, text: str) -> str:
81
+ """
82
+ Normalize text for matching.
83
+
84
+ - Lowercases (for non-Arabic)
85
+ - Strips whitespace
86
+ - Removes punctuation
87
+ - Applies full Arabic normalization (diacritics, alef variants, etc.)
88
+ """
89
+ if not text:
90
+ return ""
91
+ text = unicodedata.normalize("NFKC", text).strip()
92
+ text = text.translate(cls.ZERO_WIDTH_TRANSLATION)
93
+ text = text.translate(cls.LIGATURE_TRANSLATION)
94
+ text = cls.PUNCTUATION_PATTERN.sub("", text)
95
+
96
+ # Apply full Arabic normalization
97
+ text = cls.normalize_arabic(text)
98
+
99
+ # Lowercase for non-Arabic matching (safe after Arabic normalization)
100
+ text = text.lower()
101
+
102
+ return text.strip()
103
+
104
+ @classmethod
105
+ def is_arabic(cls, text: str) -> bool:
106
+ """Check if text contains Arabic characters."""
107
+ return bool(cls.ARABIC_PATTERN.search(text)) if text else False
108
+
109
+ @classmethod
110
+ def levenshtein_similarity(cls, s1: str, s2: str) -> float:
111
+ """
112
+ Calculate similarity score between two strings.
113
+
114
+ Returns:
115
+ Similarity score 0.0-1.0 (1.0 = identical)
116
+ """
117
+ s1, s2 = cls.normalize(s1), cls.normalize(s2)
118
+ if not s1 or not s2:
119
+ return 0.0
120
+ if s1 == s2:
121
+ return 1.0
122
+
123
+ len1, len2 = len(s1), len(s2)
124
+ if len1 > len2:
125
+ s1, s2 = s2, s1
126
+ len1, len2 = len2, len1
127
+
128
+ current_row = list(range(len1 + 1))
129
+ for i in range(1, len2 + 1):
130
+ previous_row, current_row = current_row, [i] + [0] * len1
131
+ for j in range(1, len1 + 1):
132
+ add = previous_row[j] + 1
133
+ delete = current_row[j - 1] + 1
134
+ change = previous_row[j - 1] + (0 if s1[j - 1] == s2[i - 1] else 1)
135
+ current_row[j] = min(add, delete, change)
136
+
137
+ distance = current_row[len1]
138
+ return 1.0 - (distance / max(len1, len2))
139
+
140
+
141
+ class HonorificsHandler:
142
+ """
143
+ Handler for Islamic honorific symbols.
144
+
145
+ Maps display symbols (ﷺ, ﷻ, etc.) to spoken Arabic phrases.
146
+ """
147
+
148
+ # Honorific expansions WITH HARAKAT for proper TTS pronunciation
149
+ HONORIFICS: Dict[str, str] = {
150
+ # Prophet Muhammad ﷺ - صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ
151
+ '\uFDFA': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
152
+ '\uFD46': 'صَلَّى اللهُ عَلَيْهِ وَآلِهِ وَسَلَّمَ',
153
+ 'ﷺ': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
154
+ '\uF067': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
155
+ '\uF030': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
156
+ '\uF031': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
157
+ 'PBUH': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
158
+ 'pbuh': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
159
+ '(s)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
160
+ '(S)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
161
+ '(saw)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
162
+ '(SAW)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
163
+ '(PBUH)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
164
+ '(pbuh)': 'صَلَّى اللهُ عَلَيْهِ وَسَلَّمَ',
165
+
166
+ # Allah ﷻ - عَزَّ وَجَلَّ
167
+ '\uFDFB': 'عَزَّ وَجَلَّ',
168
+ 'ﷻ': 'عَزَّ وَجَلَّ',
169
+ '\uFBF2': 'عَزَّ وَجَلَّ',
170
+ '\uFBF1': 'عَزَّ وَجَلَّ',
171
+ '\uF063': 'سُبْحَانَهُ وَتَعَالَى',
172
+ '(swt)': 'سُبْحَانَهُ وَتَعَالَى',
173
+ '(SWT)': 'سُبْحَانَهُ وَتَعَالَى',
174
+
175
+ # Other prophets - عَلَيْهِ السَّلَامُ
176
+ '\uF064': 'عَلَيْهِ السَّلَامُ',
177
+ '(as)': 'عَلَيْهِ السَّلَامُ',
178
+ '(AS)': 'عَلَيْهِ السَّلَامُ',
179
+
180
+ # Companions - رَضِيَ اللهُ عَنْهُ
181
+ '\uF065': 'رَضِيَ اللهُ عَنْهُ',
182
+ '(ra)': 'رَضِيَ اللهُ عَنْهُ',
183
+ '(RA)': 'رَضِيَ اللهُ عَنْهُ',
184
+ }
185
+
186
+ SYMBOLS = set(HONORIFICS.keys())
187
+
188
+ @classmethod
189
+ def is_honorific(cls, text: str) -> bool:
190
+ """Check if text is an honorific symbol."""
191
+ return text in cls.SYMBOLS
192
+
193
+ @classmethod
194
+ def expand(cls, symbol: str) -> Optional[str]:
195
+ """Expand honorific symbol to spoken Arabic."""
196
+ return cls.HONORIFICS.get(symbol)
197
+
198
+ @classmethod
199
+ def process_text(cls, text: str) -> tuple[str, str]:
200
+ """
201
+ Process text that may contain honorifics.
202
+
203
+ Returns:
204
+ Tuple of (display_text, spoken_text)
205
+ """
206
+ display = text
207
+ spoken = text
208
+
209
+ for symbol, expansion in cls.HONORIFICS.items():
210
+ if symbol in text:
211
+ spoken = spoken.replace(symbol, expansion)
212
+
213
+ return display, spoken
Binary file
@@ -0,0 +1,45 @@
1
+ {
2
+ "words": {
3
+ "prophet": "النبي",
4
+ "allah": "الله",
5
+ "god": "الله",
6
+ "quran": "القرآن",
7
+ "qur'an": "القرآن",
8
+ "hadith": "حديث",
9
+ "sunnah": "سنة",
10
+ "surah": "سورة",
11
+ "sura": "سورة",
12
+ "ayah": "آية",
13
+ "ayat": "آيات",
14
+ "hammazan": "هَمَّاز",
15
+ "lammazan": "لَمَّاز",
16
+ "ayyaban": "عَيَّاب",
17
+ "al-fatiha": "الفاتحة",
18
+ "al-baqarah": "البقرة",
19
+ "al-imran": "آل عمران",
20
+ "al-nisa": "النساء",
21
+ "al-ma'idah": "المائدة",
22
+ "al-an'am": "الأنعام",
23
+ "al-a'raf": "الأعراف",
24
+ "al-anfal": "الأنفال",
25
+ "al-tawbah": "التوبة",
26
+ "al-jathiyah": "الجاثية",
27
+ "al-kahf": "الكهف",
28
+ "al-isra": "الإسراء",
29
+ "al-maryam": "مريم",
30
+ "al-haj": "الحج",
31
+ "al-nur": "النور",
32
+ "al-furqan": "الفرقان",
33
+ "ya-sin": "يس",
34
+ "al-rahman": "الرحمن",
35
+ "al-mulk": "الملك",
36
+ "al-qalam": "القلم"
37
+ },
38
+ "prefixes": {
39
+ "abu": "أبو",
40
+ "ibn": "ابن",
41
+ "umm": "أم",
42
+ "bint": "بنت",
43
+ "al-": "الـ"
44
+ }
45
+ }
versed/arabic.py ADDED
@@ -0,0 +1,110 @@
1
+ """Portable Arabic text helpers for the public engine."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Iterable
6
+
7
+
8
+ ARABIC_DIACRITICS = frozenset(
9
+ {
10
+ "\u064B",
11
+ "\u064C",
12
+ "\u064D",
13
+ "\u064E",
14
+ "\u064F",
15
+ "\u0650",
16
+ "\u0651",
17
+ "\u0652",
18
+ "\u0653",
19
+ "\u0654",
20
+ "\u0655",
21
+ "\u0656",
22
+ "\u0657",
23
+ "\u0658",
24
+ "\u0670",
25
+ }
26
+ )
27
+
28
+ ARABIC_BASE_LETTERS = frozenset(chr(codepoint) for codepoint in range(0x0621, 0x064B))
29
+
30
+
31
+ def _iter_arabic_chars(text: str) -> Iterable[str]:
32
+ return (
33
+ char
34
+ for char in text
35
+ if "\u0600" <= char <= "\u06FF" or "\u0750" <= char <= "\u077F"
36
+ )
37
+
38
+
39
+ def is_arabic(text: str) -> bool:
40
+ """Return True when the string contains Arabic characters."""
41
+ return any(True for _ in _iter_arabic_chars(text))
42
+
43
+
44
+ def is_mostly_arabic(text: str, threshold: float = 0.5) -> bool:
45
+ """Return True when at least `threshold` of visible chars are Arabic."""
46
+ if not text:
47
+ return False
48
+
49
+ visible_chars = [char for char in text if not char.isspace()]
50
+ if not visible_chars:
51
+ return False
52
+
53
+ arabic_chars = [char for char in visible_chars if is_arabic(char)]
54
+ return (len(arabic_chars) / len(visible_chars)) >= threshold
55
+
56
+
57
+ def strip_diacritics(text: str) -> str:
58
+ """Remove Arabic diacritics for fuzzy matching."""
59
+ return "".join(char for char in text if char not in ARABIC_DIACRITICS)
60
+
61
+
62
+ def orphan_diacritic_rate(text: str) -> float:
63
+ """Measure how often diacritics appear without a valid Arabic base letter."""
64
+ if not text:
65
+ return 0.0
66
+
67
+ diacritic_count = 0
68
+ orphan_count = 0
69
+
70
+ for index, char in enumerate(text):
71
+ if char not in ARABIC_DIACRITICS:
72
+ continue
73
+
74
+ diacritic_count += 1
75
+
76
+ cursor = index - 1
77
+ while cursor >= 0 and text[cursor] in ARABIC_DIACRITICS:
78
+ cursor -= 1
79
+
80
+ if cursor < 0:
81
+ orphan_count += 1
82
+ elif text[cursor] not in ARABIC_BASE_LETTERS and not is_arabic(text[cursor]):
83
+ orphan_count += 1
84
+
85
+ return orphan_count / diacritic_count if diacritic_count else 0.0
86
+
87
+
88
+ def detect_batch_reversal(words_raw: list) -> bool:
89
+ """Detect likely RTL reversal in a PyMuPDF-style `get_text(\"words\")` batch."""
90
+ arabic_texts = []
91
+ for word in words_raw:
92
+ text = word[4] if len(word) > 4 else ""
93
+ if is_arabic(text) and any(char in ARABIC_DIACRITICS for char in text):
94
+ arabic_texts.append(text)
95
+
96
+ if len(arabic_texts) < 3:
97
+ return False
98
+
99
+ orphan_rates = [orphan_diacritic_rate(text) for text in arabic_texts[:30]]
100
+
101
+ if max(orphan_rates) >= 0.4:
102
+ return True
103
+
104
+ words_with_orphans = sum(1 for rate in orphan_rates if rate > 0.1)
105
+ if words_with_orphans / len(orphan_rates) >= 0.3:
106
+ return True
107
+
108
+ average_rate = sum(orphan_rates) / len(orphan_rates)
109
+ return average_rate > 0.15
110
+