arafix 0.8.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- arafix/__init__.py +210 -0
- arafix/adapters.py +83 -0
- arafix/cli.py +288 -0
- arafix/cmap.py +194 -0
- arafix/diagnose.py +417 -0
- arafix/evaluate.py +358 -0
- arafix/extractors/__init__.py +45 -0
- arafix/extractors/base.py +58 -0
- arafix/extractors/pymupdf_extractor.py +164 -0
- arafix/hygiene.py +164 -0
- arafix/integrations/__init__.py +49 -0
- arafix/integrations/markitdown_plugin.py +136 -0
- arafix/lamalef.py +231 -0
- arafix/layout.py +583 -0
- arafix/normalize.py +184 -0
- arafix/order.py +172 -0
- arafix/pipeline.py +537 -0
- arafix/py.typed +0 -0
- arafix/types.py +232 -0
- arafix/unicode_tables.py +271 -0
- arafix-0.8.0.dist-info/METADATA +645 -0
- arafix-0.8.0.dist-info/RECORD +25 -0
- arafix-0.8.0.dist-info/WHEEL +4 -0
- arafix-0.8.0.dist-info/entry_points.txt +5 -0
- arafix-0.8.0.dist-info/licenses/LICENSE +21 -0
arafix/__init__.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""
|
|
2
|
+
arafix — recover broken Arabic text from PDFs.
|
|
3
|
+
|
|
4
|
+
Graded ladder (not one hammer)::
|
|
5
|
+
|
|
6
|
+
0 diagnose() know before you fix
|
|
7
|
+
1a fold_simple_forms() presentation forms; keep ligatures atomic
|
|
8
|
+
2 fix_order() visual → logical, protect LTR runs
|
|
9
|
+
1b expand_ligatures() ﻻ → لا after order is stable
|
|
10
|
+
3 build_glyph_map() rebuild from embedded font
|
|
11
|
+
4 OCR last resort (not shipped)
|
|
12
|
+
|
|
13
|
+
Quick start::
|
|
14
|
+
|
|
15
|
+
>>> from arafix import repair_text
|
|
16
|
+
>>> repair_text("\ufee3\ufeae\ufea3\ufe92\ufe8e").text
|
|
17
|
+
'مرحبا'
|
|
18
|
+
|
|
19
|
+
>>> from arafix import repair_blocks
|
|
20
|
+
>>> repair_blocks(["\ufee3\ufeae\ufea3\ufe92\ufe8e"]).texts[0]
|
|
21
|
+
'مرحبا'
|
|
22
|
+
|
|
23
|
+
>>> from arafix import extract_pdf # doctest: +SKIP
|
|
24
|
+
>>> doc = extract_pdf("thesis.pdf") # doctest: +SKIP
|
|
25
|
+
|
|
26
|
+
MIT license. Primary long-form docs are in Arabic (README).
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
__version__ = "0.8.0"
|
|
32
|
+
__license__ = "MIT"
|
|
33
|
+
|
|
34
|
+
from .adapters import as_blocks, fix_any, fix_markitdown, fix_table
|
|
35
|
+
from .cmap import GlyphMap, build_glyph_map, decode_glyph_name
|
|
36
|
+
from .diagnose import (
|
|
37
|
+
DEFAULT_THRESHOLDS,
|
|
38
|
+
detect_mojibake,
|
|
39
|
+
detect_presentation_forms,
|
|
40
|
+
detect_pua,
|
|
41
|
+
detect_visual_order,
|
|
42
|
+
diagnose,
|
|
43
|
+
)
|
|
44
|
+
from .evaluate import (
|
|
45
|
+
EvalConfig,
|
|
46
|
+
EvalReport,
|
|
47
|
+
cer,
|
|
48
|
+
compare_extractors,
|
|
49
|
+
evaluate_pdf,
|
|
50
|
+
evaluate_text,
|
|
51
|
+
levenshtein,
|
|
52
|
+
levenshtein_reference,
|
|
53
|
+
wer,
|
|
54
|
+
)
|
|
55
|
+
from .extractors import Extractor, RawPage, get_extractor, register
|
|
56
|
+
from .hygiene import (
|
|
57
|
+
count_artifacts,
|
|
58
|
+
fold_arabic_punct_confusables,
|
|
59
|
+
sanitize_extraction,
|
|
60
|
+
)
|
|
61
|
+
from .lamalef import (
|
|
62
|
+
LamAlefReport,
|
|
63
|
+
detect_lam_alef_transposition,
|
|
64
|
+
repair_lam_alef_transposition,
|
|
65
|
+
)
|
|
66
|
+
from .layout import (
|
|
67
|
+
Glyph,
|
|
68
|
+
LayoutColumn,
|
|
69
|
+
LayoutConfig,
|
|
70
|
+
LayoutLine,
|
|
71
|
+
LayoutTable,
|
|
72
|
+
PageLayout,
|
|
73
|
+
analyze_layout,
|
|
74
|
+
cluster_to_lines,
|
|
75
|
+
table_to_markdown,
|
|
76
|
+
)
|
|
77
|
+
from .normalize import (
|
|
78
|
+
NormalizeConfig,
|
|
79
|
+
expand_deferred_forms,
|
|
80
|
+
expand_ligatures,
|
|
81
|
+
fold_presentation_forms,
|
|
82
|
+
fold_simple_forms,
|
|
83
|
+
normalize_text,
|
|
84
|
+
)
|
|
85
|
+
from .order import (
|
|
86
|
+
MIRROR_PAIRS,
|
|
87
|
+
ReorderConfig,
|
|
88
|
+
fix_order,
|
|
89
|
+
grapheme_clusters,
|
|
90
|
+
reverse_visual_line,
|
|
91
|
+
)
|
|
92
|
+
from .pipeline import (
|
|
93
|
+
PipelineConfig,
|
|
94
|
+
extract_pdf,
|
|
95
|
+
harvest_document_lexicon,
|
|
96
|
+
repair_blocks,
|
|
97
|
+
repair_text,
|
|
98
|
+
)
|
|
99
|
+
from .types import (
|
|
100
|
+
BlockResult,
|
|
101
|
+
BlocksResult,
|
|
102
|
+
Defect,
|
|
103
|
+
Diagnosis,
|
|
104
|
+
DocumentResult,
|
|
105
|
+
Evidence,
|
|
106
|
+
PageResult,
|
|
107
|
+
RepairResult,
|
|
108
|
+
Stage,
|
|
109
|
+
TextBlock,
|
|
110
|
+
)
|
|
111
|
+
from .unicode_tables import (
|
|
112
|
+
DEFERRED_PF_TO_BASE,
|
|
113
|
+
LIGATURE_PF_TO_BASE,
|
|
114
|
+
PF_TO_BASE,
|
|
115
|
+
SIMPLE_PF_TO_BASE,
|
|
116
|
+
SPACING_MARK_PF_TO_BASE,
|
|
117
|
+
JoiningForm,
|
|
118
|
+
unicode_version,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
__all__ = [
|
|
122
|
+
"__version__",
|
|
123
|
+
# الأنبوب
|
|
124
|
+
"repair_text",
|
|
125
|
+
"repair_blocks",
|
|
126
|
+
"extract_pdf",
|
|
127
|
+
"PipelineConfig",
|
|
128
|
+
"harvest_document_lexicon",
|
|
129
|
+
# مهايئات
|
|
130
|
+
"fix_any",
|
|
131
|
+
"fix_markitdown",
|
|
132
|
+
"fix_table",
|
|
133
|
+
"as_blocks",
|
|
134
|
+
# نظافة الاستخراج
|
|
135
|
+
"sanitize_extraction",
|
|
136
|
+
"count_artifacts",
|
|
137
|
+
"fold_arabic_punct_confusables",
|
|
138
|
+
# البنية
|
|
139
|
+
"Glyph",
|
|
140
|
+
"LayoutLine",
|
|
141
|
+
"LayoutColumn",
|
|
142
|
+
"LayoutTable",
|
|
143
|
+
"PageLayout",
|
|
144
|
+
"LayoutConfig",
|
|
145
|
+
"analyze_layout",
|
|
146
|
+
"cluster_to_lines",
|
|
147
|
+
"table_to_markdown",
|
|
148
|
+
# الدرجة ٠
|
|
149
|
+
"diagnose",
|
|
150
|
+
"detect_mojibake",
|
|
151
|
+
"detect_presentation_forms",
|
|
152
|
+
"detect_pua",
|
|
153
|
+
"detect_visual_order",
|
|
154
|
+
"DEFAULT_THRESHOLDS",
|
|
155
|
+
# الدرجة ١ (تمريرتان: مفردات ← اتجاه ← رباطات)
|
|
156
|
+
"normalize_text",
|
|
157
|
+
"fold_presentation_forms",
|
|
158
|
+
"fold_simple_forms",
|
|
159
|
+
"expand_deferred_forms",
|
|
160
|
+
"expand_ligatures",
|
|
161
|
+
"NormalizeConfig",
|
|
162
|
+
# لام-ألف
|
|
163
|
+
"detect_lam_alef_transposition",
|
|
164
|
+
"repair_lam_alef_transposition",
|
|
165
|
+
"LamAlefReport",
|
|
166
|
+
# الدرجة ٢
|
|
167
|
+
"fix_order",
|
|
168
|
+
"reverse_visual_line",
|
|
169
|
+
"grapheme_clusters",
|
|
170
|
+
"MIRROR_PAIRS",
|
|
171
|
+
"ReorderConfig",
|
|
172
|
+
# الدرجة ٣
|
|
173
|
+
"build_glyph_map",
|
|
174
|
+
"decode_glyph_name",
|
|
175
|
+
"GlyphMap",
|
|
176
|
+
# النماذج
|
|
177
|
+
"Defect",
|
|
178
|
+
"Stage",
|
|
179
|
+
"Evidence",
|
|
180
|
+
"Diagnosis",
|
|
181
|
+
"RepairResult",
|
|
182
|
+
"PageResult",
|
|
183
|
+
"DocumentResult",
|
|
184
|
+
"TextBlock",
|
|
185
|
+
"BlockResult",
|
|
186
|
+
"BlocksResult",
|
|
187
|
+
# القياس
|
|
188
|
+
"evaluate_text",
|
|
189
|
+
"evaluate_pdf",
|
|
190
|
+
"compare_extractors",
|
|
191
|
+
"cer",
|
|
192
|
+
"wer",
|
|
193
|
+
"levenshtein",
|
|
194
|
+
"levenshtein_reference",
|
|
195
|
+
"EvalConfig",
|
|
196
|
+
"EvalReport",
|
|
197
|
+
# المحرّكات
|
|
198
|
+
"Extractor",
|
|
199
|
+
"RawPage",
|
|
200
|
+
"get_extractor",
|
|
201
|
+
"register",
|
|
202
|
+
# الجداول
|
|
203
|
+
"PF_TO_BASE",
|
|
204
|
+
"SIMPLE_PF_TO_BASE",
|
|
205
|
+
"DEFERRED_PF_TO_BASE",
|
|
206
|
+
"LIGATURE_PF_TO_BASE",
|
|
207
|
+
"SPACING_MARK_PF_TO_BASE",
|
|
208
|
+
"JoiningForm",
|
|
209
|
+
"unicode_version",
|
|
210
|
+
]
|
arafix/adapters.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""
|
|
2
|
+
مهايئات خفيفة — مسارٌ واحد من «نصٍّ مستخرج بأيّ أداة» إلى عربيّ سليم.
|
|
3
|
+
|
|
4
|
+
from arafix.adapters import fix_markitdown, fix_any
|
|
5
|
+
|
|
6
|
+
md = MarkItDown().convert("f.pdf")
|
|
7
|
+
print(fix_markitdown(md).text)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from .pipeline import PipelineConfig, repair_blocks, repair_text
|
|
15
|
+
from .types import BlocksResult, RepairResult, TextBlock
|
|
16
|
+
|
|
17
|
+
__all__ = ["fix_any", "fix_markitdown", "fix_table", "as_blocks"]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def fix_any(text: str, config: PipelineConfig | None = None) -> RepairResult:
|
|
21
|
+
"""أيّ نصٍّ — من pdfminer أو المتصفح أو الحافظة."""
|
|
22
|
+
return repair_text(text, config)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def fix_markitdown(
|
|
26
|
+
result: Any,
|
|
27
|
+
config: PipelineConfig | None = None,
|
|
28
|
+
) -> RepairResult:
|
|
29
|
+
"""
|
|
30
|
+
يقبل ``DocumentConverterResult`` من MarkItDown (أو أيّ كائن فيه
|
|
31
|
+
``text_content`` / ``markdown``).
|
|
32
|
+
"""
|
|
33
|
+
text = (
|
|
34
|
+
getattr(result, "text_content", None)
|
|
35
|
+
or getattr(result, "markdown", None)
|
|
36
|
+
or str(result)
|
|
37
|
+
)
|
|
38
|
+
return repair_text(text, config)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def as_blocks(
|
|
42
|
+
rows: list[list[str]],
|
|
43
|
+
*,
|
|
44
|
+
id_prefix: str = "r",
|
|
45
|
+
) -> list[TextBlock]:
|
|
46
|
+
""" يحوّل جدولاً (قائمة صفوف) إلى كتل ``TextBlock`` بخلايا مستقلة."""
|
|
47
|
+
blocks: list[TextBlock] = []
|
|
48
|
+
for i, row in enumerate(rows):
|
|
49
|
+
for j, cell in enumerate(row):
|
|
50
|
+
blocks.append(
|
|
51
|
+
TextBlock(
|
|
52
|
+
text=cell or "",
|
|
53
|
+
id=f"{id_prefix}{i}c{j}",
|
|
54
|
+
role="cell",
|
|
55
|
+
meta={"row": i, "col": j},
|
|
56
|
+
)
|
|
57
|
+
)
|
|
58
|
+
return blocks
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def fix_table(
|
|
62
|
+
rows: list[list[str]],
|
|
63
|
+
config: PipelineConfig | None = None,
|
|
64
|
+
) -> list[list[str]]:
|
|
65
|
+
"""
|
|
66
|
+
يصلح خلايا جدولٍ كلٌّ على حدة ويُرجع نفس الشكل.
|
|
67
|
+
|
|
68
|
+
>>> fix_table([["\ufee3\ufeae\ufea3\ufe92\ufe8e", "OK"]])
|
|
69
|
+
[['مرحبا', 'OK']]
|
|
70
|
+
"""
|
|
71
|
+
if not rows:
|
|
72
|
+
return []
|
|
73
|
+
blocks = as_blocks(rows)
|
|
74
|
+
repaired: BlocksResult = repair_blocks(blocks, config)
|
|
75
|
+
by_id = repaired.by_id()
|
|
76
|
+
out: list[list[str]] = []
|
|
77
|
+
for i, row in enumerate(rows):
|
|
78
|
+
new_row = []
|
|
79
|
+
for j, _ in enumerate(row):
|
|
80
|
+
key = f"r{i}c{j}"
|
|
81
|
+
new_row.append(by_id[key].text if key in by_id else "")
|
|
82
|
+
out.append(new_row)
|
|
83
|
+
return out
|
arafix/cli.py
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""
|
|
2
|
+
واجهة سطر الأوامر.
|
|
3
|
+
|
|
4
|
+
arafix diagnose thesis.pdf
|
|
5
|
+
arafix extract thesis.pdf -o out.txt
|
|
6
|
+
arafix eval thesis.pdf --truth thesis.txt --compare
|
|
7
|
+
arafix text "ﺎﺒﺣﺮﻣ"
|
|
8
|
+
arafix fonts thesis.pdf
|
|
9
|
+
|
|
10
|
+
فلسفة الأمر `diagnose` أنه **لا يكتب شيئاً**. اقرأ تقريره أولاً، ثم
|
|
11
|
+
قرّر. الأداة التي تعالج قبل أن تُريك ما وجدت أداةٌ لا تُؤتمن.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
from . import __version__
|
|
21
|
+
from .diagnose import diagnose
|
|
22
|
+
from .pipeline import PipelineConfig, extract_pdf, repair_text
|
|
23
|
+
from .unicode_tables import unicode_version
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _cmd_diagnose(args: argparse.Namespace) -> int:
|
|
27
|
+
from .extractors import get_extractor
|
|
28
|
+
|
|
29
|
+
ex = get_extractor(args.extractor)
|
|
30
|
+
report = []
|
|
31
|
+
for raw in ex.pages(args.path):
|
|
32
|
+
dg = diagnose(raw.text)
|
|
33
|
+
report.append(
|
|
34
|
+
{
|
|
35
|
+
"page": raw.number,
|
|
36
|
+
"chars": dg.char_count,
|
|
37
|
+
"arabic_ratio": round(dg.arabic_ratio, 3),
|
|
38
|
+
"defects": [d.value for d in dg.defects],
|
|
39
|
+
"confidence": dg.confidence,
|
|
40
|
+
"defect_confidence": {
|
|
41
|
+
k.value: v for k, v in dg.defect_confidence.items()
|
|
42
|
+
},
|
|
43
|
+
"fonts": raw.fonts,
|
|
44
|
+
"evidence": [
|
|
45
|
+
{"name": e.name, "value": round(e.value, 3), "detail": e.detail}
|
|
46
|
+
for e in dg.evidence
|
|
47
|
+
],
|
|
48
|
+
}
|
|
49
|
+
)
|
|
50
|
+
if args.pages and raw.number >= args.pages:
|
|
51
|
+
break
|
|
52
|
+
|
|
53
|
+
if args.json:
|
|
54
|
+
print(json.dumps(report, ensure_ascii=False, indent=2))
|
|
55
|
+
return 0
|
|
56
|
+
|
|
57
|
+
for r in report:
|
|
58
|
+
print(f"── صفحة {r['page']} " + "─" * 40)
|
|
59
|
+
conf = r["defect_confidence"]
|
|
60
|
+
detail = "، ".join(f"{d} ({conf.get(d, 0):.2f})" for d in r["defects"])
|
|
61
|
+
print(f" العلل : {detail}")
|
|
62
|
+
print(f" الثقة : {r['confidence']} (أضعف حلقة)")
|
|
63
|
+
print(f" الحروف : {r['chars']} (عربية {r['arabic_ratio']:.0%})")
|
|
64
|
+
if r["fonts"]:
|
|
65
|
+
print(f" الخطوط : {', '.join(r['fonts'][:4])}")
|
|
66
|
+
if args.verbose:
|
|
67
|
+
for e in r["evidence"]:
|
|
68
|
+
print(f" · {e['name']:22s} {e['value']:+.3f} {e['detail']}")
|
|
69
|
+
return 0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _cmd_extract(args: argparse.Namespace) -> int:
|
|
73
|
+
from .layout import LayoutConfig
|
|
74
|
+
|
|
75
|
+
lay_cfg = LayoutConfig(reading_order=args.reading_order)
|
|
76
|
+
cfg = PipelineConfig(
|
|
77
|
+
extractor=args.extractor,
|
|
78
|
+
force_reorder=args.force_reorder,
|
|
79
|
+
layout=args.layout,
|
|
80
|
+
layout_config=lay_cfg,
|
|
81
|
+
)
|
|
82
|
+
doc = extract_pdf(args.path, cfg)
|
|
83
|
+
|
|
84
|
+
out = doc.text
|
|
85
|
+
if args.output:
|
|
86
|
+
with open(args.output, "w", encoding="utf-8") as fh:
|
|
87
|
+
fh.write(out)
|
|
88
|
+
print(f"كُتب {len(out)} حرفاً في {args.output}", file=sys.stderr)
|
|
89
|
+
else:
|
|
90
|
+
print(out)
|
|
91
|
+
|
|
92
|
+
print(f"الثقة الدنيا عبر {len(doc.pages)} صفحة: {doc.confidence}", file=sys.stderr)
|
|
93
|
+
if args.verbose:
|
|
94
|
+
print(
|
|
95
|
+
f"بنية: أعمدة≤{doc.metadata.get('max_columns', 1)} "
|
|
96
|
+
f"جداول={doc.metadata.get('table_count', 0)} "
|
|
97
|
+
f"layout={doc.metadata.get('layout')}",
|
|
98
|
+
file=sys.stderr,
|
|
99
|
+
)
|
|
100
|
+
for p in doc.pages:
|
|
101
|
+
if p.n_columns > 1 or p.tables:
|
|
102
|
+
print(
|
|
103
|
+
f" صفحة {p.page_number}: {p.n_columns} عمود، "
|
|
104
|
+
f"{len(p.tables)} جدول",
|
|
105
|
+
file=sys.stderr,
|
|
106
|
+
)
|
|
107
|
+
if args.tables and doc.all_tables:
|
|
108
|
+
print("\n── جداول ──", file=sys.stderr)
|
|
109
|
+
for i, grid in enumerate(doc.all_tables):
|
|
110
|
+
print(f"جدول {i + 1}: {len(grid)}×{len(grid[0]) if grid else 0}", file=sys.stderr)
|
|
111
|
+
for row in grid:
|
|
112
|
+
print(" | " + " | ".join(row) + " |")
|
|
113
|
+
if doc.confidence < 0.5:
|
|
114
|
+
print("تحذير: ثقة منخفضة — راجع `arafix diagnose -v`", file=sys.stderr)
|
|
115
|
+
return 2
|
|
116
|
+
return 0
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _cmd_text(args: argparse.Namespace) -> int:
|
|
120
|
+
src = args.text if args.text else sys.stdin.read()
|
|
121
|
+
r = repair_text(src)
|
|
122
|
+
print(r.text)
|
|
123
|
+
if args.verbose:
|
|
124
|
+
print(f"\n─ العلل: {r.diagnosis.summary()}", file=sys.stderr)
|
|
125
|
+
print(f"─ المراحل: {[s.value for s in r.stages_applied]}", file=sys.stderr)
|
|
126
|
+
print(f"─ الثقة: {r.confidence}", file=sys.stderr)
|
|
127
|
+
for n in r.notes:
|
|
128
|
+
print(f" · {n}", file=sys.stderr)
|
|
129
|
+
return 0
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _cmd_blocks(args: argparse.Namespace) -> int:
|
|
133
|
+
"""أصلح سطوراً/خلايا من stdin (سطر = كتلة) — للجداول والأنابيب."""
|
|
134
|
+
from .pipeline import repair_blocks
|
|
135
|
+
from .types import TextBlock
|
|
136
|
+
|
|
137
|
+
lines = [ln.rstrip("\n\r") for ln in sys.stdin]
|
|
138
|
+
if args.skip_empty:
|
|
139
|
+
lines = [ln for ln in lines if ln.strip()]
|
|
140
|
+
blocks = [TextBlock(text=ln, id=f"L{i}", role="line") for i, ln in enumerate(lines)]
|
|
141
|
+
out = repair_blocks(blocks)
|
|
142
|
+
for b in out.blocks:
|
|
143
|
+
print(b.text)
|
|
144
|
+
if args.verbose:
|
|
145
|
+
print(
|
|
146
|
+
f"─ {len(out.blocks)} كتلة · ثقة دنيا {out.confidence}",
|
|
147
|
+
file=sys.stderr,
|
|
148
|
+
)
|
|
149
|
+
n_changed = sum(1 for b in out.blocks if b.repair.changed)
|
|
150
|
+
print(f"─ تغيّر منها: {n_changed}", file=sys.stderr)
|
|
151
|
+
return 0
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _cmd_eval(args: argparse.Namespace) -> int:
|
|
155
|
+
from .evaluate import EvalConfig, compare_extractors, evaluate_pdf
|
|
156
|
+
|
|
157
|
+
cfg = EvalConfig(
|
|
158
|
+
ignore_diacritics=args.ignore_diacritics,
|
|
159
|
+
ignore_punctuation=args.ignore_punctuation,
|
|
160
|
+
)
|
|
161
|
+
reports = (
|
|
162
|
+
compare_extractors(args.path, args.truth, cfg)
|
|
163
|
+
if args.compare
|
|
164
|
+
else [evaluate_pdf(args.path, args.truth, args.extractor, cfg)]
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
print("─" * 68)
|
|
168
|
+
for r in reports:
|
|
169
|
+
print(" ", r)
|
|
170
|
+
print("─" * 68)
|
|
171
|
+
|
|
172
|
+
best = reports[0]
|
|
173
|
+
if args.verbose and best.worst_lines:
|
|
174
|
+
print("\nأسوأ السطور في", best.label, ":")
|
|
175
|
+
for i, ref, hyp in best.worst_lines:
|
|
176
|
+
print(f" سطر {i}")
|
|
177
|
+
print(f" المرجع : {ref[:70]!r}")
|
|
178
|
+
print(f" الناتج : {hyp[:70]!r}")
|
|
179
|
+
|
|
180
|
+
if len(reports) > 1:
|
|
181
|
+
gap = reports[-1].cer.rate - best.cer.rate
|
|
182
|
+
print(f"\nأفضل مسار: {best.label} — يسبق أسوأهم بـ {gap:.2%} في CER")
|
|
183
|
+
return 0 if best.cer.rate < 0.05 else 3
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _cmd_fonts(args: argparse.Namespace) -> int:
|
|
187
|
+
from .cmap import build_glyph_map
|
|
188
|
+
from .extractors import get_extractor
|
|
189
|
+
|
|
190
|
+
ex = get_extractor(args.extractor)
|
|
191
|
+
fonts = ex.font_bytes(args.path)
|
|
192
|
+
if not fonts:
|
|
193
|
+
print("لا خطوط مضمَّنة — الدرجة ٣ غير ممكنة على هذا الملف.")
|
|
194
|
+
return 1
|
|
195
|
+
for name, data in fonts.items():
|
|
196
|
+
try:
|
|
197
|
+
gm = build_glyph_map(data, name)
|
|
198
|
+
print(f"{name:40s} تغطية {gm.coverage:.0%} ثقة {gm.confidence} ({gm.source})")
|
|
199
|
+
for note in gm.notes:
|
|
200
|
+
print(f" ! {note}")
|
|
201
|
+
except Exception as exc:
|
|
202
|
+
print(f"{name:40s} تعذّر التحليل: {exc}")
|
|
203
|
+
return 0
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
207
|
+
p = argparse.ArgumentParser(
|
|
208
|
+
prog="arafix", description="استرجاع النص العربي من ملفات PDF المعطوبة"
|
|
209
|
+
)
|
|
210
|
+
p.add_argument(
|
|
211
|
+
"--version",
|
|
212
|
+
action="version",
|
|
213
|
+
version=f"arafix {__version__} · unicode {unicode_version()}",
|
|
214
|
+
)
|
|
215
|
+
p.add_argument("-e", "--extractor", default="auto", help="محرّك القراءة (auto|pymupdf)")
|
|
216
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
217
|
+
|
|
218
|
+
d = sub.add_parser("diagnose", help="شخّص ولا تكتب شيئاً")
|
|
219
|
+
d.add_argument("path")
|
|
220
|
+
d.add_argument("-v", "--verbose", action="store_true", help="اعرض الشواهد")
|
|
221
|
+
d.add_argument("--json", action="store_true")
|
|
222
|
+
d.add_argument("-n", "--pages", type=int, default=0, help="حدّ الصفحات")
|
|
223
|
+
d.set_defaults(func=_cmd_diagnose)
|
|
224
|
+
|
|
225
|
+
x = sub.add_parser("extract", help="استخرج وأصلح")
|
|
226
|
+
x.add_argument("path")
|
|
227
|
+
x.add_argument("-o", "--output")
|
|
228
|
+
x.add_argument("--force-reorder", action="store_true", help="اعكس بلا شاهد")
|
|
229
|
+
x.add_argument(
|
|
230
|
+
"--layout",
|
|
231
|
+
choices=["auto", "linear", "columns", "full"],
|
|
232
|
+
default="auto",
|
|
233
|
+
help="تحليل البنية: auto|linear|columns|full",
|
|
234
|
+
)
|
|
235
|
+
x.add_argument(
|
|
236
|
+
"--reading-order",
|
|
237
|
+
choices=["rtl", "ltr"],
|
|
238
|
+
default="rtl",
|
|
239
|
+
help="ترتيب قراءة الأعمدة (افتراضي rtl)",
|
|
240
|
+
)
|
|
241
|
+
x.add_argument("-v", "--verbose", action="store_true", help="اعرض ملخص البنية")
|
|
242
|
+
x.add_argument("--tables", action="store_true", help="اطبع الجداول المستخرجة")
|
|
243
|
+
x.set_defaults(func=_cmd_extract)
|
|
244
|
+
|
|
245
|
+
t = sub.add_parser("text", help="أصلح نصاً مباشراً أو من stdin")
|
|
246
|
+
t.add_argument("text", nargs="?")
|
|
247
|
+
t.add_argument("-v", "--verbose", action="store_true")
|
|
248
|
+
t.set_defaults(func=_cmd_text)
|
|
249
|
+
|
|
250
|
+
b = sub.add_parser(
|
|
251
|
+
"blocks",
|
|
252
|
+
help="أصلح كتلًا مستقلة من stdin (سطر=كتلة) — جداول وأنابيب",
|
|
253
|
+
)
|
|
254
|
+
b.add_argument("-v", "--verbose", action="store_true")
|
|
255
|
+
b.add_argument(
|
|
256
|
+
"--skip-empty",
|
|
257
|
+
action="store_true",
|
|
258
|
+
help="تجاهل الأسطر الفارغة",
|
|
259
|
+
)
|
|
260
|
+
b.set_defaults(func=_cmd_blocks)
|
|
261
|
+
|
|
262
|
+
v = sub.add_parser("eval", help="قِس مقابل حقيقةٍ مرجعية (CER/WER)")
|
|
263
|
+
v.add_argument("path")
|
|
264
|
+
v.add_argument("--truth", required=True, help="ملفٌ نصّيّ فيه النصّ الصحيح")
|
|
265
|
+
v.add_argument("--compare", action="store_true", help="قِس كل المسارات ورتّبها")
|
|
266
|
+
v.add_argument("--ignore-diacritics", action="store_true")
|
|
267
|
+
v.add_argument("--ignore-punctuation", action="store_true")
|
|
268
|
+
v.add_argument("-v", "--verbose", action="store_true", help="اسرد أسوأ السطور")
|
|
269
|
+
v.set_defaults(func=_cmd_eval)
|
|
270
|
+
|
|
271
|
+
f = sub.add_parser("fonts", help="افحص الخطوط المضمَّنة (الدرجة ٣)")
|
|
272
|
+
f.add_argument("path")
|
|
273
|
+
f.set_defaults(func=_cmd_fonts)
|
|
274
|
+
|
|
275
|
+
return p
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def main(argv: list[str] | None = None) -> int:
|
|
279
|
+
args = build_parser().parse_args(argv)
|
|
280
|
+
try:
|
|
281
|
+
return args.func(args)
|
|
282
|
+
except (RuntimeError, KeyError, FileNotFoundError) as exc:
|
|
283
|
+
print(f"خطأ: {exc}", file=sys.stderr)
|
|
284
|
+
return 1
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
if __name__ == "__main__": # pragma: no cover
|
|
288
|
+
sys.exit(main())
|