arafix 0.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
arafix/__init__.py ADDED
@@ -0,0 +1,210 @@
1
+ """
2
+ arafix — recover broken Arabic text from PDFs.
3
+
4
+ Graded ladder (not one hammer)::
5
+
6
+ 0 diagnose() know before you fix
7
+ 1a fold_simple_forms() presentation forms; keep ligatures atomic
8
+ 2 fix_order() visual → logical, protect LTR runs
9
+ 1b expand_ligatures() ﻻ → لا after order is stable
10
+ 3 build_glyph_map() rebuild from embedded font
11
+ 4 OCR last resort (not shipped)
12
+
13
+ Quick start::
14
+
15
+ >>> from arafix import repair_text
16
+ >>> repair_text("\ufee3\ufeae\ufea3\ufe92\ufe8e").text
17
+ 'مرحبا'
18
+
19
+ >>> from arafix import repair_blocks
20
+ >>> repair_blocks(["\ufee3\ufeae\ufea3\ufe92\ufe8e"]).texts[0]
21
+ 'مرحبا'
22
+
23
+ >>> from arafix import extract_pdf # doctest: +SKIP
24
+ >>> doc = extract_pdf("thesis.pdf") # doctest: +SKIP
25
+
26
+ MIT license. Primary long-form docs are in Arabic (README).
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ __version__ = "0.8.0"
32
+ __license__ = "MIT"
33
+
34
+ from .adapters import as_blocks, fix_any, fix_markitdown, fix_table
35
+ from .cmap import GlyphMap, build_glyph_map, decode_glyph_name
36
+ from .diagnose import (
37
+ DEFAULT_THRESHOLDS,
38
+ detect_mojibake,
39
+ detect_presentation_forms,
40
+ detect_pua,
41
+ detect_visual_order,
42
+ diagnose,
43
+ )
44
+ from .evaluate import (
45
+ EvalConfig,
46
+ EvalReport,
47
+ cer,
48
+ compare_extractors,
49
+ evaluate_pdf,
50
+ evaluate_text,
51
+ levenshtein,
52
+ levenshtein_reference,
53
+ wer,
54
+ )
55
+ from .extractors import Extractor, RawPage, get_extractor, register
56
+ from .hygiene import (
57
+ count_artifacts,
58
+ fold_arabic_punct_confusables,
59
+ sanitize_extraction,
60
+ )
61
+ from .lamalef import (
62
+ LamAlefReport,
63
+ detect_lam_alef_transposition,
64
+ repair_lam_alef_transposition,
65
+ )
66
+ from .layout import (
67
+ Glyph,
68
+ LayoutColumn,
69
+ LayoutConfig,
70
+ LayoutLine,
71
+ LayoutTable,
72
+ PageLayout,
73
+ analyze_layout,
74
+ cluster_to_lines,
75
+ table_to_markdown,
76
+ )
77
+ from .normalize import (
78
+ NormalizeConfig,
79
+ expand_deferred_forms,
80
+ expand_ligatures,
81
+ fold_presentation_forms,
82
+ fold_simple_forms,
83
+ normalize_text,
84
+ )
85
+ from .order import (
86
+ MIRROR_PAIRS,
87
+ ReorderConfig,
88
+ fix_order,
89
+ grapheme_clusters,
90
+ reverse_visual_line,
91
+ )
92
+ from .pipeline import (
93
+ PipelineConfig,
94
+ extract_pdf,
95
+ harvest_document_lexicon,
96
+ repair_blocks,
97
+ repair_text,
98
+ )
99
+ from .types import (
100
+ BlockResult,
101
+ BlocksResult,
102
+ Defect,
103
+ Diagnosis,
104
+ DocumentResult,
105
+ Evidence,
106
+ PageResult,
107
+ RepairResult,
108
+ Stage,
109
+ TextBlock,
110
+ )
111
+ from .unicode_tables import (
112
+ DEFERRED_PF_TO_BASE,
113
+ LIGATURE_PF_TO_BASE,
114
+ PF_TO_BASE,
115
+ SIMPLE_PF_TO_BASE,
116
+ SPACING_MARK_PF_TO_BASE,
117
+ JoiningForm,
118
+ unicode_version,
119
+ )
120
+
121
+ __all__ = [
122
+ "__version__",
123
+ # الأنبوب
124
+ "repair_text",
125
+ "repair_blocks",
126
+ "extract_pdf",
127
+ "PipelineConfig",
128
+ "harvest_document_lexicon",
129
+ # مهايئات
130
+ "fix_any",
131
+ "fix_markitdown",
132
+ "fix_table",
133
+ "as_blocks",
134
+ # نظافة الاستخراج
135
+ "sanitize_extraction",
136
+ "count_artifacts",
137
+ "fold_arabic_punct_confusables",
138
+ # البنية
139
+ "Glyph",
140
+ "LayoutLine",
141
+ "LayoutColumn",
142
+ "LayoutTable",
143
+ "PageLayout",
144
+ "LayoutConfig",
145
+ "analyze_layout",
146
+ "cluster_to_lines",
147
+ "table_to_markdown",
148
+ # الدرجة ٠
149
+ "diagnose",
150
+ "detect_mojibake",
151
+ "detect_presentation_forms",
152
+ "detect_pua",
153
+ "detect_visual_order",
154
+ "DEFAULT_THRESHOLDS",
155
+ # الدرجة ١ (تمريرتان: مفردات ← اتجاه ← رباطات)
156
+ "normalize_text",
157
+ "fold_presentation_forms",
158
+ "fold_simple_forms",
159
+ "expand_deferred_forms",
160
+ "expand_ligatures",
161
+ "NormalizeConfig",
162
+ # لام-ألف
163
+ "detect_lam_alef_transposition",
164
+ "repair_lam_alef_transposition",
165
+ "LamAlefReport",
166
+ # الدرجة ٢
167
+ "fix_order",
168
+ "reverse_visual_line",
169
+ "grapheme_clusters",
170
+ "MIRROR_PAIRS",
171
+ "ReorderConfig",
172
+ # الدرجة ٣
173
+ "build_glyph_map",
174
+ "decode_glyph_name",
175
+ "GlyphMap",
176
+ # النماذج
177
+ "Defect",
178
+ "Stage",
179
+ "Evidence",
180
+ "Diagnosis",
181
+ "RepairResult",
182
+ "PageResult",
183
+ "DocumentResult",
184
+ "TextBlock",
185
+ "BlockResult",
186
+ "BlocksResult",
187
+ # القياس
188
+ "evaluate_text",
189
+ "evaluate_pdf",
190
+ "compare_extractors",
191
+ "cer",
192
+ "wer",
193
+ "levenshtein",
194
+ "levenshtein_reference",
195
+ "EvalConfig",
196
+ "EvalReport",
197
+ # المحرّكات
198
+ "Extractor",
199
+ "RawPage",
200
+ "get_extractor",
201
+ "register",
202
+ # الجداول
203
+ "PF_TO_BASE",
204
+ "SIMPLE_PF_TO_BASE",
205
+ "DEFERRED_PF_TO_BASE",
206
+ "LIGATURE_PF_TO_BASE",
207
+ "SPACING_MARK_PF_TO_BASE",
208
+ "JoiningForm",
209
+ "unicode_version",
210
+ ]
arafix/adapters.py ADDED
@@ -0,0 +1,83 @@
1
+ """
2
+ مهايئات خفيفة — مسارٌ واحد من «نصٍّ مستخرج بأيّ أداة» إلى عربيّ سليم.
3
+
4
+ from arafix.adapters import fix_markitdown, fix_any
5
+
6
+ md = MarkItDown().convert("f.pdf")
7
+ print(fix_markitdown(md).text)
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from typing import Any
13
+
14
+ from .pipeline import PipelineConfig, repair_blocks, repair_text
15
+ from .types import BlocksResult, RepairResult, TextBlock
16
+
17
+ __all__ = ["fix_any", "fix_markitdown", "fix_table", "as_blocks"]
18
+
19
+
20
+ def fix_any(text: str, config: PipelineConfig | None = None) -> RepairResult:
21
+ """أيّ نصٍّ — من pdfminer أو المتصفح أو الحافظة."""
22
+ return repair_text(text, config)
23
+
24
+
25
+ def fix_markitdown(
26
+ result: Any,
27
+ config: PipelineConfig | None = None,
28
+ ) -> RepairResult:
29
+ """
30
+ يقبل ``DocumentConverterResult`` من MarkItDown (أو أيّ كائن فيه
31
+ ``text_content`` / ``markdown``).
32
+ """
33
+ text = (
34
+ getattr(result, "text_content", None)
35
+ or getattr(result, "markdown", None)
36
+ or str(result)
37
+ )
38
+ return repair_text(text, config)
39
+
40
+
41
+ def as_blocks(
42
+ rows: list[list[str]],
43
+ *,
44
+ id_prefix: str = "r",
45
+ ) -> list[TextBlock]:
46
+ """ يحوّل جدولاً (قائمة صفوف) إلى كتل ``TextBlock`` بخلايا مستقلة."""
47
+ blocks: list[TextBlock] = []
48
+ for i, row in enumerate(rows):
49
+ for j, cell in enumerate(row):
50
+ blocks.append(
51
+ TextBlock(
52
+ text=cell or "",
53
+ id=f"{id_prefix}{i}c{j}",
54
+ role="cell",
55
+ meta={"row": i, "col": j},
56
+ )
57
+ )
58
+ return blocks
59
+
60
+
61
+ def fix_table(
62
+ rows: list[list[str]],
63
+ config: PipelineConfig | None = None,
64
+ ) -> list[list[str]]:
65
+ """
66
+ يصلح خلايا جدولٍ كلٌّ على حدة ويُرجع نفس الشكل.
67
+
68
+ >>> fix_table([["\ufee3\ufeae\ufea3\ufe92\ufe8e", "OK"]])
69
+ [['مرحبا', 'OK']]
70
+ """
71
+ if not rows:
72
+ return []
73
+ blocks = as_blocks(rows)
74
+ repaired: BlocksResult = repair_blocks(blocks, config)
75
+ by_id = repaired.by_id()
76
+ out: list[list[str]] = []
77
+ for i, row in enumerate(rows):
78
+ new_row = []
79
+ for j, _ in enumerate(row):
80
+ key = f"r{i}c{j}"
81
+ new_row.append(by_id[key].text if key in by_id else "")
82
+ out.append(new_row)
83
+ return out
arafix/cli.py ADDED
@@ -0,0 +1,288 @@
1
+ """
2
+ واجهة سطر الأوامر.
3
+
4
+ arafix diagnose thesis.pdf
5
+ arafix extract thesis.pdf -o out.txt
6
+ arafix eval thesis.pdf --truth thesis.txt --compare
7
+ arafix text "ﺎﺒﺣﺮﻣ"
8
+ arafix fonts thesis.pdf
9
+
10
+ فلسفة الأمر `diagnose` أنه **لا يكتب شيئاً**. اقرأ تقريره أولاً، ثم
11
+ قرّر. الأداة التي تعالج قبل أن تُريك ما وجدت أداةٌ لا تُؤتمن.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import argparse
17
+ import json
18
+ import sys
19
+
20
+ from . import __version__
21
+ from .diagnose import diagnose
22
+ from .pipeline import PipelineConfig, extract_pdf, repair_text
23
+ from .unicode_tables import unicode_version
24
+
25
+
26
+ def _cmd_diagnose(args: argparse.Namespace) -> int:
27
+ from .extractors import get_extractor
28
+
29
+ ex = get_extractor(args.extractor)
30
+ report = []
31
+ for raw in ex.pages(args.path):
32
+ dg = diagnose(raw.text)
33
+ report.append(
34
+ {
35
+ "page": raw.number,
36
+ "chars": dg.char_count,
37
+ "arabic_ratio": round(dg.arabic_ratio, 3),
38
+ "defects": [d.value for d in dg.defects],
39
+ "confidence": dg.confidence,
40
+ "defect_confidence": {
41
+ k.value: v for k, v in dg.defect_confidence.items()
42
+ },
43
+ "fonts": raw.fonts,
44
+ "evidence": [
45
+ {"name": e.name, "value": round(e.value, 3), "detail": e.detail}
46
+ for e in dg.evidence
47
+ ],
48
+ }
49
+ )
50
+ if args.pages and raw.number >= args.pages:
51
+ break
52
+
53
+ if args.json:
54
+ print(json.dumps(report, ensure_ascii=False, indent=2))
55
+ return 0
56
+
57
+ for r in report:
58
+ print(f"── صفحة {r['page']} " + "─" * 40)
59
+ conf = r["defect_confidence"]
60
+ detail = "، ".join(f"{d} ({conf.get(d, 0):.2f})" for d in r["defects"])
61
+ print(f" العلل : {detail}")
62
+ print(f" الثقة : {r['confidence']} (أضعف حلقة)")
63
+ print(f" الحروف : {r['chars']} (عربية {r['arabic_ratio']:.0%})")
64
+ if r["fonts"]:
65
+ print(f" الخطوط : {', '.join(r['fonts'][:4])}")
66
+ if args.verbose:
67
+ for e in r["evidence"]:
68
+ print(f" · {e['name']:22s} {e['value']:+.3f} {e['detail']}")
69
+ return 0
70
+
71
+
72
+ def _cmd_extract(args: argparse.Namespace) -> int:
73
+ from .layout import LayoutConfig
74
+
75
+ lay_cfg = LayoutConfig(reading_order=args.reading_order)
76
+ cfg = PipelineConfig(
77
+ extractor=args.extractor,
78
+ force_reorder=args.force_reorder,
79
+ layout=args.layout,
80
+ layout_config=lay_cfg,
81
+ )
82
+ doc = extract_pdf(args.path, cfg)
83
+
84
+ out = doc.text
85
+ if args.output:
86
+ with open(args.output, "w", encoding="utf-8") as fh:
87
+ fh.write(out)
88
+ print(f"كُتب {len(out)} حرفاً في {args.output}", file=sys.stderr)
89
+ else:
90
+ print(out)
91
+
92
+ print(f"الثقة الدنيا عبر {len(doc.pages)} صفحة: {doc.confidence}", file=sys.stderr)
93
+ if args.verbose:
94
+ print(
95
+ f"بنية: أعمدة≤{doc.metadata.get('max_columns', 1)} "
96
+ f"جداول={doc.metadata.get('table_count', 0)} "
97
+ f"layout={doc.metadata.get('layout')}",
98
+ file=sys.stderr,
99
+ )
100
+ for p in doc.pages:
101
+ if p.n_columns > 1 or p.tables:
102
+ print(
103
+ f" صفحة {p.page_number}: {p.n_columns} عمود، "
104
+ f"{len(p.tables)} جدول",
105
+ file=sys.stderr,
106
+ )
107
+ if args.tables and doc.all_tables:
108
+ print("\n── جداول ──", file=sys.stderr)
109
+ for i, grid in enumerate(doc.all_tables):
110
+ print(f"جدول {i + 1}: {len(grid)}×{len(grid[0]) if grid else 0}", file=sys.stderr)
111
+ for row in grid:
112
+ print(" | " + " | ".join(row) + " |")
113
+ if doc.confidence < 0.5:
114
+ print("تحذير: ثقة منخفضة — راجع `arafix diagnose -v`", file=sys.stderr)
115
+ return 2
116
+ return 0
117
+
118
+
119
+ def _cmd_text(args: argparse.Namespace) -> int:
120
+ src = args.text if args.text else sys.stdin.read()
121
+ r = repair_text(src)
122
+ print(r.text)
123
+ if args.verbose:
124
+ print(f"\n─ العلل: {r.diagnosis.summary()}", file=sys.stderr)
125
+ print(f"─ المراحل: {[s.value for s in r.stages_applied]}", file=sys.stderr)
126
+ print(f"─ الثقة: {r.confidence}", file=sys.stderr)
127
+ for n in r.notes:
128
+ print(f" · {n}", file=sys.stderr)
129
+ return 0
130
+
131
+
132
+ def _cmd_blocks(args: argparse.Namespace) -> int:
133
+ """أصلح سطوراً/خلايا من stdin (سطر = كتلة) — للجداول والأنابيب."""
134
+ from .pipeline import repair_blocks
135
+ from .types import TextBlock
136
+
137
+ lines = [ln.rstrip("\n\r") for ln in sys.stdin]
138
+ if args.skip_empty:
139
+ lines = [ln for ln in lines if ln.strip()]
140
+ blocks = [TextBlock(text=ln, id=f"L{i}", role="line") for i, ln in enumerate(lines)]
141
+ out = repair_blocks(blocks)
142
+ for b in out.blocks:
143
+ print(b.text)
144
+ if args.verbose:
145
+ print(
146
+ f"─ {len(out.blocks)} كتلة · ثقة دنيا {out.confidence}",
147
+ file=sys.stderr,
148
+ )
149
+ n_changed = sum(1 for b in out.blocks if b.repair.changed)
150
+ print(f"─ تغيّر منها: {n_changed}", file=sys.stderr)
151
+ return 0
152
+
153
+
154
+ def _cmd_eval(args: argparse.Namespace) -> int:
155
+ from .evaluate import EvalConfig, compare_extractors, evaluate_pdf
156
+
157
+ cfg = EvalConfig(
158
+ ignore_diacritics=args.ignore_diacritics,
159
+ ignore_punctuation=args.ignore_punctuation,
160
+ )
161
+ reports = (
162
+ compare_extractors(args.path, args.truth, cfg)
163
+ if args.compare
164
+ else [evaluate_pdf(args.path, args.truth, args.extractor, cfg)]
165
+ )
166
+
167
+ print("─" * 68)
168
+ for r in reports:
169
+ print(" ", r)
170
+ print("─" * 68)
171
+
172
+ best = reports[0]
173
+ if args.verbose and best.worst_lines:
174
+ print("\nأسوأ السطور في", best.label, ":")
175
+ for i, ref, hyp in best.worst_lines:
176
+ print(f" سطر {i}")
177
+ print(f" المرجع : {ref[:70]!r}")
178
+ print(f" الناتج : {hyp[:70]!r}")
179
+
180
+ if len(reports) > 1:
181
+ gap = reports[-1].cer.rate - best.cer.rate
182
+ print(f"\nأفضل مسار: {best.label} — يسبق أسوأهم بـ {gap:.2%} في CER")
183
+ return 0 if best.cer.rate < 0.05 else 3
184
+
185
+
186
+ def _cmd_fonts(args: argparse.Namespace) -> int:
187
+ from .cmap import build_glyph_map
188
+ from .extractors import get_extractor
189
+
190
+ ex = get_extractor(args.extractor)
191
+ fonts = ex.font_bytes(args.path)
192
+ if not fonts:
193
+ print("لا خطوط مضمَّنة — الدرجة ٣ غير ممكنة على هذا الملف.")
194
+ return 1
195
+ for name, data in fonts.items():
196
+ try:
197
+ gm = build_glyph_map(data, name)
198
+ print(f"{name:40s} تغطية {gm.coverage:.0%} ثقة {gm.confidence} ({gm.source})")
199
+ for note in gm.notes:
200
+ print(f" ! {note}")
201
+ except Exception as exc:
202
+ print(f"{name:40s} تعذّر التحليل: {exc}")
203
+ return 0
204
+
205
+
206
+ def build_parser() -> argparse.ArgumentParser:
207
+ p = argparse.ArgumentParser(
208
+ prog="arafix", description="استرجاع النص العربي من ملفات PDF المعطوبة"
209
+ )
210
+ p.add_argument(
211
+ "--version",
212
+ action="version",
213
+ version=f"arafix {__version__} · unicode {unicode_version()}",
214
+ )
215
+ p.add_argument("-e", "--extractor", default="auto", help="محرّك القراءة (auto|pymupdf)")
216
+ sub = p.add_subparsers(dest="cmd", required=True)
217
+
218
+ d = sub.add_parser("diagnose", help="شخّص ولا تكتب شيئاً")
219
+ d.add_argument("path")
220
+ d.add_argument("-v", "--verbose", action="store_true", help="اعرض الشواهد")
221
+ d.add_argument("--json", action="store_true")
222
+ d.add_argument("-n", "--pages", type=int, default=0, help="حدّ الصفحات")
223
+ d.set_defaults(func=_cmd_diagnose)
224
+
225
+ x = sub.add_parser("extract", help="استخرج وأصلح")
226
+ x.add_argument("path")
227
+ x.add_argument("-o", "--output")
228
+ x.add_argument("--force-reorder", action="store_true", help="اعكس بلا شاهد")
229
+ x.add_argument(
230
+ "--layout",
231
+ choices=["auto", "linear", "columns", "full"],
232
+ default="auto",
233
+ help="تحليل البنية: auto|linear|columns|full",
234
+ )
235
+ x.add_argument(
236
+ "--reading-order",
237
+ choices=["rtl", "ltr"],
238
+ default="rtl",
239
+ help="ترتيب قراءة الأعمدة (افتراضي rtl)",
240
+ )
241
+ x.add_argument("-v", "--verbose", action="store_true", help="اعرض ملخص البنية")
242
+ x.add_argument("--tables", action="store_true", help="اطبع الجداول المستخرجة")
243
+ x.set_defaults(func=_cmd_extract)
244
+
245
+ t = sub.add_parser("text", help="أصلح نصاً مباشراً أو من stdin")
246
+ t.add_argument("text", nargs="?")
247
+ t.add_argument("-v", "--verbose", action="store_true")
248
+ t.set_defaults(func=_cmd_text)
249
+
250
+ b = sub.add_parser(
251
+ "blocks",
252
+ help="أصلح كتلًا مستقلة من stdin (سطر=كتلة) — جداول وأنابيب",
253
+ )
254
+ b.add_argument("-v", "--verbose", action="store_true")
255
+ b.add_argument(
256
+ "--skip-empty",
257
+ action="store_true",
258
+ help="تجاهل الأسطر الفارغة",
259
+ )
260
+ b.set_defaults(func=_cmd_blocks)
261
+
262
+ v = sub.add_parser("eval", help="قِس مقابل حقيقةٍ مرجعية (CER/WER)")
263
+ v.add_argument("path")
264
+ v.add_argument("--truth", required=True, help="ملفٌ نصّيّ فيه النصّ الصحيح")
265
+ v.add_argument("--compare", action="store_true", help="قِس كل المسارات ورتّبها")
266
+ v.add_argument("--ignore-diacritics", action="store_true")
267
+ v.add_argument("--ignore-punctuation", action="store_true")
268
+ v.add_argument("-v", "--verbose", action="store_true", help="اسرد أسوأ السطور")
269
+ v.set_defaults(func=_cmd_eval)
270
+
271
+ f = sub.add_parser("fonts", help="افحص الخطوط المضمَّنة (الدرجة ٣)")
272
+ f.add_argument("path")
273
+ f.set_defaults(func=_cmd_fonts)
274
+
275
+ return p
276
+
277
+
278
+ def main(argv: list[str] | None = None) -> int:
279
+ args = build_parser().parse_args(argv)
280
+ try:
281
+ return args.func(args)
282
+ except (RuntimeError, KeyError, FileNotFoundError) as exc:
283
+ print(f"خطأ: {exc}", file=sys.stderr)
284
+ return 1
285
+
286
+
287
+ if __name__ == "__main__": # pragma: no cover
288
+ sys.exit(main())