langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1894 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+
17
+ import asyncio
18
+ import logging
19
+ import math
20
+ import os
21
+ import random
22
+ import re
23
+ import sys
24
+ import threading
25
+ import unicodedata
26
+ from collections import Counter, defaultdict
27
+ from copy import deepcopy
28
+ from io import BytesIO
29
+ from timeit import default_timer as timer
30
+
31
+ import numpy as np
32
+ import pdfplumber
33
+ from PIL import Image
34
+ from sklearn.cluster import KMeans
35
+ from sklearn.metrics import silhouette_score
36
+
37
+ from .model_loader import default_model_dir
38
+ from .ocr import OCR
39
+ from .layout_recognizer import LayoutRecognizer4YOLOv10 as LayoutRecognizer
40
+ from .recognizer import Recognizer
41
+ from .table_structure_recognizer import TableStructureRecognizer
42
+ from .tokenizer import is_chinese
43
+ from .utils import extract_pdf_outlines
44
+
45
+ MAXIMUM_PAGE_NUMBER = 100000
46
+ #: Was torch.cuda.device_count() upstream (multi-GPU OCR); this port is
47
+ #: CPU-only, single-device, so parallel_limiter (see __init__) is always None
48
+ #: and this constant only matters to keep the still-present but now-dead
49
+ #: settings.PARALLEL_DEVICES reference in __images__ syntactically valid.
50
+ PARALLEL_DEVICES = 0
51
+
52
+
53
+ #: NOTE: this is a simplified reimplementation, not a verbatim port, of
54
+ #: common/misc_utils.py's thread_pool_exec. Upstream uses a per-call
55
+ #: ThreadPoolExecutor(max_workers=1) instead of loop.run_in_executor(None,
56
+ #: call) specifically to avoid a documented Python 3.13 deadlock on repeated
57
+ #: awaits within one event loop. That distinction has no present effect here
58
+ #: -- this function's only call site is inside a branch that's permanently
59
+ #: unreachable, since self.parallel_limiter is always None after this port's
60
+ #: __init__ (see below) -- but reviving multi-device parallelism in the
61
+ #: future would need to restore the per-call executor to avoid that deadlock.
62
+ async def thread_pool_exec(func, *args, **kwargs):
63
+ import asyncio
64
+ import contextvars
65
+ import functools
66
+
67
+ loop = asyncio.get_running_loop()
68
+ ctx = contextvars.copy_context()
69
+ call = functools.partial(ctx.run, func, *args, **kwargs)
70
+ return await loop.run_in_executor(None, call)
71
+
72
+ LOCK_KEY_pdfplumber = "global_shared_lock_pdfplumber"
73
+ if LOCK_KEY_pdfplumber not in sys.modules:
74
+ sys.modules[LOCK_KEY_pdfplumber] = threading.Lock()
75
+
76
+
77
+ class RAGFlowPdfParser:
78
+ def __init__(self, model_dir=None, **kwargs):
79
+ # Resolved once so every model-file consumer -- OCR, LayoutRecognizer,
80
+ # TableStructureRecognizer, and _ocr_can_represent's ocr.res lookup --
81
+ # agrees on the same directory. OCR/Recognizer already fall back to
82
+ # default_model_dir() internally when given None, so passing the
83
+ # already-resolved value here changes nothing for them.
84
+ self.model_dir = model_dir or str(default_model_dir())
85
+ self.ocr = OCR(model_dir=self.model_dir)
86
+ self.parallel_limiter = None
87
+
88
+ self.layouter = LayoutRecognizer("layout", model_dir=self.model_dir)
89
+ self.tbl_det = TableStructureRecognizer(model_dir=self.model_dir)
90
+
91
+ self.page_from = 0
92
+ self.column_num = 1
93
+
94
+ def __char_width(self, c):
95
+ return (c["x1"] - c["x0"]) // max(len(c["text"]), 1)
96
+
97
+ def __height(self, c):
98
+ return c["bottom"] - c["top"]
99
+
100
+ def _x_dis(self, a, b):
101
+ return min(abs(a["x1"] - b["x0"]), abs(a["x0"] - b["x1"]), abs(a["x0"] + a["x1"] - b["x0"] - b["x1"]) / 2)
102
+
103
+ def _y_dis(self, a, b):
104
+ return (b["top"] + b["bottom"] - a["top"] - a["bottom"]) / 2
105
+
106
+ def _match_proj(self, b):
107
+ proj_patt = [
108
+ r"第[零一二三四五六七八九十百]+章",
109
+ r"第[零一二三四五六七八九十百]+[条节]",
110
+ r"[零一二三四五六七八九十百]+[、是  ]",
111
+ r"[\((][零一二三四五六七八九十百]+[)\)]",
112
+ r"[\((][0-9]+[)\)]",
113
+ r"[0-9]+(、|\.[  ]|)|\.[^0-9./a-zA-Z_%><-]{4,})",
114
+ r"[0-9]+\.[0-9.]+(、|\.[  ])",
115
+ r"[⚫•➢①② ]",
116
+ ]
117
+ return any([re.match(p, b["text"]) for p in proj_patt])
118
+
119
+ @staticmethod
120
+ def sort_X_by_page(arr, threshold):
121
+ # sort using y1 first and then x1
122
+ arr = sorted(arr, key=lambda r: (r["page_number"], r["x0"], r["top"]))
123
+ for i in range(len(arr) - 1):
124
+ for j in range(i, -1, -1):
125
+ # restore the order using th
126
+ if abs(arr[j + 1]["x0"] - arr[j]["x0"]) < threshold and arr[j + 1]["top"] < arr[j]["top"] and arr[j + 1]["page_number"] == arr[j]["page_number"]:
127
+ tmp = arr[j]
128
+ arr[j] = arr[j + 1]
129
+ arr[j + 1] = tmp
130
+ return arr
131
+
132
+ def _has_color(self, o):
133
+ if o.get("ncs", "") == "DeviceGray":
134
+ if o["stroking_color"] and o["stroking_color"][0] == 1 and o["non_stroking_color"] and o["non_stroking_color"][0] == 1:
135
+ if re.match(r"[a-zT_\[\]\(\)-]+", o.get("text", "")):
136
+ return False
137
+ return True
138
+
139
+ # CID pattern regex for unmapped font characters from pdfminer
140
+ _CID_PATTERN = re.compile(r"\(cid\s*:\s*\d+\s*\)")
141
+
142
+ # Class-level default; a matching instance attribute (see __init__'s
143
+ # self.model_dir) shadows this once _ocr_can_represent below populates
144
+ # it, scoping the cache to the model_dir this instance was built with.
145
+ _OCR_ALPHABET = None
146
+
147
+ def _ocr_can_represent(self, text, min_coverage=0.8):
148
+ """True if the OCR recogniser's alphabet covers this text well enough to be worth OCRing."""
149
+ if not text:
150
+ return True
151
+ if self._OCR_ALPHABET is None:
152
+ res = os.path.join(self.model_dir, "ocr.res")
153
+ try:
154
+ with open(res, encoding="utf-8") as f:
155
+ self._OCR_ALPHABET = set(f.read())
156
+ except (OSError, UnicodeDecodeError) as e:
157
+ logging.warning("Could not load OCR alphabet from %s: %s; treating all text as representable.", res, e)
158
+ self._OCR_ALPHABET = set()
159
+ if not self._OCR_ALPHABET:
160
+ return True # unknown alphabet: preserve existing behaviour
161
+ letters = [c for c in text if c.strip()]
162
+ if not letters:
163
+ return True
164
+ covered = sum(1 for c in letters if c in self._OCR_ALPHABET)
165
+ return covered / len(letters) >= min_coverage
166
+
167
+ # CJK scripts (Han, Hiragana, Katakana, Hangul) do not separate words with
168
+ # spaces, so a geometric gap between their glyphs must not become one.
169
+ _CJK_PATTERN = re.compile(r"[ᄀ-ᇿ぀-ヿ㄰-㆏㐀-䶿一-鿿가-힯豈-﫿]|[\U00020000-\U0002fa1f]")
170
+
171
+ @classmethod
172
+ def _insert_word_spaces(cls, chars, gap_ratio=0.25):
173
+ """Recover missing spaces from character geometry.
174
+
175
+ Many PDFs encode no space glyphs and separate words by positioning alone.
176
+ Append a space to a char when the gap to the next exceeds ``gap_ratio`` of
177
+ the mean char width; intra-word kerns fall well below that. CJK is skipped:
178
+ it does not write inter-word spaces, so a gap between CJK glyphs is ordinary
179
+ tracking, not a boundary. ``chars`` is a list of pdfplumber-style dicts and
180
+ is mutated in place.
181
+ """
182
+ widths = [c["width"] for c in chars if c["text"] and c["text"].strip()]
183
+ mean_w = sum(widths) / len(widths) if widths else 0
184
+ if mean_w <= 0:
185
+ return
186
+ for cur, nxt in zip(chars, chars[1:]):
187
+ if (
188
+ cur["text"]
189
+ and nxt["text"]
190
+ and cur["text"].strip()
191
+ and nxt["text"].strip()
192
+ and not cls._CJK_PATTERN.search(cur["text"])
193
+ and not cls._CJK_PATTERN.search(nxt["text"])
194
+ and nxt["x0"] - cur["x1"] > mean_w * gap_ratio
195
+ ):
196
+ cur["text"] += " "
197
+
198
+ @staticmethod
199
+ def _is_garbled_char(ch):
200
+ """Check if a single character is garbled (unmappable from PDF font encoding).
201
+
202
+ A character is considered garbled if it falls into Unicode Private Use Areas
203
+ or certain replacement/control character ranges that typically indicate
204
+ pdfminer failed to map a CID to a valid Unicode codepoint.
205
+ """
206
+ if not ch:
207
+ return False
208
+ cp = ord(ch)
209
+ if 0xE000 <= cp <= 0xF8FF:
210
+ return True
211
+ if 0xF0000 <= cp <= 0xFFFFF:
212
+ return True
213
+ if 0x100000 <= cp <= 0x10FFFF:
214
+ return True
215
+ if cp == 0xFFFD:
216
+ return True
217
+ if cp < 0x20 and ch not in ("\t", "\n", "\r"):
218
+ return True
219
+ if 0x80 <= cp <= 0x9F:
220
+ return True
221
+ cat = unicodedata.category(ch)
222
+ if cat in ("Cn", "Cs"):
223
+ return True
224
+ return False
225
+
226
+ @staticmethod
227
+ def _is_garbled_text(text, threshold=0.5):
228
+ """Check if a text string contains too many garbled characters.
229
+
230
+ Examines each character and determines if the overall proportion
231
+ of garbled characters exceeds the given threshold. Also detects
232
+ pdfminer's CID placeholder patterns like '(cid:123)'.
233
+ """
234
+ if not text or not text.strip():
235
+ return False
236
+ if RAGFlowPdfParser._CID_PATTERN.search(text):
237
+ return True
238
+ garbled_count = 0
239
+ total = 0
240
+ for ch in text:
241
+ if ch.isspace():
242
+ continue
243
+ total += 1
244
+ if RAGFlowPdfParser._is_garbled_char(ch):
245
+ garbled_count += 1
246
+ if total == 0:
247
+ return False
248
+ return garbled_count / total >= threshold
249
+
250
+ @staticmethod
251
+ def _has_subset_font_prefix(fontname):
252
+ """Check if a font name has a subset prefix (e.g. 'DY1+ZLQDm1-1').
253
+
254
+ PDF subset fonts use a 6-letter uppercase tag followed by '+' before
255
+ the actual font name. Some tools use shorter tags (e.g. 'DY1+').
256
+ """
257
+ if not fontname:
258
+ return False
259
+ return bool(re.match(r"^[A-Z0-9]{2,6}\+", fontname))
260
+
261
+ @staticmethod
262
+ def _is_garbled_by_font_encoding(page_chars, min_chars=20):
263
+ """Detect garbled text caused by broken font encoding mappings.
264
+
265
+ Some PDFs (especially older Chinese standards) embed custom fonts that
266
+ map CJK glyphs to ASCII codepoints. The extracted text appears as
267
+ random ASCII punctuation/symbols instead of actual CJK characters.
268
+
269
+ Detection strategy: if a significant proportion of characters come from
270
+ subset-embedded fonts and the page produces overwhelmingly ASCII
271
+ (punctuation, digits, symbols) with virtually no CJK/Hangul/Kana
272
+ characters, the page is likely garbled due to broken font encoding.
273
+ """
274
+ if not page_chars or len(page_chars) < min_chars:
275
+ return False
276
+
277
+ subset_font_count = 0
278
+ total_non_space = 0
279
+ ascii_punct_sym = 0
280
+ cjk_like = 0
281
+
282
+ for c in page_chars:
283
+ text = c.get("text", "")
284
+ fontname = c.get("fontname", "")
285
+ if not text or text.isspace():
286
+ continue
287
+ total_non_space += 1
288
+
289
+ if RAGFlowPdfParser._has_subset_font_prefix(fontname):
290
+ subset_font_count += 1
291
+
292
+ cp = ord(text[0])
293
+ if 0x2E80 <= cp <= 0x9FFF or 0xF900 <= cp <= 0xFAFF or 0x20000 <= cp <= 0x2FA1F or 0xAC00 <= cp <= 0xD7AF or 0x3040 <= cp <= 0x30FF:
294
+ cjk_like += 1
295
+ elif 0x21 <= cp <= 0x2F or 0x3A <= cp <= 0x40 or 0x5B <= cp <= 0x60 or 0x7B <= cp <= 0x7E:
296
+ ascii_punct_sym += 1
297
+
298
+ if total_non_space < min_chars:
299
+ return False
300
+
301
+ subset_ratio = subset_font_count / total_non_space
302
+ if subset_ratio < 0.3:
303
+ return False
304
+
305
+ cjk_ratio = cjk_like / total_non_space
306
+ punct_ratio = ascii_punct_sym / total_non_space
307
+ if cjk_ratio < 0.05 and punct_ratio > 0.4:
308
+ return True
309
+
310
+ return False
311
+
312
+ def _evaluate_table_orientation(self, table_img, sample_ratio=0.3):
313
+ """
314
+ Evaluate the best rotation orientation for a table image.
315
+
316
+ Tests 4 rotation angles (0°, 90°, 180°, 270°) and uses OCR
317
+ confidence scores to determine the best orientation.
318
+
319
+ Args:
320
+ table_img: PIL Image object of the table region
321
+ sample_ratio: Sampling ratio for quick evaluation
322
+
323
+ Returns:
324
+ tuple: (best_angle, best_img, confidence_scores)
325
+ - best_angle: Best rotation angle (0, 90, 180, 270)
326
+ - best_img: Image rotated to best orientation
327
+ - confidence_scores: Dict of scores for each angle
328
+ """
329
+
330
+ rotations = [
331
+ (0, "original"),
332
+ (90, "rotate_90"), # clockwise 90°
333
+ (180, "rotate_180"), # 180°
334
+ (270, "rotate_270"), # clockwise 270° (counter-clockwise 90°)
335
+ ]
336
+
337
+ results = {}
338
+ best_score = -1
339
+ best_angle = 0
340
+ best_img = table_img
341
+ score_0 = None
342
+
343
+ for angle, name in rotations:
344
+ # Rotate image
345
+ if angle == 0:
346
+ rotated_img = table_img
347
+ else:
348
+ # PIL's rotate is counter-clockwise, use negative angle for clockwise
349
+ rotated_img = table_img.rotate(-angle, expand=True)
350
+
351
+ # Convert to numpy array for OCR
352
+ img_array = np.array(rotated_img)
353
+
354
+ # Perform OCR detection and recognition
355
+ try:
356
+ ocr_results = self.ocr(img_array)
357
+
358
+ if ocr_results:
359
+ # Calculate average confidence
360
+ scores = [conf for _, (_, conf) in ocr_results]
361
+ avg_score = sum(scores) / len(scores) if scores else 0
362
+ total_regions = len(scores)
363
+
364
+ # Combined score: considers both average confidence and number of regions
365
+ # More regions + higher confidence = better orientation
366
+ combined_score = avg_score * (1 + 0.1 * min(total_regions, 50) / 50)
367
+ else:
368
+ avg_score = 0
369
+ total_regions = 0
370
+ combined_score = 0
371
+
372
+ except Exception as e:
373
+ logging.warning(f"OCR failed for angle {angle}: {e}")
374
+ avg_score = 0
375
+ total_regions = 0
376
+ combined_score = 0
377
+
378
+ results[angle] = {"avg_confidence": avg_score, "total_regions": total_regions, "combined_score": combined_score}
379
+ if angle == 0:
380
+ score_0 = combined_score
381
+
382
+ logging.debug(f"Table orientation {angle}°: avg_conf={avg_score:.4f}, regions={total_regions}, combined={combined_score:.4f}")
383
+
384
+ if combined_score > best_score:
385
+ best_score = combined_score
386
+ best_angle = angle
387
+ best_img = rotated_img
388
+
389
+ # Absolute threshold rule:
390
+ # Only choose non-0° if it exceeds 0° by more than 0.2 and 0° score is below 0.8.
391
+ if best_angle != 0 and score_0 is not None:
392
+ if not (best_score - score_0 > 0.2 and score_0 < 0.8):
393
+ best_angle = 0
394
+ best_img = table_img
395
+ best_score = score_0
396
+
397
+ results[best_angle] = results.get(best_angle, {"avg_confidence": 0, "total_regions": 0, "combined_score": 0})
398
+
399
+ logging.info(f"Best table orientation: {best_angle}° (score={best_score:.4f})")
400
+
401
+ return best_angle, best_img, results
402
+
403
+ @staticmethod
404
+ def _map_clockwise_rotated_point_to_original(x, y, angle, width, height):
405
+ if angle == 0:
406
+ return x, y
407
+ if angle == 90:
408
+ return y, height - x
409
+ if angle == 180:
410
+ return width - x, height - y
411
+ if angle == 270:
412
+ return width - y, x
413
+ return x, y
414
+
415
+ def _table_transformer_job(self, ZM, auto_rotate=True):
416
+ """
417
+ Process table structure recognition.
418
+
419
+ When auto_rotate=True, the complete workflow:
420
+ 1. Evaluate table orientation and select the best rotation angle
421
+ 2. Use rotated image for table structure recognition (TSR)
422
+ 3. Re-OCR the rotated image
423
+ 4. Match new OCR results with TSR cell coordinates
424
+
425
+ Args:
426
+ ZM: Zoom factor
427
+ auto_rotate: Whether to enable auto orientation correction
428
+ """
429
+ logging.debug("Table processing...")
430
+ imgs, pos = [], []
431
+ tbcnt = [0]
432
+ MARGIN = 10
433
+ self.tb_cpns = []
434
+ self.table_rotations = {} # Store rotation info for each table
435
+ self.rotated_table_imgs = {} # Store rotated table images
436
+
437
+ assert len(self.page_layout) == len(self.page_images)
438
+
439
+ # Collect layout info for all tables
440
+ table_layouts = []
441
+
442
+ table_index = 0
443
+ for p, tbls in enumerate(self.page_layout): # for page
444
+ tbls = [f for f in tbls if f["type"] == "table"]
445
+ tbcnt.append(len(tbls))
446
+ if not tbls:
447
+ continue
448
+ for page_table_index, tb in enumerate(tbls): # for table
449
+ left, top, right, bott = tb["x0"] - MARGIN, tb["top"] - MARGIN, tb["x1"] + MARGIN, tb["bottom"] + MARGIN
450
+ left *= ZM
451
+ top *= ZM
452
+ right *= ZM
453
+ bott *= ZM
454
+ layoutno = f"table-{page_table_index}"
455
+ pos.append((left, top, p, table_index, layoutno))
456
+
457
+ # Record table layout info
458
+ table_layouts.append({"page": p, "table_index": table_index, "layoutno": layoutno, "layout": tb, "coords": (left, top, right, bott)})
459
+
460
+ # Crop table image
461
+ table_img = self.page_images[p].crop((left, top, right, bott))
462
+
463
+ if auto_rotate:
464
+ # Evaluate table orientation
465
+ logging.debug(f"Evaluating orientation for table {table_index} on page {p}")
466
+ best_angle, rotated_img, rotation_scores = self._evaluate_table_orientation(table_img)
467
+
468
+ # Store rotation info
469
+ self.table_rotations[table_index] = {
470
+ "page": p,
471
+ "original_pos": (left, top, right, bott),
472
+ "best_angle": best_angle,
473
+ "scores": rotation_scores,
474
+ "rotated_size": rotated_img.size, # (width, height)
475
+ }
476
+
477
+ # Store the rotated image
478
+ self.rotated_table_imgs[table_index] = rotated_img
479
+ imgs.append(rotated_img)
480
+
481
+ else:
482
+ imgs.append(table_img)
483
+ self.table_rotations[table_index] = {"page": p, "original_pos": (left, top, right, bott), "best_angle": 0, "scores": {}, "rotated_size": table_img.size}
484
+ self.rotated_table_imgs[table_index] = table_img
485
+
486
+ table_index += 1
487
+
488
+ assert len(self.page_images) == len(tbcnt) - 1
489
+ if not imgs:
490
+ return
491
+
492
+ # Perform table structure recognition (TSR)
493
+ recos = self.tbl_det(imgs)
494
+
495
+ # If tables were rotated, re-OCR the rotated images and replace table boxes
496
+ if auto_rotate:
497
+ self._ocr_rotated_tables(ZM, table_layouts, recos, tbcnt)
498
+
499
+ def _map_tsr_component_to_page_space(component, table_pos):
500
+ crop_left, crop_top, page, table_index, _ = table_pos
501
+ rotation_info = self.table_rotations.get(table_index, {})
502
+ angle = rotation_info.get("best_angle", 0)
503
+ original_pos = rotation_info.get("original_pos", (crop_left, crop_top, crop_left, crop_top))
504
+ width = original_pos[2] - original_pos[0]
505
+ height = original_pos[3] - original_pos[1]
506
+ points = [
507
+ (component["x0_rotated"], component["top_rotated"]),
508
+ (component["x1_rotated"], component["top_rotated"]),
509
+ (component["x0_rotated"], component["bottom_rotated"]),
510
+ (component["x1_rotated"], component["bottom_rotated"]),
511
+ ]
512
+ mapped = [self._map_clockwise_rotated_point_to_original(x, y, angle, width, height) for x, y in points]
513
+ xs = [p[0] for p in mapped]
514
+ ys = [p[1] for p in mapped]
515
+ component["x0"] = min(xs) / ZM + crop_left / ZM
516
+ component["x1"] = max(xs) / ZM + crop_left / ZM
517
+ component["top"] = min(ys) / ZM + crop_top / ZM + self.page_cum_height[page]
518
+ component["bottom"] = max(ys) / ZM + crop_top / ZM + self.page_cum_height[page]
519
+
520
+ # Process TSR results and align structure boxes with page-cumulative OCR boxes.
521
+ tbcnt = np.cumsum(tbcnt)
522
+ for i in range(len(tbcnt) - 1): # for page
523
+ pg = []
524
+ for j, tb_items in enumerate(recos[tbcnt[i] : tbcnt[i + 1]]): # for table
525
+ poss = pos[tbcnt[i] : tbcnt[i + 1]]
526
+ for it in tb_items: # for table components
527
+ # TSR coordinates are relative to rotated image, need to record
528
+ it["x0_rotated"] = it["x0"]
529
+ it["x1_rotated"] = it["x1"]
530
+ it["top_rotated"] = it["top"]
531
+ it["bottom_rotated"] = it["bottom"]
532
+
533
+ it["pn"] = poss[j][2] # page number
534
+ it["layoutno"] = poss[j][4]
535
+ it["table_index"] = poss[j][3] # table index
536
+ _map_tsr_component_to_page_space(it, poss[j])
537
+ pg.append(it)
538
+ self.tb_cpns.extend(pg)
539
+
540
+ def gather(kwd, fzy=10, ption=0.6):
541
+ eles = Recognizer.sort_Y_firstly([r for r in self.tb_cpns if re.match(kwd, r["label"])], fzy)
542
+ eles = Recognizer.layouts_cleanup(self.boxes, eles, 5, ption)
543
+ return Recognizer.sort_Y_firstly(eles, 0)
544
+
545
+ # add R,H,C,SP tag to boxes within table layout
546
+ headers = gather(r".*header$")
547
+ rows = gather(r".* (row|header)")
548
+ spans = gather(r".*spanning")
549
+ clmns = sorted([r for r in self.tb_cpns if re.match(r"table column$", r["label"])], key=lambda x: (x["pn"], x["layoutno"], x["x0"]))
550
+ clmns = Recognizer.layouts_cleanup(self.boxes, clmns, 5, 0.5)
551
+
552
+ for b in self.boxes:
553
+ if b.get("layout_type", "") != "table":
554
+ continue
555
+ ii = Recognizer.find_overlapped_with_threshold(b, rows, thr=0.3)
556
+ if ii is not None:
557
+ b["R"] = ii
558
+ b["R_top"] = rows[ii]["top"]
559
+ b["R_bott"] = rows[ii]["bottom"]
560
+
561
+ ii = Recognizer.find_overlapped_with_threshold(b, headers, thr=0.3)
562
+ if ii is not None:
563
+ b["H_top"] = headers[ii]["top"]
564
+ b["H_bott"] = headers[ii]["bottom"]
565
+ b["H_left"] = headers[ii]["x0"]
566
+ b["H_right"] = headers[ii]["x1"]
567
+ b["H"] = ii
568
+
569
+ ii = Recognizer.find_horizontally_tightest_fit(b, clmns)
570
+ if ii is not None:
571
+ b["C"] = ii
572
+ b["C_left"] = clmns[ii]["x0"]
573
+ b["C_right"] = clmns[ii]["x1"]
574
+
575
+ ii = Recognizer.find_overlapped_with_threshold(b, spans, thr=0.3)
576
+ if ii is not None:
577
+ b["H_top"] = spans[ii]["top"]
578
+ b["H_bott"] = spans[ii]["bottom"]
579
+ b["H_left"] = spans[ii]["x0"]
580
+ b["H_right"] = spans[ii]["x1"]
581
+ b["SP"] = ii
582
+
583
+ def _ocr_rotated_tables(self, ZM, table_layouts, tsr_results, tbcnt):
584
+ """
585
+ Re-OCR rotated table images and update self.boxes.
586
+
587
+ Args:
588
+ ZM: Zoom factor
589
+ table_layouts: List of table layout info
590
+ tsr_results: TSR recognition results
591
+ tbcnt: Cumulative table count per page
592
+ """
593
+ tbcnt = np.cumsum(tbcnt)
594
+
595
+ def _table_region(layout, page_index):
596
+ table_x0 = layout["x0"]
597
+ table_top = layout["top"]
598
+ table_x1 = layout["x1"]
599
+ table_bottom = layout["bottom"]
600
+ table_top_cum = table_top + self.page_cum_height[page_index]
601
+ table_bottom_cum = table_bottom + self.page_cum_height[page_index]
602
+ return table_x0, table_top, table_x1, table_bottom, table_top_cum, table_bottom_cum
603
+
604
+ def _collect_table_boxes(page_index, table_x0, table_x1, table_top_cum, table_bottom_cum):
605
+ indices = [
606
+ i
607
+ for i, b in enumerate(self.boxes)
608
+ if (
609
+ b.get("page_number") == page_index + self.page_from
610
+ and b.get("layout_type") == "table"
611
+ and b["x0"] >= table_x0 - 5
612
+ and b["x1"] <= table_x1 + 5
613
+ and b["top"] >= table_top_cum - 5
614
+ and b["bottom"] <= table_bottom_cum + 5
615
+ )
616
+ ]
617
+ original_boxes = [self.boxes[i] for i in indices]
618
+ insert_at = indices[0] if indices else len(self.boxes)
619
+ for i in reversed(indices):
620
+ self.boxes.pop(i)
621
+ return original_boxes, insert_at
622
+
623
+ def _restore_boxes(original_boxes, insert_at):
624
+ for b in original_boxes:
625
+ self.boxes.insert(insert_at, b)
626
+ insert_at += 1
627
+ return insert_at
628
+
629
+ def _insert_ocr_boxes(ocr_results, page_index, crop_left, crop_top, insert_at, table_index, layoutno, best_angle, table_w_px, table_h_px):
630
+ added = 0
631
+ for bbox, (text, conf) in ocr_results:
632
+ if conf < 0.5:
633
+ continue
634
+ mapped = [self._map_clockwise_rotated_point_to_original(p[0], p[1], best_angle, table_w_px, table_h_px) for p in bbox]
635
+ x_coords = [p[0] for p in mapped]
636
+ y_coords = [p[1] for p in mapped]
637
+ box_x0 = min(x_coords) / ZM
638
+ box_x1 = max(x_coords) / ZM
639
+ box_top = min(y_coords) / ZM
640
+ box_bottom = max(y_coords) / ZM
641
+ new_box = {
642
+ "text": text,
643
+ "x0": box_x0 + crop_left / ZM,
644
+ "x1": box_x1 + crop_left / ZM,
645
+ "top": box_top + crop_top / ZM + self.page_cum_height[page_index],
646
+ "bottom": box_bottom + crop_top / ZM + self.page_cum_height[page_index],
647
+ "page_number": page_index + self.page_from,
648
+ "layout_type": "table",
649
+ "layoutno": layoutno,
650
+ "_rotated": True,
651
+ "_rotation_angle": best_angle,
652
+ "_table_index": table_index,
653
+ "_rotated_x0": box_x0,
654
+ "_rotated_x1": box_x1,
655
+ "_rotated_top": box_top,
656
+ "_rotated_bottom": box_bottom,
657
+ }
658
+ self.boxes.insert(insert_at, new_box)
659
+ insert_at += 1
660
+ added += 1
661
+ return added
662
+
663
+ for tbl_info in table_layouts:
664
+ table_index = tbl_info["table_index"]
665
+ page = tbl_info["page"]
666
+ layout = tbl_info["layout"]
667
+ layoutno = tbl_info["layoutno"]
668
+ left, top, right, bott = tbl_info["coords"]
669
+
670
+ rotation_info = self.table_rotations.get(table_index, {})
671
+ best_angle = rotation_info.get("best_angle", 0)
672
+
673
+ # Get the rotated table image
674
+ rotated_img = self.rotated_table_imgs.get(table_index)
675
+ if rotated_img is None:
676
+ continue
677
+
678
+ # If no rotation, keep original OCR boxes untouched.
679
+ if best_angle == 0:
680
+ continue
681
+
682
+ # Table region is defined by layout's x0, top, x1, bottom (page-local coords)
683
+ table_x0, table_top, table_x1, table_bottom, table_top_cum, table_bottom_cum = _table_region(layout, page)
684
+ original_boxes, insert_at = _collect_table_boxes(page, table_x0, table_x1, table_top_cum, table_bottom_cum)
685
+
686
+ logging.info(f"Re-OCR table {table_index} on page {page} with rotation {best_angle}°")
687
+
688
+ # Perform OCR on rotated image
689
+ img_array = np.array(rotated_img)
690
+ ocr_results = self.ocr(img_array)
691
+
692
+ if not ocr_results:
693
+ logging.warning(f"No OCR results for rotated table {table_index}, restoring originals")
694
+ _restore_boxes(original_boxes, insert_at)
695
+ continue
696
+
697
+ # Add new OCR results to self.boxes
698
+ # OCR coordinates are relative to rotated image, map back to original table coords
699
+ table_w_px = right - left
700
+ table_h_px = bott - top
701
+ added = _insert_ocr_boxes(
702
+ ocr_results,
703
+ page,
704
+ left,
705
+ top,
706
+ insert_at,
707
+ table_index,
708
+ layoutno,
709
+ best_angle,
710
+ table_w_px,
711
+ table_h_px,
712
+ )
713
+
714
+ logging.info(f"Added {added} OCR results from rotated table {table_index}")
715
+
716
+ def __ocr(self, pagenum, img, chars, ZM=3, device_id: int | None = None):
717
+ # start = timer()
718
+ bxs = self.ocr.detect(np.array(img), device_id)
719
+ # logging.info(f"__ocr detecting boxes of an image cost ({timer() - start}s)")
720
+
721
+ # start = timer()
722
+ if not bxs:
723
+ self.boxes.append([])
724
+ return
725
+ bxs = [(line[0], line[1][0]) for line in bxs]
726
+ bxs = Recognizer.sort_Y_firstly(
727
+ [
728
+ {"x0": b[0][0] / ZM, "x1": b[1][0] / ZM, "top": b[0][1] / ZM, "text": "", "txt": t, "bottom": b[-1][1] / ZM, "chars": [], "page_number": pagenum}
729
+ for b, t in bxs
730
+ if b[0][0] <= b[1][0] and b[0][1] <= b[-1][1]
731
+ ],
732
+ self.mean_height[pagenum - 1] / 3,
733
+ )
734
+
735
+ # merge chars in the same rect
736
+ for c in chars:
737
+ ii = Recognizer.find_overlapped(c, bxs)
738
+ if ii is None:
739
+ self.lefted_chars.append(c)
740
+ continue
741
+ ch = c["bottom"] - c["top"]
742
+ bh = bxs[ii]["bottom"] - bxs[ii]["top"]
743
+ if abs(ch - bh) / max(ch, bh) >= 0.7 and c["text"] != " ":
744
+ self.lefted_chars.append(c)
745
+ continue
746
+ bxs[ii]["chars"].append(c)
747
+
748
+ for b in bxs:
749
+ if not b["chars"]:
750
+ del b["chars"]
751
+ continue
752
+ box_chars = b["chars"]
753
+ m_ht = np.mean([c["height"] for c in box_chars])
754
+ garbled_count = 0
755
+ total_count = 0
756
+ for c in Recognizer.sort_Y_firstly(box_chars, m_ht):
757
+ if c["text"] == " " and b["text"]:
758
+ if re.match(r"[0-9a-zA-Zа-яА-Я,.?;:!%%]", b["text"][-1]):
759
+ b["text"] += " "
760
+ else:
761
+ b["text"] += c["text"]
762
+ for ch in c["text"]:
763
+ if not ch.isspace():
764
+ total_count += 1
765
+ if self._is_garbled_char(ch):
766
+ garbled_count += 1
767
+ del b["chars"]
768
+
769
+ # Strategy 1: PUA / unmapped CID characters. These are genuine garbage,
770
+ # so re-OCR regardless of script.
771
+ if total_count > 0 and garbled_count / total_count >= 0.5:
772
+ logging.info(
773
+ "Page %d: detected garbled pdfplumber text (garbled=%d/%d), falling back to OCR for box at (%.1f, %.1f)",
774
+ pagenum,
775
+ garbled_count,
776
+ total_count,
777
+ b["x0"],
778
+ b["top"],
779
+ )
780
+ b["text"] = ""
781
+ continue
782
+
783
+ # Keep a clean text layer the recogniser cannot spell: ocr.res is
784
+ # CJK+Latin, so re-OCRing e.g. a Cyrillic page only produces garbage.
785
+ if total_count > 0 and not self._ocr_can_represent(b["text"]):
786
+ continue
787
+
788
+ # Strategy 2: font-encoding garbling — all chars are ASCII
789
+ # punctuation from subset fonts (no CJK output)
790
+ if total_count > 0 and self._is_garbled_by_font_encoding(box_chars, min_chars=5):
791
+ logging.info(
792
+ "Page %d: detected font-encoding garbled text (%d chars), falling back to OCR for box at (%.1f, %.1f)",
793
+ pagenum,
794
+ total_count,
795
+ b["x0"],
796
+ b["top"],
797
+ )
798
+ b["text"] = ""
799
+
800
+ # logging.info(f"__ocr sorting {len(chars)} chars cost {timer() - start}s")
801
+ # start = timer()
802
+ boxes_to_reg = []
803
+ img_np = None
804
+ for b in bxs:
805
+ if not b["text"]:
806
+ if img_np is None:
807
+ img_np = np.asarray(img)
808
+ left, right, top, bott = b["x0"] * ZM, b["x1"] * ZM, b["top"] * ZM, b["bottom"] * ZM
809
+ b["box_image"] = self.ocr.get_rotate_crop_image(img_np, np.array([[left, top], [right, top], [right, bott], [left, bott]], dtype=np.float32))
810
+ boxes_to_reg.append(b)
811
+ del b["txt"]
812
+ texts = self.ocr.recognize_batch([b["box_image"] for b in boxes_to_reg], device_id)
813
+ for i in range(len(boxes_to_reg)):
814
+ boxes_to_reg[i]["text"] = texts[i]
815
+ del boxes_to_reg[i]["box_image"]
816
+ # logging.info(f"__ocr recognize {len(bxs)} boxes cost {timer() - start}s")
817
+ bxs = [b for b in bxs if b["text"]]
818
+ if self.mean_height[pagenum - 1] == 0:
819
+ self.mean_height[pagenum - 1] = np.median([b["bottom"] - b["top"] for b in bxs])
820
+ self.boxes.append(bxs)
821
+
822
+ def _layouts_rec(self, ZM, drop=True):
823
+ assert len(self.page_images) == len(self.boxes)
824
+ self.boxes, self.page_layout = self.layouter(self.page_images, self.boxes, ZM, drop=drop)
825
+ # cumlative Y
826
+ for i in range(len(self.boxes)):
827
+ self.boxes[i]["top"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]
828
+ self.boxes[i]["bottom"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]
829
+
830
+ def _assign_column(self, boxes, zoomin=3):
831
+ if not boxes:
832
+ return boxes
833
+ if all("col_id" in b for b in boxes):
834
+ return boxes
835
+
836
+ by_page = defaultdict(list)
837
+ for b in boxes:
838
+ by_page[b["page_number"]].append(b)
839
+
840
+ page_cols = {}
841
+
842
+ for pg, bxs in by_page.items():
843
+ if not bxs:
844
+ page_cols[pg] = 1
845
+ continue
846
+
847
+ x0s_raw = np.array([b["x0"] for b in bxs], dtype=float)
848
+
849
+ min_x0 = np.min(x0s_raw)
850
+ max_x1 = np.max([b["x1"] for b in bxs])
851
+ width = max_x1 - min_x0
852
+
853
+ INDENT_TOL = width * 0.12
854
+ x0s = []
855
+ for x in x0s_raw:
856
+ if abs(x - min_x0) < INDENT_TOL:
857
+ x0s.append([min_x0])
858
+ else:
859
+ x0s.append([x])
860
+ x0s = np.array(x0s, dtype=float)
861
+
862
+ max_try = min(4, len(bxs))
863
+ if max_try < 2:
864
+ max_try = 1
865
+ best_k = 1
866
+ best_score = -1
867
+
868
+ for k in range(1, max_try + 1):
869
+ km = KMeans(n_clusters=k, n_init="auto")
870
+ labels = km.fit_predict(x0s)
871
+
872
+ centers = np.sort(km.cluster_centers_.flatten())
873
+ if len(centers) > 1:
874
+ try:
875
+ score = silhouette_score(x0s, labels)
876
+ except ValueError:
877
+ continue
878
+ else:
879
+ score = 0
880
+ if score > best_score:
881
+ best_score = score
882
+ best_k = k
883
+
884
+ page_cols[pg] = best_k
885
+ logging.info(f"[Page {pg}] best_score={best_score:.2f}, best_k={best_k}")
886
+
887
+ global_cols = Counter(page_cols.values()).most_common(1)[0][0]
888
+ logging.info(f"Global column_num decided by majority: {global_cols}")
889
+
890
+ for pg, bxs in by_page.items():
891
+ if not bxs:
892
+ continue
893
+ k = page_cols[pg]
894
+ if len(bxs) < k:
895
+ k = 1
896
+ x0s = np.array([[b["x0"]] for b in bxs], dtype=float)
897
+ km = KMeans(n_clusters=k, n_init="auto")
898
+ labels = km.fit_predict(x0s)
899
+
900
+ centers = km.cluster_centers_.flatten()
901
+ order = np.argsort(centers)
902
+
903
+ remap = {orig: new for new, orig in enumerate(order)}
904
+
905
+ for b, lb in zip(bxs, labels):
906
+ b["col_id"] = remap[lb]
907
+
908
+ grouped = defaultdict(list)
909
+ for b in bxs:
910
+ grouped[b["col_id"]].append(b)
911
+
912
+ return boxes
913
+
914
+ def _text_merge(self, zoomin=3):
915
+ # merge adjusted boxes
916
+ bxs = self._assign_column(self.boxes, zoomin)
917
+
918
+ def end_with(b, txt):
919
+ txt = txt.strip()
920
+ tt = b.get("text", "").strip()
921
+ return tt and tt.find(txt) == len(tt) - len(txt)
922
+
923
+ def start_with(b, txts):
924
+ tt = b.get("text", "").strip()
925
+ return tt and any([tt.find(t.strip()) == 0 for t in txts])
926
+
927
+ # horizontally merge adjacent box with the same layout
928
+ i = 0
929
+ while i < len(bxs) - 1:
930
+ b = bxs[i]
931
+ b_ = bxs[i + 1]
932
+
933
+ if b["page_number"] != b_["page_number"] or b.get("col_id") != b_.get("col_id"):
934
+ i += 1
935
+ continue
936
+
937
+ if b.get("layoutno", "0") != b_.get("layoutno", "1") or b.get("layout_type", "") in ["table", "figure", "equation"]:
938
+ i += 1
939
+ continue
940
+
941
+ if abs(self._y_dis(b, b_)) < self.mean_height[bxs[i]["page_number"] - 1] / 3:
942
+ # merge
943
+ bxs[i]["x1"] = b_["x1"]
944
+ bxs[i]["top"] = (b["top"] + b_["top"]) / 2
945
+ bxs[i]["bottom"] = (b["bottom"] + b_["bottom"]) / 2
946
+ bxs[i]["text"] += b_["text"]
947
+ bxs.pop(i + 1)
948
+ continue
949
+ i += 1
950
+ self.boxes = bxs
951
+
952
+ def _naive_vertical_merge(self, zoomin=3):
953
+ # bxs = self._assign_column(self.boxes, zoomin)
954
+ bxs = self.boxes
955
+
956
+ grouped = defaultdict(list)
957
+ for b in bxs:
958
+ # grouped[(b["page_number"], b.get("col_id", 0))].append(b)
959
+ grouped[(b["page_number"], "x")].append(b)
960
+
961
+ merged_boxes = []
962
+ for (pg, col), bxs in grouped.items():
963
+ bxs = sorted(bxs, key=lambda x: (x["top"], x["x0"]))
964
+ if not bxs:
965
+ continue
966
+
967
+ mh = self.mean_height[pg - 1] if self.mean_height else np.median([b["bottom"] - b["top"] for b in bxs]) or 10
968
+
969
+ i = 0
970
+ while i + 1 < len(bxs):
971
+ b = bxs[i]
972
+ b_ = bxs[i + 1]
973
+
974
+ if b["page_number"] < b_["page_number"] and re.match(r"[0-9 •一—-]+$", b["text"]):
975
+ bxs.pop(i)
976
+ continue
977
+
978
+ if not b["text"].strip():
979
+ bxs.pop(i)
980
+ continue
981
+
982
+ if not b["text"].strip() or b.get("layoutno") != b_.get("layoutno"):
983
+ i += 1
984
+ continue
985
+
986
+ if b_["top"] - b["bottom"] > mh * 1.5:
987
+ i += 1
988
+ continue
989
+
990
+ overlap = max(0, min(b["x1"], b_["x1"]) - max(b["x0"], b_["x0"]))
991
+ if overlap / max(1, min(b["x1"] - b["x0"], b_["x1"] - b_["x0"])) < 0.3:
992
+ i += 1
993
+ continue
994
+
995
+ concatting_feats = [
996
+ b["text"].strip()[-1] in ",;:'\",、‘“;:-",
997
+ len(b["text"].strip()) > 1 and b["text"].strip()[-2] in ",;:'\",‘“、;:",
998
+ b_["text"].strip() and b_["text"].strip()[0] in "。;?!?”)),,、:",
999
+ ]
1000
+ # features for not concating
1001
+ feats = [
1002
+ b.get("layoutno", 0) != b_.get("layoutno", 0),
1003
+ b["text"].strip()[-1] in "。?!?",
1004
+ self.is_english and b["text"].strip()[-1] in ".!?",
1005
+ b["page_number"] == b_["page_number"] and b_["top"] - b["bottom"] > self.mean_height[b["page_number"] - 1] * 1.5,
1006
+ b["page_number"] < b_["page_number"] and abs(b["x0"] - b_["x0"]) > self.mean_width[b["page_number"] - 1] * 4,
1007
+ ]
1008
+ # split features
1009
+ detach_feats = [b["x1"] < b_["x0"], b["x0"] > b_["x1"]]
1010
+ if (any(feats) and not any(concatting_feats)) or any(detach_feats):
1011
+ logging.debug(
1012
+ "{} {} {} {}".format(
1013
+ b["text"],
1014
+ b_["text"],
1015
+ any(feats),
1016
+ any(concatting_feats),
1017
+ )
1018
+ )
1019
+ i += 1
1020
+ continue
1021
+
1022
+ b["text"] = (b["text"].rstrip() + " " + b_["text"].lstrip()).strip()
1023
+ b["bottom"] = b_["bottom"]
1024
+ b["x0"] = min(b["x0"], b_["x0"])
1025
+ b["x1"] = max(b["x1"], b_["x1"])
1026
+ bxs.pop(i + 1)
1027
+
1028
+ merged_boxes.extend(bxs)
1029
+
1030
+ # self.boxes = sorted(merged_boxes, key=lambda x: (x["page_number"], x.get("col_id", 0), x["top"]))
1031
+ self.boxes = merged_boxes
1032
+
1033
+ def _concat_downward(self, concat_between_pages=True):
1034
+ self.boxes = Recognizer.sort_Y_firstly(self.boxes, 0)
1035
+ return
1036
+
1037
+ def _filter_forpages(self):
1038
+ if not self.boxes:
1039
+ return
1040
+ findit = False
1041
+ i = 0
1042
+ while i < len(self.boxes):
1043
+ if not re.match(r"(contents|目录|目次|table of contents|致谢|acknowledge)$", re.sub(r"( | |\u3000)+", "", self.boxes[i]["text"].lower())):
1044
+ i += 1
1045
+ continue
1046
+ findit = True
1047
+ eng = re.match(r"[0-9a-zA-Z :'.-]{5,}", self.boxes[i]["text"].strip())
1048
+ self.boxes.pop(i)
1049
+ if i >= len(self.boxes):
1050
+ break
1051
+ prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
1052
+ while not prefix:
1053
+ self.boxes.pop(i)
1054
+ if i >= len(self.boxes):
1055
+ break
1056
+ prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
1057
+ self.boxes.pop(i)
1058
+ if i >= len(self.boxes) or not prefix:
1059
+ break
1060
+ for j in range(i, min(i + 128, len(self.boxes))):
1061
+ if not re.match(prefix, self.boxes[j]["text"]):
1062
+ continue
1063
+ for k in range(i, j):
1064
+ self.boxes.pop(i)
1065
+ break
1066
+ if findit:
1067
+ return
1068
+
1069
+ page_dirty = [0] * len(self.page_images)
1070
+ for b in self.boxes:
1071
+ if re.search(r"(··|··|··)", b["text"]):
1072
+ page_dirty[b["page_number"] - 1] += 1
1073
+ page_dirty = set([i + 1 for i, t in enumerate(page_dirty) if t > 3])
1074
+ if not page_dirty:
1075
+ return
1076
+ i = 0
1077
+ while i < len(self.boxes):
1078
+ if self.boxes[i]["page_number"] in page_dirty:
1079
+ self.boxes.pop(i)
1080
+ continue
1081
+ i += 1
1082
+
1083
+ def _merge_with_same_bullet(self):
1084
+ i = 0
1085
+ while i + 1 < len(self.boxes):
1086
+ b = self.boxes[i]
1087
+ b_ = self.boxes[i + 1]
1088
+ if not b["text"].strip():
1089
+ self.boxes.pop(i)
1090
+ continue
1091
+ if not b_["text"].strip():
1092
+ self.boxes.pop(i + 1)
1093
+ continue
1094
+
1095
+ if (
1096
+ b["text"].strip()[0] != b_["text"].strip()[0]
1097
+ or b["text"].strip()[0].lower() in set("qwertyuopasdfghjklzxcvbnm")
1098
+ or is_chinese(b["text"].strip()[0])
1099
+ or b["top"] > b_["bottom"]
1100
+ ):
1101
+ i += 1
1102
+ continue
1103
+ b_["text"] = b["text"] + "\n" + b_["text"]
1104
+ b_["x0"] = min(b["x0"], b_["x0"])
1105
+ b_["x1"] = max(b["x1"], b_["x1"])
1106
+ b_["top"] = b["top"]
1107
+ self.boxes.pop(i)
1108
+
1109
+ def _extract_table_figure(self, need_image, ZM, return_html, need_position, separate_tables_figures=False):
1110
+ tables = {}
1111
+ figures = {}
1112
+ # extract figure and table boxes
1113
+ i = 0
1114
+ lst_lout_no = ""
1115
+ nomerge_lout_no = []
1116
+ while i < len(self.boxes):
1117
+ if "layoutno" not in self.boxes[i]:
1118
+ i += 1
1119
+ continue
1120
+ lout_no = str(self.boxes[i]["page_number"]) + "-" + str(self.boxes[i]["layoutno"])
1121
+ if TableStructureRecognizer.is_caption(self.boxes[i]) or self.boxes[i]["layout_type"] in ["table caption", "title", "figure caption", "reference"]:
1122
+ nomerge_lout_no.append(lst_lout_no)
1123
+ if self.boxes[i]["layout_type"] == "table":
1124
+ if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
1125
+ self.boxes.pop(i)
1126
+ continue
1127
+ if lout_no not in tables:
1128
+ tables[lout_no] = []
1129
+ tables[lout_no].append(self.boxes[i])
1130
+ self.boxes.pop(i)
1131
+ lst_lout_no = lout_no
1132
+ continue
1133
+ if need_image and self.boxes[i]["layout_type"] == "figure":
1134
+ if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
1135
+ self.boxes.pop(i)
1136
+ continue
1137
+ if lout_no not in figures:
1138
+ figures[lout_no] = []
1139
+ figures[lout_no].append(self.boxes[i])
1140
+ self.boxes.pop(i)
1141
+ lst_lout_no = lout_no
1142
+ continue
1143
+ i += 1
1144
+
1145
+ # merge table on different pages
1146
+ nomerge_lout_no = set(nomerge_lout_no)
1147
+ tbls = sorted([(k, bxs) for k, bxs in tables.items()], key=lambda x: (x[1][0]["top"], x[1][0]["x0"]))
1148
+
1149
+ i = len(tbls) - 1
1150
+ while i - 1 >= 0:
1151
+ k0, bxs0 = tbls[i - 1]
1152
+ k, bxs = tbls[i]
1153
+ i -= 1
1154
+ if k0 in nomerge_lout_no:
1155
+ continue
1156
+ if bxs[0]["page_number"] == bxs0[0]["page_number"]:
1157
+ continue
1158
+ if bxs[0]["page_number"] - bxs0[0]["page_number"] > 1:
1159
+ continue
1160
+ mh = self.mean_height[bxs[0]["page_number"] - 1]
1161
+ if self._y_dis(bxs0[-1], bxs[0]) > mh * 23:
1162
+ continue
1163
+ tables[k0].extend(tables[k])
1164
+ del tables[k]
1165
+
1166
+ def x_overlapped(a, b):
1167
+ return not any([a["x1"] < b["x0"], a["x0"] > b["x1"]])
1168
+
1169
+ # find captions and pop out
1170
+ i = 0
1171
+ while i < len(self.boxes):
1172
+ c = self.boxes[i]
1173
+ # mh = self.mean_height[c["page_number"]-1]
1174
+ if not TableStructureRecognizer.is_caption(c):
1175
+ i += 1
1176
+ continue
1177
+
1178
+ # find the nearest layouts
1179
+ def nearest(tbls):
1180
+ nonlocal c
1181
+ mink = ""
1182
+ minv = 1000000000
1183
+ for k, bxs in tbls.items():
1184
+ for b in bxs:
1185
+ if b.get("layout_type", "").find("caption") >= 0:
1186
+ continue
1187
+ y_dis = self._y_dis(c, b)
1188
+ x_dis = self._x_dis(c, b) if not x_overlapped(c, b) else 0
1189
+ dis = y_dis * y_dis + x_dis * x_dis
1190
+ if dis < minv:
1191
+ mink = k
1192
+ minv = dis
1193
+ return mink, minv
1194
+
1195
+ tk, tv = nearest(tables)
1196
+ fk, fv = nearest(figures)
1197
+ # if min(tv, fv) > 2000:
1198
+ # i += 1
1199
+ # continue
1200
+ if tv < fv and tk:
1201
+ tables[tk].insert(0, c)
1202
+ logging.debug("TABLE:" + self.boxes[i]["text"] + "; Cap: " + tk)
1203
+ elif fk:
1204
+ figures[fk].insert(0, c)
1205
+ logging.debug("FIGURE:" + self.boxes[i]["text"] + "; Cap: " + tk)
1206
+ self.boxes.pop(i)
1207
+
1208
+ def cropout(bxs, ltype, poss):
1209
+ nonlocal ZM
1210
+ max_page_index = len(self.page_images) - 1
1211
+
1212
+ def local_page_index(page_number):
1213
+ idx = page_number - 1 if page_number > 0 else 0
1214
+ if idx > max_page_index and self.page_from:
1215
+ idx = page_number - 1 - self.page_from
1216
+ return idx
1217
+
1218
+ pn = set()
1219
+ for b in bxs:
1220
+ idx = local_page_index(b["page_number"])
1221
+ if 0 <= idx <= max_page_index:
1222
+ pn.add(idx)
1223
+ else:
1224
+ logging.warning(
1225
+ "Skip out-of-range page_number %s (page_from=%s, pages=%s)",
1226
+ b.get("page_number"),
1227
+ self.page_from,
1228
+ len(self.page_images),
1229
+ )
1230
+
1231
+ if not pn:
1232
+ return None
1233
+
1234
+ if len(pn) < 2:
1235
+ pn = list(pn)[0]
1236
+ ht = self.page_cum_height[pn]
1237
+ b = {"x0": np.min([b["x0"] for b in bxs]), "top": np.min([b["top"] for b in bxs]) - ht, "x1": np.max([b["x1"] for b in bxs]), "bottom": np.max([b["bottom"] for b in bxs]) - ht}
1238
+ louts = [layout for layout in self.page_layout[pn] if layout["type"] == ltype]
1239
+ ii = Recognizer.find_overlapped(b, louts, naive=True)
1240
+ if ii is not None:
1241
+ b = louts[ii]
1242
+ else:
1243
+ logging.warning(f"Missing layout match: {pn + 1},%s" % (bxs[0].get("layoutno", "")))
1244
+
1245
+ left, top, right, bott = b["x0"], b["top"], b["x1"], b["bottom"]
1246
+ if right < left:
1247
+ right = left + 1
1248
+ poss.append((pn + self.page_from, left, right, top, bott))
1249
+ return self.page_images[pn].crop((left * ZM, top * ZM, right * ZM, bott * ZM))
1250
+ pn = {}
1251
+ for b in bxs:
1252
+ p = local_page_index(b["page_number"])
1253
+ if 0 <= p <= max_page_index:
1254
+ if p not in pn:
1255
+ pn[p] = []
1256
+ pn[p].append(b)
1257
+ pn = sorted(pn.items(), key=lambda x: x[0])
1258
+ imgs = [cropout(arr, ltype, poss) for p, arr in pn]
1259
+ imgs = [img for img in imgs if img is not None]
1260
+ if not imgs:
1261
+ return None
1262
+ pic = Image.new("RGB", (int(np.max([i.size[0] for i in imgs])), int(np.sum([m.size[1] for m in imgs]))), (245, 245, 245))
1263
+ height = 0
1264
+ for img in imgs:
1265
+ pic.paste(img, (0, int(height)))
1266
+ height += img.size[1]
1267
+ return pic
1268
+
1269
+ res = []
1270
+ positions = []
1271
+ figure_results = []
1272
+ figure_positions = []
1273
+ # crop figure out and add caption
1274
+ for k, bxs in figures.items():
1275
+ txt = "\n".join([b["text"] for b in bxs])
1276
+ if not txt:
1277
+ continue
1278
+
1279
+ poss = []
1280
+
1281
+ if separate_tables_figures:
1282
+ img = cropout(bxs, "figure", poss)
1283
+ if img is None:
1284
+ continue
1285
+ figure_results.append((img, [txt]))
1286
+ figure_positions.append(poss)
1287
+ else:
1288
+ img = cropout(bxs, "figure", poss)
1289
+ if img is None:
1290
+ continue
1291
+ res.append((img, [txt]))
1292
+ positions.append(poss)
1293
+
1294
+ for k, bxs in tables.items():
1295
+ if not bxs:
1296
+ continue
1297
+ bxs = Recognizer.sort_Y_firstly(bxs, np.mean([(b["bottom"] - b["top"]) / 2 for b in bxs]))
1298
+
1299
+ poss = []
1300
+
1301
+ img = cropout(bxs, "table", poss)
1302
+ if img is None:
1303
+ continue
1304
+ res.append((img, self.tbl_det.construct_table(bxs, html=return_html, is_english=self.is_english)))
1305
+ positions.append(poss)
1306
+
1307
+ if separate_tables_figures:
1308
+ assert len(positions) + len(figure_positions) == len(res) + len(figure_results)
1309
+ if need_position:
1310
+ return list(zip(res, positions)), list(zip(figure_results, figure_positions))
1311
+ else:
1312
+ return res, figure_results
1313
+ else:
1314
+ assert len(positions) == len(res)
1315
+ if need_position:
1316
+ return list(zip(res, positions))
1317
+ else:
1318
+ return res
1319
+
1320
+ def proj_match(self, line):
1321
+ if len(line) <= 2:
1322
+ return
1323
+ if re.match(r"[0-9 ().,%%+/-]+$", line):
1324
+ return False
1325
+ for p, j in [
1326
+ (r"第[零一二三四五六七八九十百]+章", 1),
1327
+ (r"第[零一二三四五六七八九十百]+[条节]", 2),
1328
+ (r"[零一二三四五六七八九十百]+[、  ]", 3),
1329
+ (r"[\((][零一二三四五六七八九十百]+[)\)]", 4),
1330
+ (r"[0-9]+(、|\.[  ]|\.[^0-9])", 5),
1331
+ (r"[0-9]+\.[0-9]+(、|[.  ]|[^0-9])", 6),
1332
+ (r"[0-9]+\.[0-9]+\.[0-9]+(、|[  ]|[^0-9])", 7),
1333
+ (r"[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+(、|[  ]|[^0-9])", 8),
1334
+ (r".{,48}[::??]$", 9),
1335
+ (r"[0-9]+)", 10),
1336
+ (r"[\((][0-9]+[)\)]", 11),
1337
+ (r"[零一二三四五六七八九十百]+是", 12),
1338
+ (r"[⚫•➢✓]", 12),
1339
+ ]:
1340
+ if re.match(p, line):
1341
+ return j
1342
+ return
1343
+
1344
+ def _line_tag(self, bx, ZM):
1345
+ pn = [bx["page_number"]]
1346
+ top = bx["top"] - self.page_cum_height[pn[0] - 1]
1347
+ bott = bx["bottom"] - self.page_cum_height[pn[0] - 1]
1348
+ page_images_cnt = len(self.page_images)
1349
+ if pn[-1] - 1 >= page_images_cnt:
1350
+ return ""
1351
+ while bott * ZM > self.page_images[pn[-1] - 1].size[1]:
1352
+ bott -= self.page_images[pn[-1] - 1].size[1] / ZM
1353
+ pn.append(pn[-1] + 1)
1354
+ if pn[-1] - 1 >= page_images_cnt:
1355
+ return ""
1356
+
1357
+ return "@@{}\t{:.1f}\t{:.1f}\t{:.1f}\t{:.1f}##".format("-".join([str(p) for p in pn]), bx["x0"], bx["x1"], top, bott)
1358
+
1359
+ def __filterout_scraps(self, boxes, ZM):
1360
+ def width(b):
1361
+ return b["x1"] - b["x0"]
1362
+
1363
+ def height(b):
1364
+ return b["bottom"] - b["top"]
1365
+
1366
+ def usefull(b):
1367
+ if b.get("layout_type"):
1368
+ return True
1369
+ if width(b) > self.page_images[b["page_number"] - 1].size[0] / ZM / 3:
1370
+ return True
1371
+ if b["bottom"] - b["top"] > self.mean_height[b["page_number"] - 1]:
1372
+ return True
1373
+ return False
1374
+
1375
+ res = []
1376
+ while boxes:
1377
+ lines = []
1378
+ widths = []
1379
+ pw = self.page_images[boxes[0]["page_number"] - 1].size[0] / ZM
1380
+ mh = self.mean_height[boxes[0]["page_number"] - 1]
1381
+ mj = self.proj_match(boxes[0]["text"]) or boxes[0].get("layout_type", "") == "title"
1382
+
1383
+ def dfs(line, st):
1384
+ nonlocal mh, pw, lines, widths
1385
+ lines.append(line)
1386
+ widths.append(width(line))
1387
+ mmj = self.proj_match(line["text"]) or line.get("layout_type", "") == "title"
1388
+ for i in range(st + 1, min(st + 20, len(boxes))):
1389
+ if (boxes[i]["page_number"] - line["page_number"]) > 0:
1390
+ break
1391
+ if not mmj and self._y_dis(line, boxes[i]) >= 3 * mh and height(line) < 1.5 * mh:
1392
+ break
1393
+
1394
+ if not usefull(boxes[i]):
1395
+ continue
1396
+ if mmj or (self._x_dis(boxes[i], line) < pw / 10):
1397
+ # and abs(width(boxes[i])-width_mean)/max(width(boxes[i]),width_mean)<0.5):
1398
+ # concat following
1399
+ dfs(boxes[i], i)
1400
+ boxes.pop(i)
1401
+ break
1402
+
1403
+ try:
1404
+ if usefull(boxes[0]):
1405
+ dfs(boxes[0], 0)
1406
+ else:
1407
+ logging.debug("WASTE: " + boxes[0]["text"])
1408
+ except Exception:
1409
+ pass
1410
+ boxes.pop(0)
1411
+ mw = np.mean(widths)
1412
+ if mj or mw / pw >= 0.35 or mw > 200:
1413
+ res.append("\n".join([c["text"] + self._line_tag(c, ZM) for c in lines]))
1414
+ else:
1415
+ logging.debug("REMOVED: " + "<<".join([c["text"] for c in lines]))
1416
+
1417
+ return "\n\n".join(res)
1418
+
1419
+ @staticmethod
1420
+ def total_page_number(fnm, binary=None):
1421
+ try:
1422
+ with sys.modules[LOCK_KEY_pdfplumber]:
1423
+ pdf = pdfplumber.open(fnm) if not binary else pdfplumber.open(BytesIO(binary))
1424
+ total_page = len(pdf.pages)
1425
+ pdf.close()
1426
+ return total_page
1427
+ except Exception:
1428
+ logging.exception("total_page_number")
1429
+
1430
+ def __images__(self, fnm, zoomin=3, page_from=0, page_to=MAXIMUM_PAGE_NUMBER, callback=None):
1431
+ self.lefted_chars = []
1432
+ self.mean_height = []
1433
+ self.mean_width = []
1434
+ self.boxes = []
1435
+ self.garbages = {}
1436
+ self.page_cum_height = [0]
1437
+ self.page_layout = []
1438
+ self.page_from = page_from
1439
+ start = timer()
1440
+ try:
1441
+ with sys.modules[LOCK_KEY_pdfplumber]:
1442
+ with pdfplumber.open(fnm) if isinstance(fnm, str) else pdfplumber.open(BytesIO(fnm)) as pdf:
1443
+ self.pdf = pdf
1444
+ self.page_images = [p.to_image(resolution=72 * zoomin, antialias=True).annotated for i, p in enumerate(self.pdf.pages[page_from:page_to])]
1445
+
1446
+ try:
1447
+ self.page_chars = [[c for c in page.dedupe_chars().chars if self._has_color(c)] for page in self.pdf.pages[page_from:page_to]]
1448
+ except Exception as e:
1449
+ logging.warning(f"Failed to extract characters for pages {page_from}-{page_to}: {str(e)}")
1450
+ self.page_chars = [[] for _ in range(len(self.page_images))] # If failed to extract, using empty list instead.
1451
+
1452
+ # Detect garbled pages and clear their chars so the OCR
1453
+ # path will be used instead. Two detection strategies:
1454
+ # 1) PUA / unmapped CID characters (threshold=0.3)
1455
+ # 2) Font-encoding garbling: subset fonts mapping CJK to ASCII
1456
+ for pi, page_ch in enumerate(self.page_chars):
1457
+ if not page_ch:
1458
+ continue
1459
+ # Strategy 1: PUA / CID garbling
1460
+ sample = page_ch if len(page_ch) <= 200 else page_ch[:200]
1461
+ sample_text = "".join(c.get("text", "") for c in sample)
1462
+ if self._is_garbled_text(sample_text, threshold=0.3):
1463
+ logging.warning(
1464
+ "Page %d: pdfplumber extracted mostly garbled characters (%d chars), clearing to use OCR fallback.",
1465
+ page_from + pi + 1,
1466
+ len(page_ch),
1467
+ )
1468
+ self.page_chars[pi] = []
1469
+ continue
1470
+ # Strategy 2: font-encoding garbling (CJK mapped to ASCII)
1471
+ if self._is_garbled_by_font_encoding(page_ch):
1472
+ logging.warning(
1473
+ "Page %d: detected font-encoding garbled text (subset fonts with no CJK output, %d chars), clearing to use OCR fallback.",
1474
+ page_from + pi + 1,
1475
+ len(page_ch),
1476
+ )
1477
+ self.page_chars[pi] = []
1478
+
1479
+ self.total_page = len(self.pdf.pages)
1480
+
1481
+ except Exception as e:
1482
+ logging.exception(f"RAGFlowPdfParser __images__, exception: {e}")
1483
+ logging.info(f"__images__ dedupe_chars cost {timer() - start}s")
1484
+
1485
+ logging.debug("Images converted.")
1486
+ self.is_english = [
1487
+ re.search(r"[ a-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join(random.choices([c["text"] for c in self.page_chars[i]], k=min(100, len(self.page_chars[i])))))
1488
+ for i in range(len(self.page_chars))
1489
+ ]
1490
+ if sum([1 if e else 0 for e in self.is_english]) > len(self.page_images) / 2:
1491
+ self.is_english = True
1492
+ else:
1493
+ self.is_english = False
1494
+
1495
+ async def __img_ocr(i, id, img, chars, limiter):
1496
+ self._insert_word_spaces(chars)
1497
+
1498
+ if limiter:
1499
+ async with limiter:
1500
+ await thread_pool_exec(self.__ocr, i + 1, img, chars, zoomin, id)
1501
+ else:
1502
+ self.__ocr(i + 1, img, chars, zoomin, id)
1503
+
1504
+ if callback and i % 6 == 5:
1505
+ callback((i + 1) * 0.6 / len(self.page_images))
1506
+
1507
+ async def __img_ocr_launcher():
1508
+ def __ocr_preprocess():
1509
+ chars = self.page_chars[i] if not self.is_english else []
1510
+ self.mean_height.append(np.median(sorted([c["height"] for c in chars])) if chars else 0)
1511
+ self.mean_width.append(np.median(sorted([c["width"] for c in chars])) if chars else 8)
1512
+ self.page_cum_height.append(img.size[1] / zoomin)
1513
+ return chars
1514
+
1515
+ if self.parallel_limiter:
1516
+ tasks = []
1517
+
1518
+ for i, img in enumerate(self.page_images):
1519
+ chars = __ocr_preprocess()
1520
+
1521
+ semaphore = self.parallel_limiter[i % PARALLEL_DEVICES]
1522
+
1523
+ async def wrapper(i=i, img=img, chars=chars, semaphore=semaphore):
1524
+ await __img_ocr(
1525
+ i,
1526
+ i % PARALLEL_DEVICES,
1527
+ img,
1528
+ chars,
1529
+ semaphore,
1530
+ )
1531
+
1532
+ tasks.append(asyncio.create_task(wrapper()))
1533
+ await asyncio.sleep(0)
1534
+
1535
+ try:
1536
+ await asyncio.gather(*tasks, return_exceptions=False)
1537
+ except Exception as e:
1538
+ logging.error(f"Error in OCR: {e}")
1539
+ for t in tasks:
1540
+ t.cancel()
1541
+ await asyncio.gather(*tasks, return_exceptions=True)
1542
+ raise
1543
+
1544
+ else:
1545
+ for i, img in enumerate(self.page_images):
1546
+ chars = __ocr_preprocess()
1547
+ await __img_ocr(i, 0, img, chars, None)
1548
+
1549
+ start = timer()
1550
+
1551
+ asyncio.run(__img_ocr_launcher())
1552
+
1553
+ logging.info(f"__images__ {len(self.page_images)} pages cost {timer() - start}s")
1554
+
1555
+ if not self.is_english and not any([c for c in self.page_chars]) and self.boxes:
1556
+ bxes = [b for bxs in self.boxes for b in bxs]
1557
+ self.is_english = re.search(r"[ \na-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join([b["text"] for b in random.choices(bxes, k=min(30, len(bxes)))]))
1558
+
1559
+ logging.debug(f"Is it English: {self.is_english}")
1560
+
1561
+ self.page_cum_height = np.cumsum(self.page_cum_height)
1562
+ assert len(self.page_cum_height) == len(self.page_images) + 1
1563
+ if len(self.boxes) == 0 and zoomin < 9:
1564
+ self.__images__(fnm, zoomin * 3, page_from, page_to, callback)
1565
+
1566
+ def __call__(self, fnm, need_image=True, zoomin=3, return_html=False, auto_rotate_tables=None):
1567
+ """
1568
+ Parse a PDF file.
1569
+
1570
+ Args:
1571
+ fnm: PDF file path or binary content
1572
+ need_image: Whether to extract images
1573
+ zoomin: Zoom factor
1574
+ return_html: Whether to return tables in HTML format
1575
+ auto_rotate_tables: Whether to enable auto orientation correction for tables.
1576
+ None: Use TABLE_AUTO_ROTATE env var setting (default: True)
1577
+ True: Enable auto orientation correction
1578
+ False: Disable auto orientation correction
1579
+ """
1580
+ if auto_rotate_tables is None:
1581
+ auto_rotate_tables = os.getenv("TABLE_AUTO_ROTATE", "true").lower() in ("true", "1", "yes")
1582
+
1583
+ self.outlines = extract_pdf_outlines(fnm)
1584
+ self.__images__(fnm, zoomin)
1585
+ self._layouts_rec(zoomin)
1586
+ self._table_transformer_job(zoomin, auto_rotate=auto_rotate_tables)
1587
+ self._text_merge()
1588
+ self._concat_downward()
1589
+ self._filter_forpages()
1590
+ tbls = self._extract_table_figure(need_image, zoomin, return_html, False)
1591
+ return self.__filterout_scraps(deepcopy(self.boxes), zoomin), tbls
1592
+
1593
+ def parse_into_bboxes(self, fnm, callback=None, zoomin=3, from_page=0, to_page=MAXIMUM_PAGE_NUMBER):
1594
+ self.outlines = extract_pdf_outlines(fnm)
1595
+ batch_size = max(1, int(os.getenv("PDF_PARSER_PAGE_BATCH_SIZE", "50")))
1596
+ if isinstance(fnm, str):
1597
+ total_pages = self.total_page_number(fnm)
1598
+ else:
1599
+ total_pages = self.total_page_number(fnm, binary=fnm)
1600
+
1601
+ if total_pages is None:
1602
+ effective_to_page = to_page
1603
+ logging.warning(
1604
+ "parse_into_bboxes: total_page_number returned None; using caller-supplied to_page=%s",
1605
+ to_page,
1606
+ )
1607
+ else:
1608
+ effective_to_page = min(to_page, total_pages)
1609
+
1610
+ if effective_to_page - from_page <= batch_size:
1611
+ self.__images__(fnm, zoomin, page_from=from_page, page_to=effective_to_page, callback=callback)
1612
+ return self._parse_loaded_window_into_bboxes(zoomin, callback=callback)
1613
+
1614
+ logging.info(
1615
+ "parse_into_bboxes uses chunk mode: from_page=%s, effective_to_page=%s, batch_size=%s",
1616
+ from_page,
1617
+ effective_to_page,
1618
+ batch_size,
1619
+ )
1620
+ all_boxes = []
1621
+ start = timer()
1622
+ for page_from in range(from_page, effective_to_page, batch_size):
1623
+ page_to = min(page_from + batch_size, effective_to_page)
1624
+ self.__images__(fnm, zoomin, page_from=page_from, page_to=page_to, callback=None)
1625
+ chunk_boxes = self._parse_loaded_window_into_bboxes(zoomin)
1626
+ all_boxes.extend(self._to_global_boxes(chunk_boxes))
1627
+ if callback:
1628
+ callback((page_to - from_page) / max(1, effective_to_page - from_page), f"Structured: {page_to}/{effective_to_page} pages")
1629
+
1630
+ logging.info("parse_into_bboxes chunk mode cost %.2fs", timer() - start)
1631
+ return all_boxes
1632
+
1633
+ def _parse_loaded_window_into_bboxes(self, zoomin=3, callback=None):
1634
+ start = timer()
1635
+ self._layouts_rec(zoomin)
1636
+ if callback:
1637
+ callback(0.63, "Layout analysis ({:.2f}s)".format(timer() - start))
1638
+
1639
+ auto_rotate_tables = os.getenv("TABLE_AUTO_ROTATE", "true").lower() in ("true", "1", "yes")
1640
+
1641
+ start = timer()
1642
+ self._table_transformer_job(zoomin, auto_rotate=auto_rotate_tables)
1643
+ if callback:
1644
+ callback(0.83, "Table analysis ({:.2f}s)".format(timer() - start))
1645
+
1646
+ start = timer()
1647
+ self._text_merge()
1648
+ self._concat_downward()
1649
+ self._naive_vertical_merge(zoomin)
1650
+ if callback:
1651
+ callback(0.92, "Text merged ({:.2f}s)".format(timer() - start))
1652
+
1653
+ start = timer()
1654
+ tbls, figs = self._extract_table_figure(True, zoomin, True, True, True)
1655
+
1656
+ def insert_table_figures(tbls_or_figs, layout_type):
1657
+ def min_rectangle_distance(rect1, rect2):
1658
+ pn1, left1, right1, top1, bottom1 = rect1
1659
+ pn2, left2, right2, top2, bottom2 = rect2
1660
+ if right1 >= left2 and right2 >= left1 and bottom1 >= top2 and bottom2 >= top1:
1661
+ return 0
1662
+ if right1 < left2:
1663
+ dx = left2 - right1
1664
+ elif right2 < left1:
1665
+ dx = left1 - right2
1666
+ else:
1667
+ dx = 0
1668
+ if bottom1 < top2:
1669
+ dy = top2 - bottom1
1670
+ elif bottom2 < top1:
1671
+ dy = top1 - bottom2
1672
+ else:
1673
+ dy = 0
1674
+ return math.sqrt(dx * dx + dy * dy)
1675
+
1676
+ for (img, txt), poss in tbls_or_figs:
1677
+ local_poss = []
1678
+ for pn, left, right, top, bott in poss:
1679
+ local_pn = pn - self.page_from
1680
+ if 0 <= local_pn < len(self.page_cum_height) - 1:
1681
+ local_poss.append((local_pn, left, right, top, bott))
1682
+ else:
1683
+ logging.debug(f"Skip out-of-range table/figure position pn={pn}, page_from={self.page_from}")
1684
+ if not local_poss:
1685
+ logging.debug("No valid local positions for table/figure; skip insertion.")
1686
+ continue
1687
+
1688
+ if isinstance(txt, list):
1689
+ txt = "\n".join(txt)
1690
+ pn, left, right, top, bott = local_poss[0]
1691
+ insert_at = len(self.boxes)
1692
+ bboxes = [(i, (b["page_number"], b["x0"], b["x1"], b["top"], b["bottom"])) for i, b in enumerate(self.boxes)]
1693
+ if bboxes:
1694
+ dists = [
1695
+ (min_rectangle_distance((cand_pn, cand_left, cand_right, cand_top + self.page_cum_height[cand_pn], cand_bott + self.page_cum_height[cand_pn]), rect), i)
1696
+ for i, rect in bboxes
1697
+ for cand_pn, cand_left, cand_right, cand_top, cand_bott in local_poss
1698
+ ]
1699
+ if dists:
1700
+ nearest_bbox_idx = int(np.argmin([dist for dist, _ in dists]))
1701
+ insert_at, _ = bboxes[dists[nearest_bbox_idx][-1]]
1702
+ if self.boxes[insert_at]["bottom"] < top + self.page_cum_height[pn]:
1703
+ insert_at += 1
1704
+ else:
1705
+ logging.debug("No text boxes available; append %s block directly.", layout_type)
1706
+ self.boxes.insert(
1707
+ insert_at,
1708
+ {
1709
+ "page_number": pn + 1,
1710
+ "x0": left,
1711
+ "x1": right,
1712
+ "top": top + self.page_cum_height[pn],
1713
+ "bottom": bott + self.page_cum_height[pn],
1714
+ "layout_type": layout_type,
1715
+ "text": txt,
1716
+ "image": img,
1717
+ "positions": [[pn + 1, int(left), int(right), int(top), int(bott)]],
1718
+ },
1719
+ )
1720
+
1721
+ for b in self.boxes:
1722
+ b["position_tag"] = self._line_tag(b, zoomin)
1723
+ b["image"] = self.crop(b["position_tag"], zoomin)
1724
+ b["positions"] = [[pos[0][-1] + 1, *pos[1:]] for pos in RAGFlowPdfParser.extract_positions(b["position_tag"])]
1725
+
1726
+ insert_table_figures(tbls, "table")
1727
+ insert_table_figures(figs, "figure")
1728
+ if callback:
1729
+ callback(1, "Structured ({:.2f}s)".format(timer() - start))
1730
+ return deepcopy(self.boxes)
1731
+
1732
+ @staticmethod
1733
+ def _offset_position_tag(text, page_offset):
1734
+ if not text or page_offset <= 0:
1735
+ return text
1736
+
1737
+ def _replace(match):
1738
+ pages = [str(int(p) + page_offset) for p in match.group(1).split("-")]
1739
+ return f"@@{'-'.join(pages)}\t"
1740
+
1741
+ return re.sub(r"@@([0-9-]+)\t", _replace, text)
1742
+
1743
+ def _to_global_boxes(self, boxes):
1744
+ if self.page_from <= 0:
1745
+ return boxes
1746
+
1747
+ for box in boxes:
1748
+ box["page_number"] = int(box.get("page_number", 1)) + self.page_from
1749
+ if isinstance(box.get("position_tag"), str):
1750
+ box["position_tag"] = self._offset_position_tag(box["position_tag"], self.page_from)
1751
+ if isinstance(box.get("positions"), list):
1752
+ box["positions"] = [[int(pos[0]) + self.page_from, *pos[1:]] if isinstance(pos, list) and len(pos) > 0 and isinstance(pos[0], (int, float)) else pos for pos in box["positions"]]
1753
+ return boxes
1754
+
1755
+ @staticmethod
1756
+ def remove_tag(txt):
1757
+ return re.sub(r"@@[\t0-9.-]+?##", "", txt)
1758
+
1759
+ @staticmethod
1760
+ def extract_positions(txt):
1761
+ poss = []
1762
+ for tag in re.findall(r"@@[0-9-]+\t[0-9.\t]+##", txt):
1763
+ pn, left, right, top, bottom = tag.strip("#").strip("@").split("\t")
1764
+ left, right, top, bottom = float(left), float(right), float(top), float(bottom)
1765
+ poss.append(([int(p) - 1 for p in pn.split("-")], left, right, top, bottom))
1766
+ return poss
1767
+
1768
+ def crop(self, text, ZM=3, need_position=False):
1769
+ imgs = []
1770
+ poss = self.extract_positions(text)
1771
+ if not poss:
1772
+ if need_position:
1773
+ return None, None
1774
+ return
1775
+
1776
+ if not getattr(self, "page_images", None):
1777
+ logging.warning("crop called without page images; skipping image generation.")
1778
+ if need_position:
1779
+ return None, None
1780
+ return
1781
+
1782
+ page_count = len(self.page_images)
1783
+
1784
+ filtered_poss = []
1785
+ for pns, left, right, top, bottom in poss:
1786
+ if not pns:
1787
+ logging.warning("Empty page index list in crop; skipping this position.")
1788
+ continue
1789
+ valid_pns = [p for p in pns if 0 <= p < page_count]
1790
+ if not valid_pns:
1791
+ logging.warning(f"All page indices {pns} out of range for {page_count} pages; skipping.")
1792
+ continue
1793
+ filtered_poss.append((valid_pns, left, right, top, bottom))
1794
+
1795
+ poss = filtered_poss
1796
+ if not poss:
1797
+ logging.warning("No valid positions after filtering; skip cropping.")
1798
+ if need_position:
1799
+ return None, None
1800
+ return
1801
+
1802
+ max_width = max(np.max([right - left for (_, left, right, _, _) in poss]), 6)
1803
+ GAP = 6
1804
+ pos = poss[0]
1805
+ first_page_idx = pos[0][0]
1806
+ poss.insert(0, ([first_page_idx], pos[1], pos[2], max(0, pos[3] - 120), max(pos[3] - GAP, 0)))
1807
+ pos = poss[-1]
1808
+ last_page_idx = pos[0][-1]
1809
+ if not (0 <= last_page_idx < page_count):
1810
+ logging.warning(f"Last page index {last_page_idx} out of range for {page_count} pages; skipping crop.")
1811
+ if need_position:
1812
+ return None, None
1813
+ return
1814
+ last_page_height = self.page_images[last_page_idx].size[1] / ZM
1815
+ poss.append(
1816
+ (
1817
+ [last_page_idx],
1818
+ pos[1],
1819
+ pos[2],
1820
+ min(last_page_height, pos[4] + GAP),
1821
+ min(last_page_height, pos[4] + 120),
1822
+ )
1823
+ )
1824
+
1825
+ positions = []
1826
+ for ii, (pns, left, right, top, bottom) in enumerate(poss):
1827
+ if 0 < ii < len(poss) - 1:
1828
+ right = max(left + 10, right)
1829
+ else:
1830
+ right = left + max_width
1831
+ bottom *= ZM
1832
+ for pn in pns[1:]:
1833
+ if 0 <= pn - 1 < page_count:
1834
+ bottom += self.page_images[pn - 1].size[1]
1835
+ else:
1836
+ logging.warning(f"Page index {pn}-1 out of range for {page_count} pages during crop; skipping height accumulation.")
1837
+
1838
+ if not (0 <= pns[0] < page_count):
1839
+ logging.warning(f"Base page index {pns[0]} out of range for {page_count} pages during crop; skipping this segment.")
1840
+ continue
1841
+
1842
+ imgs.append(self.page_images[pns[0]].crop((left * ZM, top * ZM, right * ZM, min(bottom, self.page_images[pns[0]].size[1]))))
1843
+ if 0 < ii < len(poss) - 1:
1844
+ positions.append((pns[0] + self.page_from, left, right, top, min(bottom, self.page_images[pns[0]].size[1]) / ZM))
1845
+ bottom -= self.page_images[pns[0]].size[1]
1846
+ for pn in pns[1:]:
1847
+ if not (0 <= pn < page_count):
1848
+ logging.warning(f"Page index {pn} out of range for {page_count} pages during crop; skipping this page.")
1849
+ continue
1850
+ imgs.append(self.page_images[pn].crop((left * ZM, 0, right * ZM, min(bottom, self.page_images[pn].size[1]))))
1851
+ if 0 < ii < len(poss) - 1:
1852
+ positions.append((pn + self.page_from, left, right, 0, min(bottom, self.page_images[pn].size[1]) / ZM))
1853
+ bottom -= self.page_images[pn].size[1]
1854
+
1855
+ if not imgs:
1856
+ if need_position:
1857
+ return None, None
1858
+ return
1859
+ height = 0
1860
+ for img in imgs:
1861
+ height += img.size[1] + GAP
1862
+ height = int(height)
1863
+ width = int(np.max([i.size[0] for i in imgs]))
1864
+ pic = Image.new("RGB", (width, height), (245, 245, 245))
1865
+ height = 0
1866
+ for ii, img in enumerate(imgs):
1867
+ if ii == 0 or ii + 1 == len(imgs):
1868
+ img = img.convert("RGBA")
1869
+ overlay = Image.new("RGBA", img.size, (0, 0, 0, 0))
1870
+ overlay.putalpha(128)
1871
+ img = Image.alpha_composite(img, overlay).convert("RGB")
1872
+ pic.paste(img, (0, int(height)))
1873
+ height += img.size[1] + GAP
1874
+
1875
+ if need_position:
1876
+ return pic, positions
1877
+ return pic
1878
+
1879
+ def get_position(self, bx, ZM):
1880
+ poss = []
1881
+ pn = bx["page_number"]
1882
+ top = bx["top"] - self.page_cum_height[pn - 1]
1883
+ bott = bx["bottom"] - self.page_cum_height[pn - 1]
1884
+ poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
1885
+ while bott * ZM > self.page_images[pn - 1].size[1]:
1886
+ bott -= self.page_images[pn - 1].size[1] / ZM
1887
+ top = 0
1888
+ pn += 1
1889
+ poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
1890
+ return poss
1891
+
1892
+
1893
+ if __name__ == "__main__":
1894
+ pass