langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1894 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
#
|
|
16
|
+
|
|
17
|
+
import asyncio
|
|
18
|
+
import logging
|
|
19
|
+
import math
|
|
20
|
+
import os
|
|
21
|
+
import random
|
|
22
|
+
import re
|
|
23
|
+
import sys
|
|
24
|
+
import threading
|
|
25
|
+
import unicodedata
|
|
26
|
+
from collections import Counter, defaultdict
|
|
27
|
+
from copy import deepcopy
|
|
28
|
+
from io import BytesIO
|
|
29
|
+
from timeit import default_timer as timer
|
|
30
|
+
|
|
31
|
+
import numpy as np
|
|
32
|
+
import pdfplumber
|
|
33
|
+
from PIL import Image
|
|
34
|
+
from sklearn.cluster import KMeans
|
|
35
|
+
from sklearn.metrics import silhouette_score
|
|
36
|
+
|
|
37
|
+
from .model_loader import default_model_dir
|
|
38
|
+
from .ocr import OCR
|
|
39
|
+
from .layout_recognizer import LayoutRecognizer4YOLOv10 as LayoutRecognizer
|
|
40
|
+
from .recognizer import Recognizer
|
|
41
|
+
from .table_structure_recognizer import TableStructureRecognizer
|
|
42
|
+
from .tokenizer import is_chinese
|
|
43
|
+
from .utils import extract_pdf_outlines
|
|
44
|
+
|
|
45
|
+
MAXIMUM_PAGE_NUMBER = 100000
|
|
46
|
+
#: Was torch.cuda.device_count() upstream (multi-GPU OCR); this port is
|
|
47
|
+
#: CPU-only, single-device, so parallel_limiter (see __init__) is always None
|
|
48
|
+
#: and this constant only matters to keep the still-present but now-dead
|
|
49
|
+
#: settings.PARALLEL_DEVICES reference in __images__ syntactically valid.
|
|
50
|
+
PARALLEL_DEVICES = 0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
#: NOTE: this is a simplified reimplementation, not a verbatim port, of
|
|
54
|
+
#: common/misc_utils.py's thread_pool_exec. Upstream uses a per-call
|
|
55
|
+
#: ThreadPoolExecutor(max_workers=1) instead of loop.run_in_executor(None,
|
|
56
|
+
#: call) specifically to avoid a documented Python 3.13 deadlock on repeated
|
|
57
|
+
#: awaits within one event loop. That distinction has no present effect here
|
|
58
|
+
#: -- this function's only call site is inside a branch that's permanently
|
|
59
|
+
#: unreachable, since self.parallel_limiter is always None after this port's
|
|
60
|
+
#: __init__ (see below) -- but reviving multi-device parallelism in the
|
|
61
|
+
#: future would need to restore the per-call executor to avoid that deadlock.
|
|
62
|
+
async def thread_pool_exec(func, *args, **kwargs):
|
|
63
|
+
import asyncio
|
|
64
|
+
import contextvars
|
|
65
|
+
import functools
|
|
66
|
+
|
|
67
|
+
loop = asyncio.get_running_loop()
|
|
68
|
+
ctx = contextvars.copy_context()
|
|
69
|
+
call = functools.partial(ctx.run, func, *args, **kwargs)
|
|
70
|
+
return await loop.run_in_executor(None, call)
|
|
71
|
+
|
|
72
|
+
LOCK_KEY_pdfplumber = "global_shared_lock_pdfplumber"
|
|
73
|
+
if LOCK_KEY_pdfplumber not in sys.modules:
|
|
74
|
+
sys.modules[LOCK_KEY_pdfplumber] = threading.Lock()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class RAGFlowPdfParser:
|
|
78
|
+
def __init__(self, model_dir=None, **kwargs):
|
|
79
|
+
# Resolved once so every model-file consumer -- OCR, LayoutRecognizer,
|
|
80
|
+
# TableStructureRecognizer, and _ocr_can_represent's ocr.res lookup --
|
|
81
|
+
# agrees on the same directory. OCR/Recognizer already fall back to
|
|
82
|
+
# default_model_dir() internally when given None, so passing the
|
|
83
|
+
# already-resolved value here changes nothing for them.
|
|
84
|
+
self.model_dir = model_dir or str(default_model_dir())
|
|
85
|
+
self.ocr = OCR(model_dir=self.model_dir)
|
|
86
|
+
self.parallel_limiter = None
|
|
87
|
+
|
|
88
|
+
self.layouter = LayoutRecognizer("layout", model_dir=self.model_dir)
|
|
89
|
+
self.tbl_det = TableStructureRecognizer(model_dir=self.model_dir)
|
|
90
|
+
|
|
91
|
+
self.page_from = 0
|
|
92
|
+
self.column_num = 1
|
|
93
|
+
|
|
94
|
+
def __char_width(self, c):
|
|
95
|
+
return (c["x1"] - c["x0"]) // max(len(c["text"]), 1)
|
|
96
|
+
|
|
97
|
+
def __height(self, c):
|
|
98
|
+
return c["bottom"] - c["top"]
|
|
99
|
+
|
|
100
|
+
def _x_dis(self, a, b):
|
|
101
|
+
return min(abs(a["x1"] - b["x0"]), abs(a["x0"] - b["x1"]), abs(a["x0"] + a["x1"] - b["x0"] - b["x1"]) / 2)
|
|
102
|
+
|
|
103
|
+
def _y_dis(self, a, b):
|
|
104
|
+
return (b["top"] + b["bottom"] - a["top"] - a["bottom"]) / 2
|
|
105
|
+
|
|
106
|
+
def _match_proj(self, b):
|
|
107
|
+
proj_patt = [
|
|
108
|
+
r"第[零一二三四五六七八九十百]+章",
|
|
109
|
+
r"第[零一二三四五六七八九十百]+[条节]",
|
|
110
|
+
r"[零一二三四五六七八九十百]+[、是 ]",
|
|
111
|
+
r"[\((][零一二三四五六七八九十百]+[)\)]",
|
|
112
|
+
r"[\((][0-9]+[)\)]",
|
|
113
|
+
r"[0-9]+(、|\.[ ]|)|\.[^0-9./a-zA-Z_%><-]{4,})",
|
|
114
|
+
r"[0-9]+\.[0-9.]+(、|\.[ ])",
|
|
115
|
+
r"[⚫•➢①② ]",
|
|
116
|
+
]
|
|
117
|
+
return any([re.match(p, b["text"]) for p in proj_patt])
|
|
118
|
+
|
|
119
|
+
@staticmethod
|
|
120
|
+
def sort_X_by_page(arr, threshold):
|
|
121
|
+
# sort using y1 first and then x1
|
|
122
|
+
arr = sorted(arr, key=lambda r: (r["page_number"], r["x0"], r["top"]))
|
|
123
|
+
for i in range(len(arr) - 1):
|
|
124
|
+
for j in range(i, -1, -1):
|
|
125
|
+
# restore the order using th
|
|
126
|
+
if abs(arr[j + 1]["x0"] - arr[j]["x0"]) < threshold and arr[j + 1]["top"] < arr[j]["top"] and arr[j + 1]["page_number"] == arr[j]["page_number"]:
|
|
127
|
+
tmp = arr[j]
|
|
128
|
+
arr[j] = arr[j + 1]
|
|
129
|
+
arr[j + 1] = tmp
|
|
130
|
+
return arr
|
|
131
|
+
|
|
132
|
+
def _has_color(self, o):
|
|
133
|
+
if o.get("ncs", "") == "DeviceGray":
|
|
134
|
+
if o["stroking_color"] and o["stroking_color"][0] == 1 and o["non_stroking_color"] and o["non_stroking_color"][0] == 1:
|
|
135
|
+
if re.match(r"[a-zT_\[\]\(\)-]+", o.get("text", "")):
|
|
136
|
+
return False
|
|
137
|
+
return True
|
|
138
|
+
|
|
139
|
+
# CID pattern regex for unmapped font characters from pdfminer
|
|
140
|
+
_CID_PATTERN = re.compile(r"\(cid\s*:\s*\d+\s*\)")
|
|
141
|
+
|
|
142
|
+
# Class-level default; a matching instance attribute (see __init__'s
|
|
143
|
+
# self.model_dir) shadows this once _ocr_can_represent below populates
|
|
144
|
+
# it, scoping the cache to the model_dir this instance was built with.
|
|
145
|
+
_OCR_ALPHABET = None
|
|
146
|
+
|
|
147
|
+
def _ocr_can_represent(self, text, min_coverage=0.8):
|
|
148
|
+
"""True if the OCR recogniser's alphabet covers this text well enough to be worth OCRing."""
|
|
149
|
+
if not text:
|
|
150
|
+
return True
|
|
151
|
+
if self._OCR_ALPHABET is None:
|
|
152
|
+
res = os.path.join(self.model_dir, "ocr.res")
|
|
153
|
+
try:
|
|
154
|
+
with open(res, encoding="utf-8") as f:
|
|
155
|
+
self._OCR_ALPHABET = set(f.read())
|
|
156
|
+
except (OSError, UnicodeDecodeError) as e:
|
|
157
|
+
logging.warning("Could not load OCR alphabet from %s: %s; treating all text as representable.", res, e)
|
|
158
|
+
self._OCR_ALPHABET = set()
|
|
159
|
+
if not self._OCR_ALPHABET:
|
|
160
|
+
return True # unknown alphabet: preserve existing behaviour
|
|
161
|
+
letters = [c for c in text if c.strip()]
|
|
162
|
+
if not letters:
|
|
163
|
+
return True
|
|
164
|
+
covered = sum(1 for c in letters if c in self._OCR_ALPHABET)
|
|
165
|
+
return covered / len(letters) >= min_coverage
|
|
166
|
+
|
|
167
|
+
# CJK scripts (Han, Hiragana, Katakana, Hangul) do not separate words with
|
|
168
|
+
# spaces, so a geometric gap between their glyphs must not become one.
|
|
169
|
+
_CJK_PATTERN = re.compile(r"[ᄀ-ᇿ-ヿ-㐀-䶿一-鿿가-豈-]|[\U00020000-\U0002fa1f]")
|
|
170
|
+
|
|
171
|
+
@classmethod
|
|
172
|
+
def _insert_word_spaces(cls, chars, gap_ratio=0.25):
|
|
173
|
+
"""Recover missing spaces from character geometry.
|
|
174
|
+
|
|
175
|
+
Many PDFs encode no space glyphs and separate words by positioning alone.
|
|
176
|
+
Append a space to a char when the gap to the next exceeds ``gap_ratio`` of
|
|
177
|
+
the mean char width; intra-word kerns fall well below that. CJK is skipped:
|
|
178
|
+
it does not write inter-word spaces, so a gap between CJK glyphs is ordinary
|
|
179
|
+
tracking, not a boundary. ``chars`` is a list of pdfplumber-style dicts and
|
|
180
|
+
is mutated in place.
|
|
181
|
+
"""
|
|
182
|
+
widths = [c["width"] for c in chars if c["text"] and c["text"].strip()]
|
|
183
|
+
mean_w = sum(widths) / len(widths) if widths else 0
|
|
184
|
+
if mean_w <= 0:
|
|
185
|
+
return
|
|
186
|
+
for cur, nxt in zip(chars, chars[1:]):
|
|
187
|
+
if (
|
|
188
|
+
cur["text"]
|
|
189
|
+
and nxt["text"]
|
|
190
|
+
and cur["text"].strip()
|
|
191
|
+
and nxt["text"].strip()
|
|
192
|
+
and not cls._CJK_PATTERN.search(cur["text"])
|
|
193
|
+
and not cls._CJK_PATTERN.search(nxt["text"])
|
|
194
|
+
and nxt["x0"] - cur["x1"] > mean_w * gap_ratio
|
|
195
|
+
):
|
|
196
|
+
cur["text"] += " "
|
|
197
|
+
|
|
198
|
+
@staticmethod
|
|
199
|
+
def _is_garbled_char(ch):
|
|
200
|
+
"""Check if a single character is garbled (unmappable from PDF font encoding).
|
|
201
|
+
|
|
202
|
+
A character is considered garbled if it falls into Unicode Private Use Areas
|
|
203
|
+
or certain replacement/control character ranges that typically indicate
|
|
204
|
+
pdfminer failed to map a CID to a valid Unicode codepoint.
|
|
205
|
+
"""
|
|
206
|
+
if not ch:
|
|
207
|
+
return False
|
|
208
|
+
cp = ord(ch)
|
|
209
|
+
if 0xE000 <= cp <= 0xF8FF:
|
|
210
|
+
return True
|
|
211
|
+
if 0xF0000 <= cp <= 0xFFFFF:
|
|
212
|
+
return True
|
|
213
|
+
if 0x100000 <= cp <= 0x10FFFF:
|
|
214
|
+
return True
|
|
215
|
+
if cp == 0xFFFD:
|
|
216
|
+
return True
|
|
217
|
+
if cp < 0x20 and ch not in ("\t", "\n", "\r"):
|
|
218
|
+
return True
|
|
219
|
+
if 0x80 <= cp <= 0x9F:
|
|
220
|
+
return True
|
|
221
|
+
cat = unicodedata.category(ch)
|
|
222
|
+
if cat in ("Cn", "Cs"):
|
|
223
|
+
return True
|
|
224
|
+
return False
|
|
225
|
+
|
|
226
|
+
@staticmethod
|
|
227
|
+
def _is_garbled_text(text, threshold=0.5):
|
|
228
|
+
"""Check if a text string contains too many garbled characters.
|
|
229
|
+
|
|
230
|
+
Examines each character and determines if the overall proportion
|
|
231
|
+
of garbled characters exceeds the given threshold. Also detects
|
|
232
|
+
pdfminer's CID placeholder patterns like '(cid:123)'.
|
|
233
|
+
"""
|
|
234
|
+
if not text or not text.strip():
|
|
235
|
+
return False
|
|
236
|
+
if RAGFlowPdfParser._CID_PATTERN.search(text):
|
|
237
|
+
return True
|
|
238
|
+
garbled_count = 0
|
|
239
|
+
total = 0
|
|
240
|
+
for ch in text:
|
|
241
|
+
if ch.isspace():
|
|
242
|
+
continue
|
|
243
|
+
total += 1
|
|
244
|
+
if RAGFlowPdfParser._is_garbled_char(ch):
|
|
245
|
+
garbled_count += 1
|
|
246
|
+
if total == 0:
|
|
247
|
+
return False
|
|
248
|
+
return garbled_count / total >= threshold
|
|
249
|
+
|
|
250
|
+
@staticmethod
|
|
251
|
+
def _has_subset_font_prefix(fontname):
|
|
252
|
+
"""Check if a font name has a subset prefix (e.g. 'DY1+ZLQDm1-1').
|
|
253
|
+
|
|
254
|
+
PDF subset fonts use a 6-letter uppercase tag followed by '+' before
|
|
255
|
+
the actual font name. Some tools use shorter tags (e.g. 'DY1+').
|
|
256
|
+
"""
|
|
257
|
+
if not fontname:
|
|
258
|
+
return False
|
|
259
|
+
return bool(re.match(r"^[A-Z0-9]{2,6}\+", fontname))
|
|
260
|
+
|
|
261
|
+
@staticmethod
|
|
262
|
+
def _is_garbled_by_font_encoding(page_chars, min_chars=20):
|
|
263
|
+
"""Detect garbled text caused by broken font encoding mappings.
|
|
264
|
+
|
|
265
|
+
Some PDFs (especially older Chinese standards) embed custom fonts that
|
|
266
|
+
map CJK glyphs to ASCII codepoints. The extracted text appears as
|
|
267
|
+
random ASCII punctuation/symbols instead of actual CJK characters.
|
|
268
|
+
|
|
269
|
+
Detection strategy: if a significant proportion of characters come from
|
|
270
|
+
subset-embedded fonts and the page produces overwhelmingly ASCII
|
|
271
|
+
(punctuation, digits, symbols) with virtually no CJK/Hangul/Kana
|
|
272
|
+
characters, the page is likely garbled due to broken font encoding.
|
|
273
|
+
"""
|
|
274
|
+
if not page_chars or len(page_chars) < min_chars:
|
|
275
|
+
return False
|
|
276
|
+
|
|
277
|
+
subset_font_count = 0
|
|
278
|
+
total_non_space = 0
|
|
279
|
+
ascii_punct_sym = 0
|
|
280
|
+
cjk_like = 0
|
|
281
|
+
|
|
282
|
+
for c in page_chars:
|
|
283
|
+
text = c.get("text", "")
|
|
284
|
+
fontname = c.get("fontname", "")
|
|
285
|
+
if not text or text.isspace():
|
|
286
|
+
continue
|
|
287
|
+
total_non_space += 1
|
|
288
|
+
|
|
289
|
+
if RAGFlowPdfParser._has_subset_font_prefix(fontname):
|
|
290
|
+
subset_font_count += 1
|
|
291
|
+
|
|
292
|
+
cp = ord(text[0])
|
|
293
|
+
if 0x2E80 <= cp <= 0x9FFF or 0xF900 <= cp <= 0xFAFF or 0x20000 <= cp <= 0x2FA1F or 0xAC00 <= cp <= 0xD7AF or 0x3040 <= cp <= 0x30FF:
|
|
294
|
+
cjk_like += 1
|
|
295
|
+
elif 0x21 <= cp <= 0x2F or 0x3A <= cp <= 0x40 or 0x5B <= cp <= 0x60 or 0x7B <= cp <= 0x7E:
|
|
296
|
+
ascii_punct_sym += 1
|
|
297
|
+
|
|
298
|
+
if total_non_space < min_chars:
|
|
299
|
+
return False
|
|
300
|
+
|
|
301
|
+
subset_ratio = subset_font_count / total_non_space
|
|
302
|
+
if subset_ratio < 0.3:
|
|
303
|
+
return False
|
|
304
|
+
|
|
305
|
+
cjk_ratio = cjk_like / total_non_space
|
|
306
|
+
punct_ratio = ascii_punct_sym / total_non_space
|
|
307
|
+
if cjk_ratio < 0.05 and punct_ratio > 0.4:
|
|
308
|
+
return True
|
|
309
|
+
|
|
310
|
+
return False
|
|
311
|
+
|
|
312
|
+
def _evaluate_table_orientation(self, table_img, sample_ratio=0.3):
|
|
313
|
+
"""
|
|
314
|
+
Evaluate the best rotation orientation for a table image.
|
|
315
|
+
|
|
316
|
+
Tests 4 rotation angles (0°, 90°, 180°, 270°) and uses OCR
|
|
317
|
+
confidence scores to determine the best orientation.
|
|
318
|
+
|
|
319
|
+
Args:
|
|
320
|
+
table_img: PIL Image object of the table region
|
|
321
|
+
sample_ratio: Sampling ratio for quick evaluation
|
|
322
|
+
|
|
323
|
+
Returns:
|
|
324
|
+
tuple: (best_angle, best_img, confidence_scores)
|
|
325
|
+
- best_angle: Best rotation angle (0, 90, 180, 270)
|
|
326
|
+
- best_img: Image rotated to best orientation
|
|
327
|
+
- confidence_scores: Dict of scores for each angle
|
|
328
|
+
"""
|
|
329
|
+
|
|
330
|
+
rotations = [
|
|
331
|
+
(0, "original"),
|
|
332
|
+
(90, "rotate_90"), # clockwise 90°
|
|
333
|
+
(180, "rotate_180"), # 180°
|
|
334
|
+
(270, "rotate_270"), # clockwise 270° (counter-clockwise 90°)
|
|
335
|
+
]
|
|
336
|
+
|
|
337
|
+
results = {}
|
|
338
|
+
best_score = -1
|
|
339
|
+
best_angle = 0
|
|
340
|
+
best_img = table_img
|
|
341
|
+
score_0 = None
|
|
342
|
+
|
|
343
|
+
for angle, name in rotations:
|
|
344
|
+
# Rotate image
|
|
345
|
+
if angle == 0:
|
|
346
|
+
rotated_img = table_img
|
|
347
|
+
else:
|
|
348
|
+
# PIL's rotate is counter-clockwise, use negative angle for clockwise
|
|
349
|
+
rotated_img = table_img.rotate(-angle, expand=True)
|
|
350
|
+
|
|
351
|
+
# Convert to numpy array for OCR
|
|
352
|
+
img_array = np.array(rotated_img)
|
|
353
|
+
|
|
354
|
+
# Perform OCR detection and recognition
|
|
355
|
+
try:
|
|
356
|
+
ocr_results = self.ocr(img_array)
|
|
357
|
+
|
|
358
|
+
if ocr_results:
|
|
359
|
+
# Calculate average confidence
|
|
360
|
+
scores = [conf for _, (_, conf) in ocr_results]
|
|
361
|
+
avg_score = sum(scores) / len(scores) if scores else 0
|
|
362
|
+
total_regions = len(scores)
|
|
363
|
+
|
|
364
|
+
# Combined score: considers both average confidence and number of regions
|
|
365
|
+
# More regions + higher confidence = better orientation
|
|
366
|
+
combined_score = avg_score * (1 + 0.1 * min(total_regions, 50) / 50)
|
|
367
|
+
else:
|
|
368
|
+
avg_score = 0
|
|
369
|
+
total_regions = 0
|
|
370
|
+
combined_score = 0
|
|
371
|
+
|
|
372
|
+
except Exception as e:
|
|
373
|
+
logging.warning(f"OCR failed for angle {angle}: {e}")
|
|
374
|
+
avg_score = 0
|
|
375
|
+
total_regions = 0
|
|
376
|
+
combined_score = 0
|
|
377
|
+
|
|
378
|
+
results[angle] = {"avg_confidence": avg_score, "total_regions": total_regions, "combined_score": combined_score}
|
|
379
|
+
if angle == 0:
|
|
380
|
+
score_0 = combined_score
|
|
381
|
+
|
|
382
|
+
logging.debug(f"Table orientation {angle}°: avg_conf={avg_score:.4f}, regions={total_regions}, combined={combined_score:.4f}")
|
|
383
|
+
|
|
384
|
+
if combined_score > best_score:
|
|
385
|
+
best_score = combined_score
|
|
386
|
+
best_angle = angle
|
|
387
|
+
best_img = rotated_img
|
|
388
|
+
|
|
389
|
+
# Absolute threshold rule:
|
|
390
|
+
# Only choose non-0° if it exceeds 0° by more than 0.2 and 0° score is below 0.8.
|
|
391
|
+
if best_angle != 0 and score_0 is not None:
|
|
392
|
+
if not (best_score - score_0 > 0.2 and score_0 < 0.8):
|
|
393
|
+
best_angle = 0
|
|
394
|
+
best_img = table_img
|
|
395
|
+
best_score = score_0
|
|
396
|
+
|
|
397
|
+
results[best_angle] = results.get(best_angle, {"avg_confidence": 0, "total_regions": 0, "combined_score": 0})
|
|
398
|
+
|
|
399
|
+
logging.info(f"Best table orientation: {best_angle}° (score={best_score:.4f})")
|
|
400
|
+
|
|
401
|
+
return best_angle, best_img, results
|
|
402
|
+
|
|
403
|
+
@staticmethod
|
|
404
|
+
def _map_clockwise_rotated_point_to_original(x, y, angle, width, height):
|
|
405
|
+
if angle == 0:
|
|
406
|
+
return x, y
|
|
407
|
+
if angle == 90:
|
|
408
|
+
return y, height - x
|
|
409
|
+
if angle == 180:
|
|
410
|
+
return width - x, height - y
|
|
411
|
+
if angle == 270:
|
|
412
|
+
return width - y, x
|
|
413
|
+
return x, y
|
|
414
|
+
|
|
415
|
+
def _table_transformer_job(self, ZM, auto_rotate=True):
|
|
416
|
+
"""
|
|
417
|
+
Process table structure recognition.
|
|
418
|
+
|
|
419
|
+
When auto_rotate=True, the complete workflow:
|
|
420
|
+
1. Evaluate table orientation and select the best rotation angle
|
|
421
|
+
2. Use rotated image for table structure recognition (TSR)
|
|
422
|
+
3. Re-OCR the rotated image
|
|
423
|
+
4. Match new OCR results with TSR cell coordinates
|
|
424
|
+
|
|
425
|
+
Args:
|
|
426
|
+
ZM: Zoom factor
|
|
427
|
+
auto_rotate: Whether to enable auto orientation correction
|
|
428
|
+
"""
|
|
429
|
+
logging.debug("Table processing...")
|
|
430
|
+
imgs, pos = [], []
|
|
431
|
+
tbcnt = [0]
|
|
432
|
+
MARGIN = 10
|
|
433
|
+
self.tb_cpns = []
|
|
434
|
+
self.table_rotations = {} # Store rotation info for each table
|
|
435
|
+
self.rotated_table_imgs = {} # Store rotated table images
|
|
436
|
+
|
|
437
|
+
assert len(self.page_layout) == len(self.page_images)
|
|
438
|
+
|
|
439
|
+
# Collect layout info for all tables
|
|
440
|
+
table_layouts = []
|
|
441
|
+
|
|
442
|
+
table_index = 0
|
|
443
|
+
for p, tbls in enumerate(self.page_layout): # for page
|
|
444
|
+
tbls = [f for f in tbls if f["type"] == "table"]
|
|
445
|
+
tbcnt.append(len(tbls))
|
|
446
|
+
if not tbls:
|
|
447
|
+
continue
|
|
448
|
+
for page_table_index, tb in enumerate(tbls): # for table
|
|
449
|
+
left, top, right, bott = tb["x0"] - MARGIN, tb["top"] - MARGIN, tb["x1"] + MARGIN, tb["bottom"] + MARGIN
|
|
450
|
+
left *= ZM
|
|
451
|
+
top *= ZM
|
|
452
|
+
right *= ZM
|
|
453
|
+
bott *= ZM
|
|
454
|
+
layoutno = f"table-{page_table_index}"
|
|
455
|
+
pos.append((left, top, p, table_index, layoutno))
|
|
456
|
+
|
|
457
|
+
# Record table layout info
|
|
458
|
+
table_layouts.append({"page": p, "table_index": table_index, "layoutno": layoutno, "layout": tb, "coords": (left, top, right, bott)})
|
|
459
|
+
|
|
460
|
+
# Crop table image
|
|
461
|
+
table_img = self.page_images[p].crop((left, top, right, bott))
|
|
462
|
+
|
|
463
|
+
if auto_rotate:
|
|
464
|
+
# Evaluate table orientation
|
|
465
|
+
logging.debug(f"Evaluating orientation for table {table_index} on page {p}")
|
|
466
|
+
best_angle, rotated_img, rotation_scores = self._evaluate_table_orientation(table_img)
|
|
467
|
+
|
|
468
|
+
# Store rotation info
|
|
469
|
+
self.table_rotations[table_index] = {
|
|
470
|
+
"page": p,
|
|
471
|
+
"original_pos": (left, top, right, bott),
|
|
472
|
+
"best_angle": best_angle,
|
|
473
|
+
"scores": rotation_scores,
|
|
474
|
+
"rotated_size": rotated_img.size, # (width, height)
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
# Store the rotated image
|
|
478
|
+
self.rotated_table_imgs[table_index] = rotated_img
|
|
479
|
+
imgs.append(rotated_img)
|
|
480
|
+
|
|
481
|
+
else:
|
|
482
|
+
imgs.append(table_img)
|
|
483
|
+
self.table_rotations[table_index] = {"page": p, "original_pos": (left, top, right, bott), "best_angle": 0, "scores": {}, "rotated_size": table_img.size}
|
|
484
|
+
self.rotated_table_imgs[table_index] = table_img
|
|
485
|
+
|
|
486
|
+
table_index += 1
|
|
487
|
+
|
|
488
|
+
assert len(self.page_images) == len(tbcnt) - 1
|
|
489
|
+
if not imgs:
|
|
490
|
+
return
|
|
491
|
+
|
|
492
|
+
# Perform table structure recognition (TSR)
|
|
493
|
+
recos = self.tbl_det(imgs)
|
|
494
|
+
|
|
495
|
+
# If tables were rotated, re-OCR the rotated images and replace table boxes
|
|
496
|
+
if auto_rotate:
|
|
497
|
+
self._ocr_rotated_tables(ZM, table_layouts, recos, tbcnt)
|
|
498
|
+
|
|
499
|
+
def _map_tsr_component_to_page_space(component, table_pos):
|
|
500
|
+
crop_left, crop_top, page, table_index, _ = table_pos
|
|
501
|
+
rotation_info = self.table_rotations.get(table_index, {})
|
|
502
|
+
angle = rotation_info.get("best_angle", 0)
|
|
503
|
+
original_pos = rotation_info.get("original_pos", (crop_left, crop_top, crop_left, crop_top))
|
|
504
|
+
width = original_pos[2] - original_pos[0]
|
|
505
|
+
height = original_pos[3] - original_pos[1]
|
|
506
|
+
points = [
|
|
507
|
+
(component["x0_rotated"], component["top_rotated"]),
|
|
508
|
+
(component["x1_rotated"], component["top_rotated"]),
|
|
509
|
+
(component["x0_rotated"], component["bottom_rotated"]),
|
|
510
|
+
(component["x1_rotated"], component["bottom_rotated"]),
|
|
511
|
+
]
|
|
512
|
+
mapped = [self._map_clockwise_rotated_point_to_original(x, y, angle, width, height) for x, y in points]
|
|
513
|
+
xs = [p[0] for p in mapped]
|
|
514
|
+
ys = [p[1] for p in mapped]
|
|
515
|
+
component["x0"] = min(xs) / ZM + crop_left / ZM
|
|
516
|
+
component["x1"] = max(xs) / ZM + crop_left / ZM
|
|
517
|
+
component["top"] = min(ys) / ZM + crop_top / ZM + self.page_cum_height[page]
|
|
518
|
+
component["bottom"] = max(ys) / ZM + crop_top / ZM + self.page_cum_height[page]
|
|
519
|
+
|
|
520
|
+
# Process TSR results and align structure boxes with page-cumulative OCR boxes.
|
|
521
|
+
tbcnt = np.cumsum(tbcnt)
|
|
522
|
+
for i in range(len(tbcnt) - 1): # for page
|
|
523
|
+
pg = []
|
|
524
|
+
for j, tb_items in enumerate(recos[tbcnt[i] : tbcnt[i + 1]]): # for table
|
|
525
|
+
poss = pos[tbcnt[i] : tbcnt[i + 1]]
|
|
526
|
+
for it in tb_items: # for table components
|
|
527
|
+
# TSR coordinates are relative to rotated image, need to record
|
|
528
|
+
it["x0_rotated"] = it["x0"]
|
|
529
|
+
it["x1_rotated"] = it["x1"]
|
|
530
|
+
it["top_rotated"] = it["top"]
|
|
531
|
+
it["bottom_rotated"] = it["bottom"]
|
|
532
|
+
|
|
533
|
+
it["pn"] = poss[j][2] # page number
|
|
534
|
+
it["layoutno"] = poss[j][4]
|
|
535
|
+
it["table_index"] = poss[j][3] # table index
|
|
536
|
+
_map_tsr_component_to_page_space(it, poss[j])
|
|
537
|
+
pg.append(it)
|
|
538
|
+
self.tb_cpns.extend(pg)
|
|
539
|
+
|
|
540
|
+
def gather(kwd, fzy=10, ption=0.6):
|
|
541
|
+
eles = Recognizer.sort_Y_firstly([r for r in self.tb_cpns if re.match(kwd, r["label"])], fzy)
|
|
542
|
+
eles = Recognizer.layouts_cleanup(self.boxes, eles, 5, ption)
|
|
543
|
+
return Recognizer.sort_Y_firstly(eles, 0)
|
|
544
|
+
|
|
545
|
+
# add R,H,C,SP tag to boxes within table layout
|
|
546
|
+
headers = gather(r".*header$")
|
|
547
|
+
rows = gather(r".* (row|header)")
|
|
548
|
+
spans = gather(r".*spanning")
|
|
549
|
+
clmns = sorted([r for r in self.tb_cpns if re.match(r"table column$", r["label"])], key=lambda x: (x["pn"], x["layoutno"], x["x0"]))
|
|
550
|
+
clmns = Recognizer.layouts_cleanup(self.boxes, clmns, 5, 0.5)
|
|
551
|
+
|
|
552
|
+
for b in self.boxes:
|
|
553
|
+
if b.get("layout_type", "") != "table":
|
|
554
|
+
continue
|
|
555
|
+
ii = Recognizer.find_overlapped_with_threshold(b, rows, thr=0.3)
|
|
556
|
+
if ii is not None:
|
|
557
|
+
b["R"] = ii
|
|
558
|
+
b["R_top"] = rows[ii]["top"]
|
|
559
|
+
b["R_bott"] = rows[ii]["bottom"]
|
|
560
|
+
|
|
561
|
+
ii = Recognizer.find_overlapped_with_threshold(b, headers, thr=0.3)
|
|
562
|
+
if ii is not None:
|
|
563
|
+
b["H_top"] = headers[ii]["top"]
|
|
564
|
+
b["H_bott"] = headers[ii]["bottom"]
|
|
565
|
+
b["H_left"] = headers[ii]["x0"]
|
|
566
|
+
b["H_right"] = headers[ii]["x1"]
|
|
567
|
+
b["H"] = ii
|
|
568
|
+
|
|
569
|
+
ii = Recognizer.find_horizontally_tightest_fit(b, clmns)
|
|
570
|
+
if ii is not None:
|
|
571
|
+
b["C"] = ii
|
|
572
|
+
b["C_left"] = clmns[ii]["x0"]
|
|
573
|
+
b["C_right"] = clmns[ii]["x1"]
|
|
574
|
+
|
|
575
|
+
ii = Recognizer.find_overlapped_with_threshold(b, spans, thr=0.3)
|
|
576
|
+
if ii is not None:
|
|
577
|
+
b["H_top"] = spans[ii]["top"]
|
|
578
|
+
b["H_bott"] = spans[ii]["bottom"]
|
|
579
|
+
b["H_left"] = spans[ii]["x0"]
|
|
580
|
+
b["H_right"] = spans[ii]["x1"]
|
|
581
|
+
b["SP"] = ii
|
|
582
|
+
|
|
583
|
+
def _ocr_rotated_tables(self, ZM, table_layouts, tsr_results, tbcnt):
|
|
584
|
+
"""
|
|
585
|
+
Re-OCR rotated table images and update self.boxes.
|
|
586
|
+
|
|
587
|
+
Args:
|
|
588
|
+
ZM: Zoom factor
|
|
589
|
+
table_layouts: List of table layout info
|
|
590
|
+
tsr_results: TSR recognition results
|
|
591
|
+
tbcnt: Cumulative table count per page
|
|
592
|
+
"""
|
|
593
|
+
tbcnt = np.cumsum(tbcnt)
|
|
594
|
+
|
|
595
|
+
def _table_region(layout, page_index):
|
|
596
|
+
table_x0 = layout["x0"]
|
|
597
|
+
table_top = layout["top"]
|
|
598
|
+
table_x1 = layout["x1"]
|
|
599
|
+
table_bottom = layout["bottom"]
|
|
600
|
+
table_top_cum = table_top + self.page_cum_height[page_index]
|
|
601
|
+
table_bottom_cum = table_bottom + self.page_cum_height[page_index]
|
|
602
|
+
return table_x0, table_top, table_x1, table_bottom, table_top_cum, table_bottom_cum
|
|
603
|
+
|
|
604
|
+
def _collect_table_boxes(page_index, table_x0, table_x1, table_top_cum, table_bottom_cum):
|
|
605
|
+
indices = [
|
|
606
|
+
i
|
|
607
|
+
for i, b in enumerate(self.boxes)
|
|
608
|
+
if (
|
|
609
|
+
b.get("page_number") == page_index + self.page_from
|
|
610
|
+
and b.get("layout_type") == "table"
|
|
611
|
+
and b["x0"] >= table_x0 - 5
|
|
612
|
+
and b["x1"] <= table_x1 + 5
|
|
613
|
+
and b["top"] >= table_top_cum - 5
|
|
614
|
+
and b["bottom"] <= table_bottom_cum + 5
|
|
615
|
+
)
|
|
616
|
+
]
|
|
617
|
+
original_boxes = [self.boxes[i] for i in indices]
|
|
618
|
+
insert_at = indices[0] if indices else len(self.boxes)
|
|
619
|
+
for i in reversed(indices):
|
|
620
|
+
self.boxes.pop(i)
|
|
621
|
+
return original_boxes, insert_at
|
|
622
|
+
|
|
623
|
+
def _restore_boxes(original_boxes, insert_at):
|
|
624
|
+
for b in original_boxes:
|
|
625
|
+
self.boxes.insert(insert_at, b)
|
|
626
|
+
insert_at += 1
|
|
627
|
+
return insert_at
|
|
628
|
+
|
|
629
|
+
def _insert_ocr_boxes(ocr_results, page_index, crop_left, crop_top, insert_at, table_index, layoutno, best_angle, table_w_px, table_h_px):
|
|
630
|
+
added = 0
|
|
631
|
+
for bbox, (text, conf) in ocr_results:
|
|
632
|
+
if conf < 0.5:
|
|
633
|
+
continue
|
|
634
|
+
mapped = [self._map_clockwise_rotated_point_to_original(p[0], p[1], best_angle, table_w_px, table_h_px) for p in bbox]
|
|
635
|
+
x_coords = [p[0] for p in mapped]
|
|
636
|
+
y_coords = [p[1] for p in mapped]
|
|
637
|
+
box_x0 = min(x_coords) / ZM
|
|
638
|
+
box_x1 = max(x_coords) / ZM
|
|
639
|
+
box_top = min(y_coords) / ZM
|
|
640
|
+
box_bottom = max(y_coords) / ZM
|
|
641
|
+
new_box = {
|
|
642
|
+
"text": text,
|
|
643
|
+
"x0": box_x0 + crop_left / ZM,
|
|
644
|
+
"x1": box_x1 + crop_left / ZM,
|
|
645
|
+
"top": box_top + crop_top / ZM + self.page_cum_height[page_index],
|
|
646
|
+
"bottom": box_bottom + crop_top / ZM + self.page_cum_height[page_index],
|
|
647
|
+
"page_number": page_index + self.page_from,
|
|
648
|
+
"layout_type": "table",
|
|
649
|
+
"layoutno": layoutno,
|
|
650
|
+
"_rotated": True,
|
|
651
|
+
"_rotation_angle": best_angle,
|
|
652
|
+
"_table_index": table_index,
|
|
653
|
+
"_rotated_x0": box_x0,
|
|
654
|
+
"_rotated_x1": box_x1,
|
|
655
|
+
"_rotated_top": box_top,
|
|
656
|
+
"_rotated_bottom": box_bottom,
|
|
657
|
+
}
|
|
658
|
+
self.boxes.insert(insert_at, new_box)
|
|
659
|
+
insert_at += 1
|
|
660
|
+
added += 1
|
|
661
|
+
return added
|
|
662
|
+
|
|
663
|
+
for tbl_info in table_layouts:
|
|
664
|
+
table_index = tbl_info["table_index"]
|
|
665
|
+
page = tbl_info["page"]
|
|
666
|
+
layout = tbl_info["layout"]
|
|
667
|
+
layoutno = tbl_info["layoutno"]
|
|
668
|
+
left, top, right, bott = tbl_info["coords"]
|
|
669
|
+
|
|
670
|
+
rotation_info = self.table_rotations.get(table_index, {})
|
|
671
|
+
best_angle = rotation_info.get("best_angle", 0)
|
|
672
|
+
|
|
673
|
+
# Get the rotated table image
|
|
674
|
+
rotated_img = self.rotated_table_imgs.get(table_index)
|
|
675
|
+
if rotated_img is None:
|
|
676
|
+
continue
|
|
677
|
+
|
|
678
|
+
# If no rotation, keep original OCR boxes untouched.
|
|
679
|
+
if best_angle == 0:
|
|
680
|
+
continue
|
|
681
|
+
|
|
682
|
+
# Table region is defined by layout's x0, top, x1, bottom (page-local coords)
|
|
683
|
+
table_x0, table_top, table_x1, table_bottom, table_top_cum, table_bottom_cum = _table_region(layout, page)
|
|
684
|
+
original_boxes, insert_at = _collect_table_boxes(page, table_x0, table_x1, table_top_cum, table_bottom_cum)
|
|
685
|
+
|
|
686
|
+
logging.info(f"Re-OCR table {table_index} on page {page} with rotation {best_angle}°")
|
|
687
|
+
|
|
688
|
+
# Perform OCR on rotated image
|
|
689
|
+
img_array = np.array(rotated_img)
|
|
690
|
+
ocr_results = self.ocr(img_array)
|
|
691
|
+
|
|
692
|
+
if not ocr_results:
|
|
693
|
+
logging.warning(f"No OCR results for rotated table {table_index}, restoring originals")
|
|
694
|
+
_restore_boxes(original_boxes, insert_at)
|
|
695
|
+
continue
|
|
696
|
+
|
|
697
|
+
# Add new OCR results to self.boxes
|
|
698
|
+
# OCR coordinates are relative to rotated image, map back to original table coords
|
|
699
|
+
table_w_px = right - left
|
|
700
|
+
table_h_px = bott - top
|
|
701
|
+
added = _insert_ocr_boxes(
|
|
702
|
+
ocr_results,
|
|
703
|
+
page,
|
|
704
|
+
left,
|
|
705
|
+
top,
|
|
706
|
+
insert_at,
|
|
707
|
+
table_index,
|
|
708
|
+
layoutno,
|
|
709
|
+
best_angle,
|
|
710
|
+
table_w_px,
|
|
711
|
+
table_h_px,
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
logging.info(f"Added {added} OCR results from rotated table {table_index}")
|
|
715
|
+
|
|
716
|
+
def __ocr(self, pagenum, img, chars, ZM=3, device_id: int | None = None):
|
|
717
|
+
# start = timer()
|
|
718
|
+
bxs = self.ocr.detect(np.array(img), device_id)
|
|
719
|
+
# logging.info(f"__ocr detecting boxes of an image cost ({timer() - start}s)")
|
|
720
|
+
|
|
721
|
+
# start = timer()
|
|
722
|
+
if not bxs:
|
|
723
|
+
self.boxes.append([])
|
|
724
|
+
return
|
|
725
|
+
bxs = [(line[0], line[1][0]) for line in bxs]
|
|
726
|
+
bxs = Recognizer.sort_Y_firstly(
|
|
727
|
+
[
|
|
728
|
+
{"x0": b[0][0] / ZM, "x1": b[1][0] / ZM, "top": b[0][1] / ZM, "text": "", "txt": t, "bottom": b[-1][1] / ZM, "chars": [], "page_number": pagenum}
|
|
729
|
+
for b, t in bxs
|
|
730
|
+
if b[0][0] <= b[1][0] and b[0][1] <= b[-1][1]
|
|
731
|
+
],
|
|
732
|
+
self.mean_height[pagenum - 1] / 3,
|
|
733
|
+
)
|
|
734
|
+
|
|
735
|
+
# merge chars in the same rect
|
|
736
|
+
for c in chars:
|
|
737
|
+
ii = Recognizer.find_overlapped(c, bxs)
|
|
738
|
+
if ii is None:
|
|
739
|
+
self.lefted_chars.append(c)
|
|
740
|
+
continue
|
|
741
|
+
ch = c["bottom"] - c["top"]
|
|
742
|
+
bh = bxs[ii]["bottom"] - bxs[ii]["top"]
|
|
743
|
+
if abs(ch - bh) / max(ch, bh) >= 0.7 and c["text"] != " ":
|
|
744
|
+
self.lefted_chars.append(c)
|
|
745
|
+
continue
|
|
746
|
+
bxs[ii]["chars"].append(c)
|
|
747
|
+
|
|
748
|
+
for b in bxs:
|
|
749
|
+
if not b["chars"]:
|
|
750
|
+
del b["chars"]
|
|
751
|
+
continue
|
|
752
|
+
box_chars = b["chars"]
|
|
753
|
+
m_ht = np.mean([c["height"] for c in box_chars])
|
|
754
|
+
garbled_count = 0
|
|
755
|
+
total_count = 0
|
|
756
|
+
for c in Recognizer.sort_Y_firstly(box_chars, m_ht):
|
|
757
|
+
if c["text"] == " " and b["text"]:
|
|
758
|
+
if re.match(r"[0-9a-zA-Zа-яА-Я,.?;:!%%]", b["text"][-1]):
|
|
759
|
+
b["text"] += " "
|
|
760
|
+
else:
|
|
761
|
+
b["text"] += c["text"]
|
|
762
|
+
for ch in c["text"]:
|
|
763
|
+
if not ch.isspace():
|
|
764
|
+
total_count += 1
|
|
765
|
+
if self._is_garbled_char(ch):
|
|
766
|
+
garbled_count += 1
|
|
767
|
+
del b["chars"]
|
|
768
|
+
|
|
769
|
+
# Strategy 1: PUA / unmapped CID characters. These are genuine garbage,
|
|
770
|
+
# so re-OCR regardless of script.
|
|
771
|
+
if total_count > 0 and garbled_count / total_count >= 0.5:
|
|
772
|
+
logging.info(
|
|
773
|
+
"Page %d: detected garbled pdfplumber text (garbled=%d/%d), falling back to OCR for box at (%.1f, %.1f)",
|
|
774
|
+
pagenum,
|
|
775
|
+
garbled_count,
|
|
776
|
+
total_count,
|
|
777
|
+
b["x0"],
|
|
778
|
+
b["top"],
|
|
779
|
+
)
|
|
780
|
+
b["text"] = ""
|
|
781
|
+
continue
|
|
782
|
+
|
|
783
|
+
# Keep a clean text layer the recogniser cannot spell: ocr.res is
|
|
784
|
+
# CJK+Latin, so re-OCRing e.g. a Cyrillic page only produces garbage.
|
|
785
|
+
if total_count > 0 and not self._ocr_can_represent(b["text"]):
|
|
786
|
+
continue
|
|
787
|
+
|
|
788
|
+
# Strategy 2: font-encoding garbling — all chars are ASCII
|
|
789
|
+
# punctuation from subset fonts (no CJK output)
|
|
790
|
+
if total_count > 0 and self._is_garbled_by_font_encoding(box_chars, min_chars=5):
|
|
791
|
+
logging.info(
|
|
792
|
+
"Page %d: detected font-encoding garbled text (%d chars), falling back to OCR for box at (%.1f, %.1f)",
|
|
793
|
+
pagenum,
|
|
794
|
+
total_count,
|
|
795
|
+
b["x0"],
|
|
796
|
+
b["top"],
|
|
797
|
+
)
|
|
798
|
+
b["text"] = ""
|
|
799
|
+
|
|
800
|
+
# logging.info(f"__ocr sorting {len(chars)} chars cost {timer() - start}s")
|
|
801
|
+
# start = timer()
|
|
802
|
+
boxes_to_reg = []
|
|
803
|
+
img_np = None
|
|
804
|
+
for b in bxs:
|
|
805
|
+
if not b["text"]:
|
|
806
|
+
if img_np is None:
|
|
807
|
+
img_np = np.asarray(img)
|
|
808
|
+
left, right, top, bott = b["x0"] * ZM, b["x1"] * ZM, b["top"] * ZM, b["bottom"] * ZM
|
|
809
|
+
b["box_image"] = self.ocr.get_rotate_crop_image(img_np, np.array([[left, top], [right, top], [right, bott], [left, bott]], dtype=np.float32))
|
|
810
|
+
boxes_to_reg.append(b)
|
|
811
|
+
del b["txt"]
|
|
812
|
+
texts = self.ocr.recognize_batch([b["box_image"] for b in boxes_to_reg], device_id)
|
|
813
|
+
for i in range(len(boxes_to_reg)):
|
|
814
|
+
boxes_to_reg[i]["text"] = texts[i]
|
|
815
|
+
del boxes_to_reg[i]["box_image"]
|
|
816
|
+
# logging.info(f"__ocr recognize {len(bxs)} boxes cost {timer() - start}s")
|
|
817
|
+
bxs = [b for b in bxs if b["text"]]
|
|
818
|
+
if self.mean_height[pagenum - 1] == 0:
|
|
819
|
+
self.mean_height[pagenum - 1] = np.median([b["bottom"] - b["top"] for b in bxs])
|
|
820
|
+
self.boxes.append(bxs)
|
|
821
|
+
|
|
822
|
+
def _layouts_rec(self, ZM, drop=True):
|
|
823
|
+
assert len(self.page_images) == len(self.boxes)
|
|
824
|
+
self.boxes, self.page_layout = self.layouter(self.page_images, self.boxes, ZM, drop=drop)
|
|
825
|
+
# cumlative Y
|
|
826
|
+
for i in range(len(self.boxes)):
|
|
827
|
+
self.boxes[i]["top"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]
|
|
828
|
+
self.boxes[i]["bottom"] += self.page_cum_height[self.boxes[i]["page_number"] - 1]
|
|
829
|
+
|
|
830
|
+
def _assign_column(self, boxes, zoomin=3):
|
|
831
|
+
if not boxes:
|
|
832
|
+
return boxes
|
|
833
|
+
if all("col_id" in b for b in boxes):
|
|
834
|
+
return boxes
|
|
835
|
+
|
|
836
|
+
by_page = defaultdict(list)
|
|
837
|
+
for b in boxes:
|
|
838
|
+
by_page[b["page_number"]].append(b)
|
|
839
|
+
|
|
840
|
+
page_cols = {}
|
|
841
|
+
|
|
842
|
+
for pg, bxs in by_page.items():
|
|
843
|
+
if not bxs:
|
|
844
|
+
page_cols[pg] = 1
|
|
845
|
+
continue
|
|
846
|
+
|
|
847
|
+
x0s_raw = np.array([b["x0"] for b in bxs], dtype=float)
|
|
848
|
+
|
|
849
|
+
min_x0 = np.min(x0s_raw)
|
|
850
|
+
max_x1 = np.max([b["x1"] for b in bxs])
|
|
851
|
+
width = max_x1 - min_x0
|
|
852
|
+
|
|
853
|
+
INDENT_TOL = width * 0.12
|
|
854
|
+
x0s = []
|
|
855
|
+
for x in x0s_raw:
|
|
856
|
+
if abs(x - min_x0) < INDENT_TOL:
|
|
857
|
+
x0s.append([min_x0])
|
|
858
|
+
else:
|
|
859
|
+
x0s.append([x])
|
|
860
|
+
x0s = np.array(x0s, dtype=float)
|
|
861
|
+
|
|
862
|
+
max_try = min(4, len(bxs))
|
|
863
|
+
if max_try < 2:
|
|
864
|
+
max_try = 1
|
|
865
|
+
best_k = 1
|
|
866
|
+
best_score = -1
|
|
867
|
+
|
|
868
|
+
for k in range(1, max_try + 1):
|
|
869
|
+
km = KMeans(n_clusters=k, n_init="auto")
|
|
870
|
+
labels = km.fit_predict(x0s)
|
|
871
|
+
|
|
872
|
+
centers = np.sort(km.cluster_centers_.flatten())
|
|
873
|
+
if len(centers) > 1:
|
|
874
|
+
try:
|
|
875
|
+
score = silhouette_score(x0s, labels)
|
|
876
|
+
except ValueError:
|
|
877
|
+
continue
|
|
878
|
+
else:
|
|
879
|
+
score = 0
|
|
880
|
+
if score > best_score:
|
|
881
|
+
best_score = score
|
|
882
|
+
best_k = k
|
|
883
|
+
|
|
884
|
+
page_cols[pg] = best_k
|
|
885
|
+
logging.info(f"[Page {pg}] best_score={best_score:.2f}, best_k={best_k}")
|
|
886
|
+
|
|
887
|
+
global_cols = Counter(page_cols.values()).most_common(1)[0][0]
|
|
888
|
+
logging.info(f"Global column_num decided by majority: {global_cols}")
|
|
889
|
+
|
|
890
|
+
for pg, bxs in by_page.items():
|
|
891
|
+
if not bxs:
|
|
892
|
+
continue
|
|
893
|
+
k = page_cols[pg]
|
|
894
|
+
if len(bxs) < k:
|
|
895
|
+
k = 1
|
|
896
|
+
x0s = np.array([[b["x0"]] for b in bxs], dtype=float)
|
|
897
|
+
km = KMeans(n_clusters=k, n_init="auto")
|
|
898
|
+
labels = km.fit_predict(x0s)
|
|
899
|
+
|
|
900
|
+
centers = km.cluster_centers_.flatten()
|
|
901
|
+
order = np.argsort(centers)
|
|
902
|
+
|
|
903
|
+
remap = {orig: new for new, orig in enumerate(order)}
|
|
904
|
+
|
|
905
|
+
for b, lb in zip(bxs, labels):
|
|
906
|
+
b["col_id"] = remap[lb]
|
|
907
|
+
|
|
908
|
+
grouped = defaultdict(list)
|
|
909
|
+
for b in bxs:
|
|
910
|
+
grouped[b["col_id"]].append(b)
|
|
911
|
+
|
|
912
|
+
return boxes
|
|
913
|
+
|
|
914
|
+
def _text_merge(self, zoomin=3):
|
|
915
|
+
# merge adjusted boxes
|
|
916
|
+
bxs = self._assign_column(self.boxes, zoomin)
|
|
917
|
+
|
|
918
|
+
def end_with(b, txt):
|
|
919
|
+
txt = txt.strip()
|
|
920
|
+
tt = b.get("text", "").strip()
|
|
921
|
+
return tt and tt.find(txt) == len(tt) - len(txt)
|
|
922
|
+
|
|
923
|
+
def start_with(b, txts):
|
|
924
|
+
tt = b.get("text", "").strip()
|
|
925
|
+
return tt and any([tt.find(t.strip()) == 0 for t in txts])
|
|
926
|
+
|
|
927
|
+
# horizontally merge adjacent box with the same layout
|
|
928
|
+
i = 0
|
|
929
|
+
while i < len(bxs) - 1:
|
|
930
|
+
b = bxs[i]
|
|
931
|
+
b_ = bxs[i + 1]
|
|
932
|
+
|
|
933
|
+
if b["page_number"] != b_["page_number"] or b.get("col_id") != b_.get("col_id"):
|
|
934
|
+
i += 1
|
|
935
|
+
continue
|
|
936
|
+
|
|
937
|
+
if b.get("layoutno", "0") != b_.get("layoutno", "1") or b.get("layout_type", "") in ["table", "figure", "equation"]:
|
|
938
|
+
i += 1
|
|
939
|
+
continue
|
|
940
|
+
|
|
941
|
+
if abs(self._y_dis(b, b_)) < self.mean_height[bxs[i]["page_number"] - 1] / 3:
|
|
942
|
+
# merge
|
|
943
|
+
bxs[i]["x1"] = b_["x1"]
|
|
944
|
+
bxs[i]["top"] = (b["top"] + b_["top"]) / 2
|
|
945
|
+
bxs[i]["bottom"] = (b["bottom"] + b_["bottom"]) / 2
|
|
946
|
+
bxs[i]["text"] += b_["text"]
|
|
947
|
+
bxs.pop(i + 1)
|
|
948
|
+
continue
|
|
949
|
+
i += 1
|
|
950
|
+
self.boxes = bxs
|
|
951
|
+
|
|
952
|
+
def _naive_vertical_merge(self, zoomin=3):
|
|
953
|
+
# bxs = self._assign_column(self.boxes, zoomin)
|
|
954
|
+
bxs = self.boxes
|
|
955
|
+
|
|
956
|
+
grouped = defaultdict(list)
|
|
957
|
+
for b in bxs:
|
|
958
|
+
# grouped[(b["page_number"], b.get("col_id", 0))].append(b)
|
|
959
|
+
grouped[(b["page_number"], "x")].append(b)
|
|
960
|
+
|
|
961
|
+
merged_boxes = []
|
|
962
|
+
for (pg, col), bxs in grouped.items():
|
|
963
|
+
bxs = sorted(bxs, key=lambda x: (x["top"], x["x0"]))
|
|
964
|
+
if not bxs:
|
|
965
|
+
continue
|
|
966
|
+
|
|
967
|
+
mh = self.mean_height[pg - 1] if self.mean_height else np.median([b["bottom"] - b["top"] for b in bxs]) or 10
|
|
968
|
+
|
|
969
|
+
i = 0
|
|
970
|
+
while i + 1 < len(bxs):
|
|
971
|
+
b = bxs[i]
|
|
972
|
+
b_ = bxs[i + 1]
|
|
973
|
+
|
|
974
|
+
if b["page_number"] < b_["page_number"] and re.match(r"[0-9 •一—-]+$", b["text"]):
|
|
975
|
+
bxs.pop(i)
|
|
976
|
+
continue
|
|
977
|
+
|
|
978
|
+
if not b["text"].strip():
|
|
979
|
+
bxs.pop(i)
|
|
980
|
+
continue
|
|
981
|
+
|
|
982
|
+
if not b["text"].strip() or b.get("layoutno") != b_.get("layoutno"):
|
|
983
|
+
i += 1
|
|
984
|
+
continue
|
|
985
|
+
|
|
986
|
+
if b_["top"] - b["bottom"] > mh * 1.5:
|
|
987
|
+
i += 1
|
|
988
|
+
continue
|
|
989
|
+
|
|
990
|
+
overlap = max(0, min(b["x1"], b_["x1"]) - max(b["x0"], b_["x0"]))
|
|
991
|
+
if overlap / max(1, min(b["x1"] - b["x0"], b_["x1"] - b_["x0"])) < 0.3:
|
|
992
|
+
i += 1
|
|
993
|
+
continue
|
|
994
|
+
|
|
995
|
+
concatting_feats = [
|
|
996
|
+
b["text"].strip()[-1] in ",;:'\",、‘“;:-",
|
|
997
|
+
len(b["text"].strip()) > 1 and b["text"].strip()[-2] in ",;:'\",‘“、;:",
|
|
998
|
+
b_["text"].strip() and b_["text"].strip()[0] in "。;?!?”)),,、:",
|
|
999
|
+
]
|
|
1000
|
+
# features for not concating
|
|
1001
|
+
feats = [
|
|
1002
|
+
b.get("layoutno", 0) != b_.get("layoutno", 0),
|
|
1003
|
+
b["text"].strip()[-1] in "。?!?",
|
|
1004
|
+
self.is_english and b["text"].strip()[-1] in ".!?",
|
|
1005
|
+
b["page_number"] == b_["page_number"] and b_["top"] - b["bottom"] > self.mean_height[b["page_number"] - 1] * 1.5,
|
|
1006
|
+
b["page_number"] < b_["page_number"] and abs(b["x0"] - b_["x0"]) > self.mean_width[b["page_number"] - 1] * 4,
|
|
1007
|
+
]
|
|
1008
|
+
# split features
|
|
1009
|
+
detach_feats = [b["x1"] < b_["x0"], b["x0"] > b_["x1"]]
|
|
1010
|
+
if (any(feats) and not any(concatting_feats)) or any(detach_feats):
|
|
1011
|
+
logging.debug(
|
|
1012
|
+
"{} {} {} {}".format(
|
|
1013
|
+
b["text"],
|
|
1014
|
+
b_["text"],
|
|
1015
|
+
any(feats),
|
|
1016
|
+
any(concatting_feats),
|
|
1017
|
+
)
|
|
1018
|
+
)
|
|
1019
|
+
i += 1
|
|
1020
|
+
continue
|
|
1021
|
+
|
|
1022
|
+
b["text"] = (b["text"].rstrip() + " " + b_["text"].lstrip()).strip()
|
|
1023
|
+
b["bottom"] = b_["bottom"]
|
|
1024
|
+
b["x0"] = min(b["x0"], b_["x0"])
|
|
1025
|
+
b["x1"] = max(b["x1"], b_["x1"])
|
|
1026
|
+
bxs.pop(i + 1)
|
|
1027
|
+
|
|
1028
|
+
merged_boxes.extend(bxs)
|
|
1029
|
+
|
|
1030
|
+
# self.boxes = sorted(merged_boxes, key=lambda x: (x["page_number"], x.get("col_id", 0), x["top"]))
|
|
1031
|
+
self.boxes = merged_boxes
|
|
1032
|
+
|
|
1033
|
+
def _concat_downward(self, concat_between_pages=True):
|
|
1034
|
+
self.boxes = Recognizer.sort_Y_firstly(self.boxes, 0)
|
|
1035
|
+
return
|
|
1036
|
+
|
|
1037
|
+
def _filter_forpages(self):
|
|
1038
|
+
if not self.boxes:
|
|
1039
|
+
return
|
|
1040
|
+
findit = False
|
|
1041
|
+
i = 0
|
|
1042
|
+
while i < len(self.boxes):
|
|
1043
|
+
if not re.match(r"(contents|目录|目次|table of contents|致谢|acknowledge)$", re.sub(r"( | |\u3000)+", "", self.boxes[i]["text"].lower())):
|
|
1044
|
+
i += 1
|
|
1045
|
+
continue
|
|
1046
|
+
findit = True
|
|
1047
|
+
eng = re.match(r"[0-9a-zA-Z :'.-]{5,}", self.boxes[i]["text"].strip())
|
|
1048
|
+
self.boxes.pop(i)
|
|
1049
|
+
if i >= len(self.boxes):
|
|
1050
|
+
break
|
|
1051
|
+
prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
|
|
1052
|
+
while not prefix:
|
|
1053
|
+
self.boxes.pop(i)
|
|
1054
|
+
if i >= len(self.boxes):
|
|
1055
|
+
break
|
|
1056
|
+
prefix = self.boxes[i]["text"].strip()[:3] if not eng else " ".join(self.boxes[i]["text"].strip().split()[:2])
|
|
1057
|
+
self.boxes.pop(i)
|
|
1058
|
+
if i >= len(self.boxes) or not prefix:
|
|
1059
|
+
break
|
|
1060
|
+
for j in range(i, min(i + 128, len(self.boxes))):
|
|
1061
|
+
if not re.match(prefix, self.boxes[j]["text"]):
|
|
1062
|
+
continue
|
|
1063
|
+
for k in range(i, j):
|
|
1064
|
+
self.boxes.pop(i)
|
|
1065
|
+
break
|
|
1066
|
+
if findit:
|
|
1067
|
+
return
|
|
1068
|
+
|
|
1069
|
+
page_dirty = [0] * len(self.page_images)
|
|
1070
|
+
for b in self.boxes:
|
|
1071
|
+
if re.search(r"(··|··|··)", b["text"]):
|
|
1072
|
+
page_dirty[b["page_number"] - 1] += 1
|
|
1073
|
+
page_dirty = set([i + 1 for i, t in enumerate(page_dirty) if t > 3])
|
|
1074
|
+
if not page_dirty:
|
|
1075
|
+
return
|
|
1076
|
+
i = 0
|
|
1077
|
+
while i < len(self.boxes):
|
|
1078
|
+
if self.boxes[i]["page_number"] in page_dirty:
|
|
1079
|
+
self.boxes.pop(i)
|
|
1080
|
+
continue
|
|
1081
|
+
i += 1
|
|
1082
|
+
|
|
1083
|
+
def _merge_with_same_bullet(self):
|
|
1084
|
+
i = 0
|
|
1085
|
+
while i + 1 < len(self.boxes):
|
|
1086
|
+
b = self.boxes[i]
|
|
1087
|
+
b_ = self.boxes[i + 1]
|
|
1088
|
+
if not b["text"].strip():
|
|
1089
|
+
self.boxes.pop(i)
|
|
1090
|
+
continue
|
|
1091
|
+
if not b_["text"].strip():
|
|
1092
|
+
self.boxes.pop(i + 1)
|
|
1093
|
+
continue
|
|
1094
|
+
|
|
1095
|
+
if (
|
|
1096
|
+
b["text"].strip()[0] != b_["text"].strip()[0]
|
|
1097
|
+
or b["text"].strip()[0].lower() in set("qwertyuopasdfghjklzxcvbnm")
|
|
1098
|
+
or is_chinese(b["text"].strip()[0])
|
|
1099
|
+
or b["top"] > b_["bottom"]
|
|
1100
|
+
):
|
|
1101
|
+
i += 1
|
|
1102
|
+
continue
|
|
1103
|
+
b_["text"] = b["text"] + "\n" + b_["text"]
|
|
1104
|
+
b_["x0"] = min(b["x0"], b_["x0"])
|
|
1105
|
+
b_["x1"] = max(b["x1"], b_["x1"])
|
|
1106
|
+
b_["top"] = b["top"]
|
|
1107
|
+
self.boxes.pop(i)
|
|
1108
|
+
|
|
1109
|
+
def _extract_table_figure(self, need_image, ZM, return_html, need_position, separate_tables_figures=False):
|
|
1110
|
+
tables = {}
|
|
1111
|
+
figures = {}
|
|
1112
|
+
# extract figure and table boxes
|
|
1113
|
+
i = 0
|
|
1114
|
+
lst_lout_no = ""
|
|
1115
|
+
nomerge_lout_no = []
|
|
1116
|
+
while i < len(self.boxes):
|
|
1117
|
+
if "layoutno" not in self.boxes[i]:
|
|
1118
|
+
i += 1
|
|
1119
|
+
continue
|
|
1120
|
+
lout_no = str(self.boxes[i]["page_number"]) + "-" + str(self.boxes[i]["layoutno"])
|
|
1121
|
+
if TableStructureRecognizer.is_caption(self.boxes[i]) or self.boxes[i]["layout_type"] in ["table caption", "title", "figure caption", "reference"]:
|
|
1122
|
+
nomerge_lout_no.append(lst_lout_no)
|
|
1123
|
+
if self.boxes[i]["layout_type"] == "table":
|
|
1124
|
+
if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
|
|
1125
|
+
self.boxes.pop(i)
|
|
1126
|
+
continue
|
|
1127
|
+
if lout_no not in tables:
|
|
1128
|
+
tables[lout_no] = []
|
|
1129
|
+
tables[lout_no].append(self.boxes[i])
|
|
1130
|
+
self.boxes.pop(i)
|
|
1131
|
+
lst_lout_no = lout_no
|
|
1132
|
+
continue
|
|
1133
|
+
if need_image and self.boxes[i]["layout_type"] == "figure":
|
|
1134
|
+
if re.match(r"(数据|资料|图表)*来源[:: ]", self.boxes[i]["text"]):
|
|
1135
|
+
self.boxes.pop(i)
|
|
1136
|
+
continue
|
|
1137
|
+
if lout_no not in figures:
|
|
1138
|
+
figures[lout_no] = []
|
|
1139
|
+
figures[lout_no].append(self.boxes[i])
|
|
1140
|
+
self.boxes.pop(i)
|
|
1141
|
+
lst_lout_no = lout_no
|
|
1142
|
+
continue
|
|
1143
|
+
i += 1
|
|
1144
|
+
|
|
1145
|
+
# merge table on different pages
|
|
1146
|
+
nomerge_lout_no = set(nomerge_lout_no)
|
|
1147
|
+
tbls = sorted([(k, bxs) for k, bxs in tables.items()], key=lambda x: (x[1][0]["top"], x[1][0]["x0"]))
|
|
1148
|
+
|
|
1149
|
+
i = len(tbls) - 1
|
|
1150
|
+
while i - 1 >= 0:
|
|
1151
|
+
k0, bxs0 = tbls[i - 1]
|
|
1152
|
+
k, bxs = tbls[i]
|
|
1153
|
+
i -= 1
|
|
1154
|
+
if k0 in nomerge_lout_no:
|
|
1155
|
+
continue
|
|
1156
|
+
if bxs[0]["page_number"] == bxs0[0]["page_number"]:
|
|
1157
|
+
continue
|
|
1158
|
+
if bxs[0]["page_number"] - bxs0[0]["page_number"] > 1:
|
|
1159
|
+
continue
|
|
1160
|
+
mh = self.mean_height[bxs[0]["page_number"] - 1]
|
|
1161
|
+
if self._y_dis(bxs0[-1], bxs[0]) > mh * 23:
|
|
1162
|
+
continue
|
|
1163
|
+
tables[k0].extend(tables[k])
|
|
1164
|
+
del tables[k]
|
|
1165
|
+
|
|
1166
|
+
def x_overlapped(a, b):
|
|
1167
|
+
return not any([a["x1"] < b["x0"], a["x0"] > b["x1"]])
|
|
1168
|
+
|
|
1169
|
+
# find captions and pop out
|
|
1170
|
+
i = 0
|
|
1171
|
+
while i < len(self.boxes):
|
|
1172
|
+
c = self.boxes[i]
|
|
1173
|
+
# mh = self.mean_height[c["page_number"]-1]
|
|
1174
|
+
if not TableStructureRecognizer.is_caption(c):
|
|
1175
|
+
i += 1
|
|
1176
|
+
continue
|
|
1177
|
+
|
|
1178
|
+
# find the nearest layouts
|
|
1179
|
+
def nearest(tbls):
|
|
1180
|
+
nonlocal c
|
|
1181
|
+
mink = ""
|
|
1182
|
+
minv = 1000000000
|
|
1183
|
+
for k, bxs in tbls.items():
|
|
1184
|
+
for b in bxs:
|
|
1185
|
+
if b.get("layout_type", "").find("caption") >= 0:
|
|
1186
|
+
continue
|
|
1187
|
+
y_dis = self._y_dis(c, b)
|
|
1188
|
+
x_dis = self._x_dis(c, b) if not x_overlapped(c, b) else 0
|
|
1189
|
+
dis = y_dis * y_dis + x_dis * x_dis
|
|
1190
|
+
if dis < minv:
|
|
1191
|
+
mink = k
|
|
1192
|
+
minv = dis
|
|
1193
|
+
return mink, minv
|
|
1194
|
+
|
|
1195
|
+
tk, tv = nearest(tables)
|
|
1196
|
+
fk, fv = nearest(figures)
|
|
1197
|
+
# if min(tv, fv) > 2000:
|
|
1198
|
+
# i += 1
|
|
1199
|
+
# continue
|
|
1200
|
+
if tv < fv and tk:
|
|
1201
|
+
tables[tk].insert(0, c)
|
|
1202
|
+
logging.debug("TABLE:" + self.boxes[i]["text"] + "; Cap: " + tk)
|
|
1203
|
+
elif fk:
|
|
1204
|
+
figures[fk].insert(0, c)
|
|
1205
|
+
logging.debug("FIGURE:" + self.boxes[i]["text"] + "; Cap: " + tk)
|
|
1206
|
+
self.boxes.pop(i)
|
|
1207
|
+
|
|
1208
|
+
def cropout(bxs, ltype, poss):
|
|
1209
|
+
nonlocal ZM
|
|
1210
|
+
max_page_index = len(self.page_images) - 1
|
|
1211
|
+
|
|
1212
|
+
def local_page_index(page_number):
|
|
1213
|
+
idx = page_number - 1 if page_number > 0 else 0
|
|
1214
|
+
if idx > max_page_index and self.page_from:
|
|
1215
|
+
idx = page_number - 1 - self.page_from
|
|
1216
|
+
return idx
|
|
1217
|
+
|
|
1218
|
+
pn = set()
|
|
1219
|
+
for b in bxs:
|
|
1220
|
+
idx = local_page_index(b["page_number"])
|
|
1221
|
+
if 0 <= idx <= max_page_index:
|
|
1222
|
+
pn.add(idx)
|
|
1223
|
+
else:
|
|
1224
|
+
logging.warning(
|
|
1225
|
+
"Skip out-of-range page_number %s (page_from=%s, pages=%s)",
|
|
1226
|
+
b.get("page_number"),
|
|
1227
|
+
self.page_from,
|
|
1228
|
+
len(self.page_images),
|
|
1229
|
+
)
|
|
1230
|
+
|
|
1231
|
+
if not pn:
|
|
1232
|
+
return None
|
|
1233
|
+
|
|
1234
|
+
if len(pn) < 2:
|
|
1235
|
+
pn = list(pn)[0]
|
|
1236
|
+
ht = self.page_cum_height[pn]
|
|
1237
|
+
b = {"x0": np.min([b["x0"] for b in bxs]), "top": np.min([b["top"] for b in bxs]) - ht, "x1": np.max([b["x1"] for b in bxs]), "bottom": np.max([b["bottom"] for b in bxs]) - ht}
|
|
1238
|
+
louts = [layout for layout in self.page_layout[pn] if layout["type"] == ltype]
|
|
1239
|
+
ii = Recognizer.find_overlapped(b, louts, naive=True)
|
|
1240
|
+
if ii is not None:
|
|
1241
|
+
b = louts[ii]
|
|
1242
|
+
else:
|
|
1243
|
+
logging.warning(f"Missing layout match: {pn + 1},%s" % (bxs[0].get("layoutno", "")))
|
|
1244
|
+
|
|
1245
|
+
left, top, right, bott = b["x0"], b["top"], b["x1"], b["bottom"]
|
|
1246
|
+
if right < left:
|
|
1247
|
+
right = left + 1
|
|
1248
|
+
poss.append((pn + self.page_from, left, right, top, bott))
|
|
1249
|
+
return self.page_images[pn].crop((left * ZM, top * ZM, right * ZM, bott * ZM))
|
|
1250
|
+
pn = {}
|
|
1251
|
+
for b in bxs:
|
|
1252
|
+
p = local_page_index(b["page_number"])
|
|
1253
|
+
if 0 <= p <= max_page_index:
|
|
1254
|
+
if p not in pn:
|
|
1255
|
+
pn[p] = []
|
|
1256
|
+
pn[p].append(b)
|
|
1257
|
+
pn = sorted(pn.items(), key=lambda x: x[0])
|
|
1258
|
+
imgs = [cropout(arr, ltype, poss) for p, arr in pn]
|
|
1259
|
+
imgs = [img for img in imgs if img is not None]
|
|
1260
|
+
if not imgs:
|
|
1261
|
+
return None
|
|
1262
|
+
pic = Image.new("RGB", (int(np.max([i.size[0] for i in imgs])), int(np.sum([m.size[1] for m in imgs]))), (245, 245, 245))
|
|
1263
|
+
height = 0
|
|
1264
|
+
for img in imgs:
|
|
1265
|
+
pic.paste(img, (0, int(height)))
|
|
1266
|
+
height += img.size[1]
|
|
1267
|
+
return pic
|
|
1268
|
+
|
|
1269
|
+
res = []
|
|
1270
|
+
positions = []
|
|
1271
|
+
figure_results = []
|
|
1272
|
+
figure_positions = []
|
|
1273
|
+
# crop figure out and add caption
|
|
1274
|
+
for k, bxs in figures.items():
|
|
1275
|
+
txt = "\n".join([b["text"] for b in bxs])
|
|
1276
|
+
if not txt:
|
|
1277
|
+
continue
|
|
1278
|
+
|
|
1279
|
+
poss = []
|
|
1280
|
+
|
|
1281
|
+
if separate_tables_figures:
|
|
1282
|
+
img = cropout(bxs, "figure", poss)
|
|
1283
|
+
if img is None:
|
|
1284
|
+
continue
|
|
1285
|
+
figure_results.append((img, [txt]))
|
|
1286
|
+
figure_positions.append(poss)
|
|
1287
|
+
else:
|
|
1288
|
+
img = cropout(bxs, "figure", poss)
|
|
1289
|
+
if img is None:
|
|
1290
|
+
continue
|
|
1291
|
+
res.append((img, [txt]))
|
|
1292
|
+
positions.append(poss)
|
|
1293
|
+
|
|
1294
|
+
for k, bxs in tables.items():
|
|
1295
|
+
if not bxs:
|
|
1296
|
+
continue
|
|
1297
|
+
bxs = Recognizer.sort_Y_firstly(bxs, np.mean([(b["bottom"] - b["top"]) / 2 for b in bxs]))
|
|
1298
|
+
|
|
1299
|
+
poss = []
|
|
1300
|
+
|
|
1301
|
+
img = cropout(bxs, "table", poss)
|
|
1302
|
+
if img is None:
|
|
1303
|
+
continue
|
|
1304
|
+
res.append((img, self.tbl_det.construct_table(bxs, html=return_html, is_english=self.is_english)))
|
|
1305
|
+
positions.append(poss)
|
|
1306
|
+
|
|
1307
|
+
if separate_tables_figures:
|
|
1308
|
+
assert len(positions) + len(figure_positions) == len(res) + len(figure_results)
|
|
1309
|
+
if need_position:
|
|
1310
|
+
return list(zip(res, positions)), list(zip(figure_results, figure_positions))
|
|
1311
|
+
else:
|
|
1312
|
+
return res, figure_results
|
|
1313
|
+
else:
|
|
1314
|
+
assert len(positions) == len(res)
|
|
1315
|
+
if need_position:
|
|
1316
|
+
return list(zip(res, positions))
|
|
1317
|
+
else:
|
|
1318
|
+
return res
|
|
1319
|
+
|
|
1320
|
+
def proj_match(self, line):
|
|
1321
|
+
if len(line) <= 2:
|
|
1322
|
+
return
|
|
1323
|
+
if re.match(r"[0-9 ().,%%+/-]+$", line):
|
|
1324
|
+
return False
|
|
1325
|
+
for p, j in [
|
|
1326
|
+
(r"第[零一二三四五六七八九十百]+章", 1),
|
|
1327
|
+
(r"第[零一二三四五六七八九十百]+[条节]", 2),
|
|
1328
|
+
(r"[零一二三四五六七八九十百]+[、 ]", 3),
|
|
1329
|
+
(r"[\((][零一二三四五六七八九十百]+[)\)]", 4),
|
|
1330
|
+
(r"[0-9]+(、|\.[ ]|\.[^0-9])", 5),
|
|
1331
|
+
(r"[0-9]+\.[0-9]+(、|[. ]|[^0-9])", 6),
|
|
1332
|
+
(r"[0-9]+\.[0-9]+\.[0-9]+(、|[ ]|[^0-9])", 7),
|
|
1333
|
+
(r"[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+(、|[ ]|[^0-9])", 8),
|
|
1334
|
+
(r".{,48}[::??]$", 9),
|
|
1335
|
+
(r"[0-9]+)", 10),
|
|
1336
|
+
(r"[\((][0-9]+[)\)]", 11),
|
|
1337
|
+
(r"[零一二三四五六七八九十百]+是", 12),
|
|
1338
|
+
(r"[⚫•➢✓]", 12),
|
|
1339
|
+
]:
|
|
1340
|
+
if re.match(p, line):
|
|
1341
|
+
return j
|
|
1342
|
+
return
|
|
1343
|
+
|
|
1344
|
+
def _line_tag(self, bx, ZM):
|
|
1345
|
+
pn = [bx["page_number"]]
|
|
1346
|
+
top = bx["top"] - self.page_cum_height[pn[0] - 1]
|
|
1347
|
+
bott = bx["bottom"] - self.page_cum_height[pn[0] - 1]
|
|
1348
|
+
page_images_cnt = len(self.page_images)
|
|
1349
|
+
if pn[-1] - 1 >= page_images_cnt:
|
|
1350
|
+
return ""
|
|
1351
|
+
while bott * ZM > self.page_images[pn[-1] - 1].size[1]:
|
|
1352
|
+
bott -= self.page_images[pn[-1] - 1].size[1] / ZM
|
|
1353
|
+
pn.append(pn[-1] + 1)
|
|
1354
|
+
if pn[-1] - 1 >= page_images_cnt:
|
|
1355
|
+
return ""
|
|
1356
|
+
|
|
1357
|
+
return "@@{}\t{:.1f}\t{:.1f}\t{:.1f}\t{:.1f}##".format("-".join([str(p) for p in pn]), bx["x0"], bx["x1"], top, bott)
|
|
1358
|
+
|
|
1359
|
+
def __filterout_scraps(self, boxes, ZM):
|
|
1360
|
+
def width(b):
|
|
1361
|
+
return b["x1"] - b["x0"]
|
|
1362
|
+
|
|
1363
|
+
def height(b):
|
|
1364
|
+
return b["bottom"] - b["top"]
|
|
1365
|
+
|
|
1366
|
+
def usefull(b):
|
|
1367
|
+
if b.get("layout_type"):
|
|
1368
|
+
return True
|
|
1369
|
+
if width(b) > self.page_images[b["page_number"] - 1].size[0] / ZM / 3:
|
|
1370
|
+
return True
|
|
1371
|
+
if b["bottom"] - b["top"] > self.mean_height[b["page_number"] - 1]:
|
|
1372
|
+
return True
|
|
1373
|
+
return False
|
|
1374
|
+
|
|
1375
|
+
res = []
|
|
1376
|
+
while boxes:
|
|
1377
|
+
lines = []
|
|
1378
|
+
widths = []
|
|
1379
|
+
pw = self.page_images[boxes[0]["page_number"] - 1].size[0] / ZM
|
|
1380
|
+
mh = self.mean_height[boxes[0]["page_number"] - 1]
|
|
1381
|
+
mj = self.proj_match(boxes[0]["text"]) or boxes[0].get("layout_type", "") == "title"
|
|
1382
|
+
|
|
1383
|
+
def dfs(line, st):
|
|
1384
|
+
nonlocal mh, pw, lines, widths
|
|
1385
|
+
lines.append(line)
|
|
1386
|
+
widths.append(width(line))
|
|
1387
|
+
mmj = self.proj_match(line["text"]) or line.get("layout_type", "") == "title"
|
|
1388
|
+
for i in range(st + 1, min(st + 20, len(boxes))):
|
|
1389
|
+
if (boxes[i]["page_number"] - line["page_number"]) > 0:
|
|
1390
|
+
break
|
|
1391
|
+
if not mmj and self._y_dis(line, boxes[i]) >= 3 * mh and height(line) < 1.5 * mh:
|
|
1392
|
+
break
|
|
1393
|
+
|
|
1394
|
+
if not usefull(boxes[i]):
|
|
1395
|
+
continue
|
|
1396
|
+
if mmj or (self._x_dis(boxes[i], line) < pw / 10):
|
|
1397
|
+
# and abs(width(boxes[i])-width_mean)/max(width(boxes[i]),width_mean)<0.5):
|
|
1398
|
+
# concat following
|
|
1399
|
+
dfs(boxes[i], i)
|
|
1400
|
+
boxes.pop(i)
|
|
1401
|
+
break
|
|
1402
|
+
|
|
1403
|
+
try:
|
|
1404
|
+
if usefull(boxes[0]):
|
|
1405
|
+
dfs(boxes[0], 0)
|
|
1406
|
+
else:
|
|
1407
|
+
logging.debug("WASTE: " + boxes[0]["text"])
|
|
1408
|
+
except Exception:
|
|
1409
|
+
pass
|
|
1410
|
+
boxes.pop(0)
|
|
1411
|
+
mw = np.mean(widths)
|
|
1412
|
+
if mj or mw / pw >= 0.35 or mw > 200:
|
|
1413
|
+
res.append("\n".join([c["text"] + self._line_tag(c, ZM) for c in lines]))
|
|
1414
|
+
else:
|
|
1415
|
+
logging.debug("REMOVED: " + "<<".join([c["text"] for c in lines]))
|
|
1416
|
+
|
|
1417
|
+
return "\n\n".join(res)
|
|
1418
|
+
|
|
1419
|
+
@staticmethod
|
|
1420
|
+
def total_page_number(fnm, binary=None):
|
|
1421
|
+
try:
|
|
1422
|
+
with sys.modules[LOCK_KEY_pdfplumber]:
|
|
1423
|
+
pdf = pdfplumber.open(fnm) if not binary else pdfplumber.open(BytesIO(binary))
|
|
1424
|
+
total_page = len(pdf.pages)
|
|
1425
|
+
pdf.close()
|
|
1426
|
+
return total_page
|
|
1427
|
+
except Exception:
|
|
1428
|
+
logging.exception("total_page_number")
|
|
1429
|
+
|
|
1430
|
+
def __images__(self, fnm, zoomin=3, page_from=0, page_to=MAXIMUM_PAGE_NUMBER, callback=None):
|
|
1431
|
+
self.lefted_chars = []
|
|
1432
|
+
self.mean_height = []
|
|
1433
|
+
self.mean_width = []
|
|
1434
|
+
self.boxes = []
|
|
1435
|
+
self.garbages = {}
|
|
1436
|
+
self.page_cum_height = [0]
|
|
1437
|
+
self.page_layout = []
|
|
1438
|
+
self.page_from = page_from
|
|
1439
|
+
start = timer()
|
|
1440
|
+
try:
|
|
1441
|
+
with sys.modules[LOCK_KEY_pdfplumber]:
|
|
1442
|
+
with pdfplumber.open(fnm) if isinstance(fnm, str) else pdfplumber.open(BytesIO(fnm)) as pdf:
|
|
1443
|
+
self.pdf = pdf
|
|
1444
|
+
self.page_images = [p.to_image(resolution=72 * zoomin, antialias=True).annotated for i, p in enumerate(self.pdf.pages[page_from:page_to])]
|
|
1445
|
+
|
|
1446
|
+
try:
|
|
1447
|
+
self.page_chars = [[c for c in page.dedupe_chars().chars if self._has_color(c)] for page in self.pdf.pages[page_from:page_to]]
|
|
1448
|
+
except Exception as e:
|
|
1449
|
+
logging.warning(f"Failed to extract characters for pages {page_from}-{page_to}: {str(e)}")
|
|
1450
|
+
self.page_chars = [[] for _ in range(len(self.page_images))] # If failed to extract, using empty list instead.
|
|
1451
|
+
|
|
1452
|
+
# Detect garbled pages and clear their chars so the OCR
|
|
1453
|
+
# path will be used instead. Two detection strategies:
|
|
1454
|
+
# 1) PUA / unmapped CID characters (threshold=0.3)
|
|
1455
|
+
# 2) Font-encoding garbling: subset fonts mapping CJK to ASCII
|
|
1456
|
+
for pi, page_ch in enumerate(self.page_chars):
|
|
1457
|
+
if not page_ch:
|
|
1458
|
+
continue
|
|
1459
|
+
# Strategy 1: PUA / CID garbling
|
|
1460
|
+
sample = page_ch if len(page_ch) <= 200 else page_ch[:200]
|
|
1461
|
+
sample_text = "".join(c.get("text", "") for c in sample)
|
|
1462
|
+
if self._is_garbled_text(sample_text, threshold=0.3):
|
|
1463
|
+
logging.warning(
|
|
1464
|
+
"Page %d: pdfplumber extracted mostly garbled characters (%d chars), clearing to use OCR fallback.",
|
|
1465
|
+
page_from + pi + 1,
|
|
1466
|
+
len(page_ch),
|
|
1467
|
+
)
|
|
1468
|
+
self.page_chars[pi] = []
|
|
1469
|
+
continue
|
|
1470
|
+
# Strategy 2: font-encoding garbling (CJK mapped to ASCII)
|
|
1471
|
+
if self._is_garbled_by_font_encoding(page_ch):
|
|
1472
|
+
logging.warning(
|
|
1473
|
+
"Page %d: detected font-encoding garbled text (subset fonts with no CJK output, %d chars), clearing to use OCR fallback.",
|
|
1474
|
+
page_from + pi + 1,
|
|
1475
|
+
len(page_ch),
|
|
1476
|
+
)
|
|
1477
|
+
self.page_chars[pi] = []
|
|
1478
|
+
|
|
1479
|
+
self.total_page = len(self.pdf.pages)
|
|
1480
|
+
|
|
1481
|
+
except Exception as e:
|
|
1482
|
+
logging.exception(f"RAGFlowPdfParser __images__, exception: {e}")
|
|
1483
|
+
logging.info(f"__images__ dedupe_chars cost {timer() - start}s")
|
|
1484
|
+
|
|
1485
|
+
logging.debug("Images converted.")
|
|
1486
|
+
self.is_english = [
|
|
1487
|
+
re.search(r"[ a-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join(random.choices([c["text"] for c in self.page_chars[i]], k=min(100, len(self.page_chars[i])))))
|
|
1488
|
+
for i in range(len(self.page_chars))
|
|
1489
|
+
]
|
|
1490
|
+
if sum([1 if e else 0 for e in self.is_english]) > len(self.page_images) / 2:
|
|
1491
|
+
self.is_english = True
|
|
1492
|
+
else:
|
|
1493
|
+
self.is_english = False
|
|
1494
|
+
|
|
1495
|
+
async def __img_ocr(i, id, img, chars, limiter):
|
|
1496
|
+
self._insert_word_spaces(chars)
|
|
1497
|
+
|
|
1498
|
+
if limiter:
|
|
1499
|
+
async with limiter:
|
|
1500
|
+
await thread_pool_exec(self.__ocr, i + 1, img, chars, zoomin, id)
|
|
1501
|
+
else:
|
|
1502
|
+
self.__ocr(i + 1, img, chars, zoomin, id)
|
|
1503
|
+
|
|
1504
|
+
if callback and i % 6 == 5:
|
|
1505
|
+
callback((i + 1) * 0.6 / len(self.page_images))
|
|
1506
|
+
|
|
1507
|
+
async def __img_ocr_launcher():
|
|
1508
|
+
def __ocr_preprocess():
|
|
1509
|
+
chars = self.page_chars[i] if not self.is_english else []
|
|
1510
|
+
self.mean_height.append(np.median(sorted([c["height"] for c in chars])) if chars else 0)
|
|
1511
|
+
self.mean_width.append(np.median(sorted([c["width"] for c in chars])) if chars else 8)
|
|
1512
|
+
self.page_cum_height.append(img.size[1] / zoomin)
|
|
1513
|
+
return chars
|
|
1514
|
+
|
|
1515
|
+
if self.parallel_limiter:
|
|
1516
|
+
tasks = []
|
|
1517
|
+
|
|
1518
|
+
for i, img in enumerate(self.page_images):
|
|
1519
|
+
chars = __ocr_preprocess()
|
|
1520
|
+
|
|
1521
|
+
semaphore = self.parallel_limiter[i % PARALLEL_DEVICES]
|
|
1522
|
+
|
|
1523
|
+
async def wrapper(i=i, img=img, chars=chars, semaphore=semaphore):
|
|
1524
|
+
await __img_ocr(
|
|
1525
|
+
i,
|
|
1526
|
+
i % PARALLEL_DEVICES,
|
|
1527
|
+
img,
|
|
1528
|
+
chars,
|
|
1529
|
+
semaphore,
|
|
1530
|
+
)
|
|
1531
|
+
|
|
1532
|
+
tasks.append(asyncio.create_task(wrapper()))
|
|
1533
|
+
await asyncio.sleep(0)
|
|
1534
|
+
|
|
1535
|
+
try:
|
|
1536
|
+
await asyncio.gather(*tasks, return_exceptions=False)
|
|
1537
|
+
except Exception as e:
|
|
1538
|
+
logging.error(f"Error in OCR: {e}")
|
|
1539
|
+
for t in tasks:
|
|
1540
|
+
t.cancel()
|
|
1541
|
+
await asyncio.gather(*tasks, return_exceptions=True)
|
|
1542
|
+
raise
|
|
1543
|
+
|
|
1544
|
+
else:
|
|
1545
|
+
for i, img in enumerate(self.page_images):
|
|
1546
|
+
chars = __ocr_preprocess()
|
|
1547
|
+
await __img_ocr(i, 0, img, chars, None)
|
|
1548
|
+
|
|
1549
|
+
start = timer()
|
|
1550
|
+
|
|
1551
|
+
asyncio.run(__img_ocr_launcher())
|
|
1552
|
+
|
|
1553
|
+
logging.info(f"__images__ {len(self.page_images)} pages cost {timer() - start}s")
|
|
1554
|
+
|
|
1555
|
+
if not self.is_english and not any([c for c in self.page_chars]) and self.boxes:
|
|
1556
|
+
bxes = [b for bxs in self.boxes for b in bxs]
|
|
1557
|
+
self.is_english = re.search(r"[ \na-zA-Z0-9,/¸;:'\[\]\(\)!@#$%^&*\"?<>._-]{30,}", "".join([b["text"] for b in random.choices(bxes, k=min(30, len(bxes)))]))
|
|
1558
|
+
|
|
1559
|
+
logging.debug(f"Is it English: {self.is_english}")
|
|
1560
|
+
|
|
1561
|
+
self.page_cum_height = np.cumsum(self.page_cum_height)
|
|
1562
|
+
assert len(self.page_cum_height) == len(self.page_images) + 1
|
|
1563
|
+
if len(self.boxes) == 0 and zoomin < 9:
|
|
1564
|
+
self.__images__(fnm, zoomin * 3, page_from, page_to, callback)
|
|
1565
|
+
|
|
1566
|
+
def __call__(self, fnm, need_image=True, zoomin=3, return_html=False, auto_rotate_tables=None):
|
|
1567
|
+
"""
|
|
1568
|
+
Parse a PDF file.
|
|
1569
|
+
|
|
1570
|
+
Args:
|
|
1571
|
+
fnm: PDF file path or binary content
|
|
1572
|
+
need_image: Whether to extract images
|
|
1573
|
+
zoomin: Zoom factor
|
|
1574
|
+
return_html: Whether to return tables in HTML format
|
|
1575
|
+
auto_rotate_tables: Whether to enable auto orientation correction for tables.
|
|
1576
|
+
None: Use TABLE_AUTO_ROTATE env var setting (default: True)
|
|
1577
|
+
True: Enable auto orientation correction
|
|
1578
|
+
False: Disable auto orientation correction
|
|
1579
|
+
"""
|
|
1580
|
+
if auto_rotate_tables is None:
|
|
1581
|
+
auto_rotate_tables = os.getenv("TABLE_AUTO_ROTATE", "true").lower() in ("true", "1", "yes")
|
|
1582
|
+
|
|
1583
|
+
self.outlines = extract_pdf_outlines(fnm)
|
|
1584
|
+
self.__images__(fnm, zoomin)
|
|
1585
|
+
self._layouts_rec(zoomin)
|
|
1586
|
+
self._table_transformer_job(zoomin, auto_rotate=auto_rotate_tables)
|
|
1587
|
+
self._text_merge()
|
|
1588
|
+
self._concat_downward()
|
|
1589
|
+
self._filter_forpages()
|
|
1590
|
+
tbls = self._extract_table_figure(need_image, zoomin, return_html, False)
|
|
1591
|
+
return self.__filterout_scraps(deepcopy(self.boxes), zoomin), tbls
|
|
1592
|
+
|
|
1593
|
+
def parse_into_bboxes(self, fnm, callback=None, zoomin=3, from_page=0, to_page=MAXIMUM_PAGE_NUMBER):
|
|
1594
|
+
self.outlines = extract_pdf_outlines(fnm)
|
|
1595
|
+
batch_size = max(1, int(os.getenv("PDF_PARSER_PAGE_BATCH_SIZE", "50")))
|
|
1596
|
+
if isinstance(fnm, str):
|
|
1597
|
+
total_pages = self.total_page_number(fnm)
|
|
1598
|
+
else:
|
|
1599
|
+
total_pages = self.total_page_number(fnm, binary=fnm)
|
|
1600
|
+
|
|
1601
|
+
if total_pages is None:
|
|
1602
|
+
effective_to_page = to_page
|
|
1603
|
+
logging.warning(
|
|
1604
|
+
"parse_into_bboxes: total_page_number returned None; using caller-supplied to_page=%s",
|
|
1605
|
+
to_page,
|
|
1606
|
+
)
|
|
1607
|
+
else:
|
|
1608
|
+
effective_to_page = min(to_page, total_pages)
|
|
1609
|
+
|
|
1610
|
+
if effective_to_page - from_page <= batch_size:
|
|
1611
|
+
self.__images__(fnm, zoomin, page_from=from_page, page_to=effective_to_page, callback=callback)
|
|
1612
|
+
return self._parse_loaded_window_into_bboxes(zoomin, callback=callback)
|
|
1613
|
+
|
|
1614
|
+
logging.info(
|
|
1615
|
+
"parse_into_bboxes uses chunk mode: from_page=%s, effective_to_page=%s, batch_size=%s",
|
|
1616
|
+
from_page,
|
|
1617
|
+
effective_to_page,
|
|
1618
|
+
batch_size,
|
|
1619
|
+
)
|
|
1620
|
+
all_boxes = []
|
|
1621
|
+
start = timer()
|
|
1622
|
+
for page_from in range(from_page, effective_to_page, batch_size):
|
|
1623
|
+
page_to = min(page_from + batch_size, effective_to_page)
|
|
1624
|
+
self.__images__(fnm, zoomin, page_from=page_from, page_to=page_to, callback=None)
|
|
1625
|
+
chunk_boxes = self._parse_loaded_window_into_bboxes(zoomin)
|
|
1626
|
+
all_boxes.extend(self._to_global_boxes(chunk_boxes))
|
|
1627
|
+
if callback:
|
|
1628
|
+
callback((page_to - from_page) / max(1, effective_to_page - from_page), f"Structured: {page_to}/{effective_to_page} pages")
|
|
1629
|
+
|
|
1630
|
+
logging.info("parse_into_bboxes chunk mode cost %.2fs", timer() - start)
|
|
1631
|
+
return all_boxes
|
|
1632
|
+
|
|
1633
|
+
def _parse_loaded_window_into_bboxes(self, zoomin=3, callback=None):
|
|
1634
|
+
start = timer()
|
|
1635
|
+
self._layouts_rec(zoomin)
|
|
1636
|
+
if callback:
|
|
1637
|
+
callback(0.63, "Layout analysis ({:.2f}s)".format(timer() - start))
|
|
1638
|
+
|
|
1639
|
+
auto_rotate_tables = os.getenv("TABLE_AUTO_ROTATE", "true").lower() in ("true", "1", "yes")
|
|
1640
|
+
|
|
1641
|
+
start = timer()
|
|
1642
|
+
self._table_transformer_job(zoomin, auto_rotate=auto_rotate_tables)
|
|
1643
|
+
if callback:
|
|
1644
|
+
callback(0.83, "Table analysis ({:.2f}s)".format(timer() - start))
|
|
1645
|
+
|
|
1646
|
+
start = timer()
|
|
1647
|
+
self._text_merge()
|
|
1648
|
+
self._concat_downward()
|
|
1649
|
+
self._naive_vertical_merge(zoomin)
|
|
1650
|
+
if callback:
|
|
1651
|
+
callback(0.92, "Text merged ({:.2f}s)".format(timer() - start))
|
|
1652
|
+
|
|
1653
|
+
start = timer()
|
|
1654
|
+
tbls, figs = self._extract_table_figure(True, zoomin, True, True, True)
|
|
1655
|
+
|
|
1656
|
+
def insert_table_figures(tbls_or_figs, layout_type):
|
|
1657
|
+
def min_rectangle_distance(rect1, rect2):
|
|
1658
|
+
pn1, left1, right1, top1, bottom1 = rect1
|
|
1659
|
+
pn2, left2, right2, top2, bottom2 = rect2
|
|
1660
|
+
if right1 >= left2 and right2 >= left1 and bottom1 >= top2 and bottom2 >= top1:
|
|
1661
|
+
return 0
|
|
1662
|
+
if right1 < left2:
|
|
1663
|
+
dx = left2 - right1
|
|
1664
|
+
elif right2 < left1:
|
|
1665
|
+
dx = left1 - right2
|
|
1666
|
+
else:
|
|
1667
|
+
dx = 0
|
|
1668
|
+
if bottom1 < top2:
|
|
1669
|
+
dy = top2 - bottom1
|
|
1670
|
+
elif bottom2 < top1:
|
|
1671
|
+
dy = top1 - bottom2
|
|
1672
|
+
else:
|
|
1673
|
+
dy = 0
|
|
1674
|
+
return math.sqrt(dx * dx + dy * dy)
|
|
1675
|
+
|
|
1676
|
+
for (img, txt), poss in tbls_or_figs:
|
|
1677
|
+
local_poss = []
|
|
1678
|
+
for pn, left, right, top, bott in poss:
|
|
1679
|
+
local_pn = pn - self.page_from
|
|
1680
|
+
if 0 <= local_pn < len(self.page_cum_height) - 1:
|
|
1681
|
+
local_poss.append((local_pn, left, right, top, bott))
|
|
1682
|
+
else:
|
|
1683
|
+
logging.debug(f"Skip out-of-range table/figure position pn={pn}, page_from={self.page_from}")
|
|
1684
|
+
if not local_poss:
|
|
1685
|
+
logging.debug("No valid local positions for table/figure; skip insertion.")
|
|
1686
|
+
continue
|
|
1687
|
+
|
|
1688
|
+
if isinstance(txt, list):
|
|
1689
|
+
txt = "\n".join(txt)
|
|
1690
|
+
pn, left, right, top, bott = local_poss[0]
|
|
1691
|
+
insert_at = len(self.boxes)
|
|
1692
|
+
bboxes = [(i, (b["page_number"], b["x0"], b["x1"], b["top"], b["bottom"])) for i, b in enumerate(self.boxes)]
|
|
1693
|
+
if bboxes:
|
|
1694
|
+
dists = [
|
|
1695
|
+
(min_rectangle_distance((cand_pn, cand_left, cand_right, cand_top + self.page_cum_height[cand_pn], cand_bott + self.page_cum_height[cand_pn]), rect), i)
|
|
1696
|
+
for i, rect in bboxes
|
|
1697
|
+
for cand_pn, cand_left, cand_right, cand_top, cand_bott in local_poss
|
|
1698
|
+
]
|
|
1699
|
+
if dists:
|
|
1700
|
+
nearest_bbox_idx = int(np.argmin([dist for dist, _ in dists]))
|
|
1701
|
+
insert_at, _ = bboxes[dists[nearest_bbox_idx][-1]]
|
|
1702
|
+
if self.boxes[insert_at]["bottom"] < top + self.page_cum_height[pn]:
|
|
1703
|
+
insert_at += 1
|
|
1704
|
+
else:
|
|
1705
|
+
logging.debug("No text boxes available; append %s block directly.", layout_type)
|
|
1706
|
+
self.boxes.insert(
|
|
1707
|
+
insert_at,
|
|
1708
|
+
{
|
|
1709
|
+
"page_number": pn + 1,
|
|
1710
|
+
"x0": left,
|
|
1711
|
+
"x1": right,
|
|
1712
|
+
"top": top + self.page_cum_height[pn],
|
|
1713
|
+
"bottom": bott + self.page_cum_height[pn],
|
|
1714
|
+
"layout_type": layout_type,
|
|
1715
|
+
"text": txt,
|
|
1716
|
+
"image": img,
|
|
1717
|
+
"positions": [[pn + 1, int(left), int(right), int(top), int(bott)]],
|
|
1718
|
+
},
|
|
1719
|
+
)
|
|
1720
|
+
|
|
1721
|
+
for b in self.boxes:
|
|
1722
|
+
b["position_tag"] = self._line_tag(b, zoomin)
|
|
1723
|
+
b["image"] = self.crop(b["position_tag"], zoomin)
|
|
1724
|
+
b["positions"] = [[pos[0][-1] + 1, *pos[1:]] for pos in RAGFlowPdfParser.extract_positions(b["position_tag"])]
|
|
1725
|
+
|
|
1726
|
+
insert_table_figures(tbls, "table")
|
|
1727
|
+
insert_table_figures(figs, "figure")
|
|
1728
|
+
if callback:
|
|
1729
|
+
callback(1, "Structured ({:.2f}s)".format(timer() - start))
|
|
1730
|
+
return deepcopy(self.boxes)
|
|
1731
|
+
|
|
1732
|
+
@staticmethod
|
|
1733
|
+
def _offset_position_tag(text, page_offset):
|
|
1734
|
+
if not text or page_offset <= 0:
|
|
1735
|
+
return text
|
|
1736
|
+
|
|
1737
|
+
def _replace(match):
|
|
1738
|
+
pages = [str(int(p) + page_offset) for p in match.group(1).split("-")]
|
|
1739
|
+
return f"@@{'-'.join(pages)}\t"
|
|
1740
|
+
|
|
1741
|
+
return re.sub(r"@@([0-9-]+)\t", _replace, text)
|
|
1742
|
+
|
|
1743
|
+
def _to_global_boxes(self, boxes):
|
|
1744
|
+
if self.page_from <= 0:
|
|
1745
|
+
return boxes
|
|
1746
|
+
|
|
1747
|
+
for box in boxes:
|
|
1748
|
+
box["page_number"] = int(box.get("page_number", 1)) + self.page_from
|
|
1749
|
+
if isinstance(box.get("position_tag"), str):
|
|
1750
|
+
box["position_tag"] = self._offset_position_tag(box["position_tag"], self.page_from)
|
|
1751
|
+
if isinstance(box.get("positions"), list):
|
|
1752
|
+
box["positions"] = [[int(pos[0]) + self.page_from, *pos[1:]] if isinstance(pos, list) and len(pos) > 0 and isinstance(pos[0], (int, float)) else pos for pos in box["positions"]]
|
|
1753
|
+
return boxes
|
|
1754
|
+
|
|
1755
|
+
@staticmethod
|
|
1756
|
+
def remove_tag(txt):
|
|
1757
|
+
return re.sub(r"@@[\t0-9.-]+?##", "", txt)
|
|
1758
|
+
|
|
1759
|
+
@staticmethod
|
|
1760
|
+
def extract_positions(txt):
|
|
1761
|
+
poss = []
|
|
1762
|
+
for tag in re.findall(r"@@[0-9-]+\t[0-9.\t]+##", txt):
|
|
1763
|
+
pn, left, right, top, bottom = tag.strip("#").strip("@").split("\t")
|
|
1764
|
+
left, right, top, bottom = float(left), float(right), float(top), float(bottom)
|
|
1765
|
+
poss.append(([int(p) - 1 for p in pn.split("-")], left, right, top, bottom))
|
|
1766
|
+
return poss
|
|
1767
|
+
|
|
1768
|
+
def crop(self, text, ZM=3, need_position=False):
|
|
1769
|
+
imgs = []
|
|
1770
|
+
poss = self.extract_positions(text)
|
|
1771
|
+
if not poss:
|
|
1772
|
+
if need_position:
|
|
1773
|
+
return None, None
|
|
1774
|
+
return
|
|
1775
|
+
|
|
1776
|
+
if not getattr(self, "page_images", None):
|
|
1777
|
+
logging.warning("crop called without page images; skipping image generation.")
|
|
1778
|
+
if need_position:
|
|
1779
|
+
return None, None
|
|
1780
|
+
return
|
|
1781
|
+
|
|
1782
|
+
page_count = len(self.page_images)
|
|
1783
|
+
|
|
1784
|
+
filtered_poss = []
|
|
1785
|
+
for pns, left, right, top, bottom in poss:
|
|
1786
|
+
if not pns:
|
|
1787
|
+
logging.warning("Empty page index list in crop; skipping this position.")
|
|
1788
|
+
continue
|
|
1789
|
+
valid_pns = [p for p in pns if 0 <= p < page_count]
|
|
1790
|
+
if not valid_pns:
|
|
1791
|
+
logging.warning(f"All page indices {pns} out of range for {page_count} pages; skipping.")
|
|
1792
|
+
continue
|
|
1793
|
+
filtered_poss.append((valid_pns, left, right, top, bottom))
|
|
1794
|
+
|
|
1795
|
+
poss = filtered_poss
|
|
1796
|
+
if not poss:
|
|
1797
|
+
logging.warning("No valid positions after filtering; skip cropping.")
|
|
1798
|
+
if need_position:
|
|
1799
|
+
return None, None
|
|
1800
|
+
return
|
|
1801
|
+
|
|
1802
|
+
max_width = max(np.max([right - left for (_, left, right, _, _) in poss]), 6)
|
|
1803
|
+
GAP = 6
|
|
1804
|
+
pos = poss[0]
|
|
1805
|
+
first_page_idx = pos[0][0]
|
|
1806
|
+
poss.insert(0, ([first_page_idx], pos[1], pos[2], max(0, pos[3] - 120), max(pos[3] - GAP, 0)))
|
|
1807
|
+
pos = poss[-1]
|
|
1808
|
+
last_page_idx = pos[0][-1]
|
|
1809
|
+
if not (0 <= last_page_idx < page_count):
|
|
1810
|
+
logging.warning(f"Last page index {last_page_idx} out of range for {page_count} pages; skipping crop.")
|
|
1811
|
+
if need_position:
|
|
1812
|
+
return None, None
|
|
1813
|
+
return
|
|
1814
|
+
last_page_height = self.page_images[last_page_idx].size[1] / ZM
|
|
1815
|
+
poss.append(
|
|
1816
|
+
(
|
|
1817
|
+
[last_page_idx],
|
|
1818
|
+
pos[1],
|
|
1819
|
+
pos[2],
|
|
1820
|
+
min(last_page_height, pos[4] + GAP),
|
|
1821
|
+
min(last_page_height, pos[4] + 120),
|
|
1822
|
+
)
|
|
1823
|
+
)
|
|
1824
|
+
|
|
1825
|
+
positions = []
|
|
1826
|
+
for ii, (pns, left, right, top, bottom) in enumerate(poss):
|
|
1827
|
+
if 0 < ii < len(poss) - 1:
|
|
1828
|
+
right = max(left + 10, right)
|
|
1829
|
+
else:
|
|
1830
|
+
right = left + max_width
|
|
1831
|
+
bottom *= ZM
|
|
1832
|
+
for pn in pns[1:]:
|
|
1833
|
+
if 0 <= pn - 1 < page_count:
|
|
1834
|
+
bottom += self.page_images[pn - 1].size[1]
|
|
1835
|
+
else:
|
|
1836
|
+
logging.warning(f"Page index {pn}-1 out of range for {page_count} pages during crop; skipping height accumulation.")
|
|
1837
|
+
|
|
1838
|
+
if not (0 <= pns[0] < page_count):
|
|
1839
|
+
logging.warning(f"Base page index {pns[0]} out of range for {page_count} pages during crop; skipping this segment.")
|
|
1840
|
+
continue
|
|
1841
|
+
|
|
1842
|
+
imgs.append(self.page_images[pns[0]].crop((left * ZM, top * ZM, right * ZM, min(bottom, self.page_images[pns[0]].size[1]))))
|
|
1843
|
+
if 0 < ii < len(poss) - 1:
|
|
1844
|
+
positions.append((pns[0] + self.page_from, left, right, top, min(bottom, self.page_images[pns[0]].size[1]) / ZM))
|
|
1845
|
+
bottom -= self.page_images[pns[0]].size[1]
|
|
1846
|
+
for pn in pns[1:]:
|
|
1847
|
+
if not (0 <= pn < page_count):
|
|
1848
|
+
logging.warning(f"Page index {pn} out of range for {page_count} pages during crop; skipping this page.")
|
|
1849
|
+
continue
|
|
1850
|
+
imgs.append(self.page_images[pn].crop((left * ZM, 0, right * ZM, min(bottom, self.page_images[pn].size[1]))))
|
|
1851
|
+
if 0 < ii < len(poss) - 1:
|
|
1852
|
+
positions.append((pn + self.page_from, left, right, 0, min(bottom, self.page_images[pn].size[1]) / ZM))
|
|
1853
|
+
bottom -= self.page_images[pn].size[1]
|
|
1854
|
+
|
|
1855
|
+
if not imgs:
|
|
1856
|
+
if need_position:
|
|
1857
|
+
return None, None
|
|
1858
|
+
return
|
|
1859
|
+
height = 0
|
|
1860
|
+
for img in imgs:
|
|
1861
|
+
height += img.size[1] + GAP
|
|
1862
|
+
height = int(height)
|
|
1863
|
+
width = int(np.max([i.size[0] for i in imgs]))
|
|
1864
|
+
pic = Image.new("RGB", (width, height), (245, 245, 245))
|
|
1865
|
+
height = 0
|
|
1866
|
+
for ii, img in enumerate(imgs):
|
|
1867
|
+
if ii == 0 or ii + 1 == len(imgs):
|
|
1868
|
+
img = img.convert("RGBA")
|
|
1869
|
+
overlay = Image.new("RGBA", img.size, (0, 0, 0, 0))
|
|
1870
|
+
overlay.putalpha(128)
|
|
1871
|
+
img = Image.alpha_composite(img, overlay).convert("RGB")
|
|
1872
|
+
pic.paste(img, (0, int(height)))
|
|
1873
|
+
height += img.size[1] + GAP
|
|
1874
|
+
|
|
1875
|
+
if need_position:
|
|
1876
|
+
return pic, positions
|
|
1877
|
+
return pic
|
|
1878
|
+
|
|
1879
|
+
def get_position(self, bx, ZM):
|
|
1880
|
+
poss = []
|
|
1881
|
+
pn = bx["page_number"]
|
|
1882
|
+
top = bx["top"] - self.page_cum_height[pn - 1]
|
|
1883
|
+
bott = bx["bottom"] - self.page_cum_height[pn - 1]
|
|
1884
|
+
poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
|
|
1885
|
+
while bott * ZM > self.page_images[pn - 1].size[1]:
|
|
1886
|
+
bott -= self.page_images[pn - 1].size[1] / ZM
|
|
1887
|
+
top = 0
|
|
1888
|
+
pn += 1
|
|
1889
|
+
poss.append((pn, bx["x0"], bx["x1"], top, min(bott, self.page_images[pn - 1].size[1] / ZM)))
|
|
1890
|
+
return poss
|
|
1891
|
+
|
|
1892
|
+
|
|
1893
|
+
if __name__ == "__main__":
|
|
1894
|
+
pass
|