langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,559 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
#
|
|
16
|
+
import logging
|
|
17
|
+
import re
|
|
18
|
+
from collections import Counter
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
from .model_loader import default_model_dir
|
|
23
|
+
from .tokenizer import tag, tokenize
|
|
24
|
+
|
|
25
|
+
from .recognizer import Recognizer
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class TableStructureRecognizer(Recognizer):
|
|
29
|
+
labels = [
|
|
30
|
+
"table",
|
|
31
|
+
"table column",
|
|
32
|
+
"table row",
|
|
33
|
+
"table column header",
|
|
34
|
+
"table projected row header",
|
|
35
|
+
"table spanning cell",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
def __init__(self, model_dir=None):
|
|
39
|
+
model_dir = model_dir or str(default_model_dir())
|
|
40
|
+
super().__init__(self.labels, "tsr", model_dir)
|
|
41
|
+
|
|
42
|
+
def __call__(self, images, thr=0.2):
|
|
43
|
+
tbls = super().__call__(images, thr)
|
|
44
|
+
|
|
45
|
+
res = []
|
|
46
|
+
# align left&right for rows, align top&bottom for columns
|
|
47
|
+
for tbl in tbls:
|
|
48
|
+
lts = [
|
|
49
|
+
{
|
|
50
|
+
"label": b["type"],
|
|
51
|
+
"score": b["score"],
|
|
52
|
+
"x0": b["bbox"][0],
|
|
53
|
+
"x1": b["bbox"][2],
|
|
54
|
+
"top": b["bbox"][1],
|
|
55
|
+
"bottom": b["bbox"][-1],
|
|
56
|
+
}
|
|
57
|
+
for b in tbl
|
|
58
|
+
]
|
|
59
|
+
if not lts:
|
|
60
|
+
continue
|
|
61
|
+
|
|
62
|
+
left = [b["x0"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
|
|
63
|
+
right = [b["x1"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
|
|
64
|
+
if not left:
|
|
65
|
+
continue
|
|
66
|
+
left = np.mean(left) if len(left) > 4 else np.min(left)
|
|
67
|
+
right = np.mean(right) if len(right) > 4 else np.max(right)
|
|
68
|
+
for b in lts:
|
|
69
|
+
if b["label"].find("row") > 0 or b["label"].find("header") > 0:
|
|
70
|
+
if b["x0"] > left:
|
|
71
|
+
b["x0"] = left
|
|
72
|
+
if b["x1"] < right:
|
|
73
|
+
b["x1"] = right
|
|
74
|
+
|
|
75
|
+
top = [b["top"] for b in lts if b["label"] == "table column"]
|
|
76
|
+
bottom = [b["bottom"] for b in lts if b["label"] == "table column"]
|
|
77
|
+
if not top:
|
|
78
|
+
res.append(lts)
|
|
79
|
+
continue
|
|
80
|
+
top = np.median(top) if len(top) > 4 else np.min(top)
|
|
81
|
+
bottom = np.median(bottom) if len(bottom) > 4 else np.max(bottom)
|
|
82
|
+
for b in lts:
|
|
83
|
+
if b["label"] == "table column":
|
|
84
|
+
if b["top"] > top:
|
|
85
|
+
b["top"] = top
|
|
86
|
+
if b["bottom"] < bottom:
|
|
87
|
+
b["bottom"] = bottom
|
|
88
|
+
|
|
89
|
+
res.append(lts)
|
|
90
|
+
return res
|
|
91
|
+
|
|
92
|
+
@staticmethod
|
|
93
|
+
def is_caption(bx):
|
|
94
|
+
patt = [
|
|
95
|
+
r"[图表]+[ 0-9::]{2,}",
|
|
96
|
+
r"(?i)Fig\.?\s*\d+",
|
|
97
|
+
r"(?i)Figure\s+\d+",
|
|
98
|
+
r"(?i)Table\s+\d+",
|
|
99
|
+
]
|
|
100
|
+
if any([re.match(p, bx["text"].strip()) for p in patt]) or bx.get("layout_type", "").find("caption") >= 0:
|
|
101
|
+
return True
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
@staticmethod
|
|
105
|
+
def blockType(b):
|
|
106
|
+
patt = [
|
|
107
|
+
("^(20|19)[0-9]{2}[年/-][0-9]{1,2}[月/-][0-9]{1,2}日*$", "Dt"),
|
|
108
|
+
(r"^(20|19)[0-9]{2}年$", "Dt"),
|
|
109
|
+
(r"^(20|19)[0-9]{2}[年-][0-9]{1,2}月*$", "Dt"),
|
|
110
|
+
("^[0-9]{1,2}[月-][0-9]{1,2}日*$", "Dt"),
|
|
111
|
+
(r"^第*[一二三四1-4]季度$", "Dt"),
|
|
112
|
+
(r"^(20|19)[0-9]{2}年*[一二三四1-4]季度$", "Dt"),
|
|
113
|
+
(r"^(20|19)[0-9]{2}[ABCDE]$", "Dt"),
|
|
114
|
+
("^[0-9.,+%/ -]+$", "Nu"),
|
|
115
|
+
(r"^[0-9A-Z/\._~-]+$", "Ca"),
|
|
116
|
+
(r"^[A-Z]*[a-z' -]+$", "En"),
|
|
117
|
+
(r"^[0-9.,+-]+[0-9A-Za-z/$¥%<>()()' -]+$", "NE"),
|
|
118
|
+
(r"^.{1}$", "Sg"),
|
|
119
|
+
]
|
|
120
|
+
for p, n in patt:
|
|
121
|
+
if re.search(p, b["text"].strip()):
|
|
122
|
+
return n
|
|
123
|
+
tks = [t for t in tokenize(b["text"]).split() if len(t) > 1]
|
|
124
|
+
if len(tks) > 3:
|
|
125
|
+
if len(tks) < 12:
|
|
126
|
+
return "Tx"
|
|
127
|
+
else:
|
|
128
|
+
return "Lx"
|
|
129
|
+
|
|
130
|
+
if len(tks) == 1 and tag(tks[0]) == "nr":
|
|
131
|
+
return "Nr"
|
|
132
|
+
|
|
133
|
+
return "Ot"
|
|
134
|
+
|
|
135
|
+
@staticmethod
|
|
136
|
+
def construct_table(boxes, is_english=False, html=True, **kwargs):
|
|
137
|
+
cap = ""
|
|
138
|
+
i = 0
|
|
139
|
+
while i < len(boxes):
|
|
140
|
+
if TableStructureRecognizer.is_caption(boxes[i]):
|
|
141
|
+
if is_english:
|
|
142
|
+
cap += " "
|
|
143
|
+
cap += boxes[i]["text"]
|
|
144
|
+
boxes.pop(i)
|
|
145
|
+
i -= 1
|
|
146
|
+
i += 1
|
|
147
|
+
|
|
148
|
+
if not boxes:
|
|
149
|
+
return []
|
|
150
|
+
for b in boxes:
|
|
151
|
+
b["btype"] = TableStructureRecognizer.blockType(b)
|
|
152
|
+
max_type = Counter([b["btype"] for b in boxes]).items()
|
|
153
|
+
max_type = max(max_type, key=lambda x: x[1])[0] if max_type else ""
|
|
154
|
+
logging.debug("MAXTYPE: " + max_type)
|
|
155
|
+
|
|
156
|
+
rowh = [b["R_bott"] - b["R_top"] for b in boxes if "R" in b]
|
|
157
|
+
rowh = np.min(rowh) if rowh else 0
|
|
158
|
+
boxes = Recognizer.sort_R_firstly(boxes, rowh / 2)
|
|
159
|
+
# for b in boxes:print(b)
|
|
160
|
+
boxes[0]["rn"] = 0
|
|
161
|
+
rows = [[boxes[0]]]
|
|
162
|
+
btm = boxes[0]["bottom"]
|
|
163
|
+
for b in boxes[1:]:
|
|
164
|
+
b["rn"] = len(rows) - 1
|
|
165
|
+
lst_r = rows[-1]
|
|
166
|
+
if lst_r[-1].get("R", "") != b.get("R", "") or (b["top"] >= btm - 3 and lst_r[-1].get("R", "-1") != b.get("R", "-2")): # new row
|
|
167
|
+
btm = b["bottom"]
|
|
168
|
+
b["rn"] += 1
|
|
169
|
+
rows.append([b])
|
|
170
|
+
continue
|
|
171
|
+
btm = (btm + b["bottom"]) / 2.0
|
|
172
|
+
rows[-1].append(b)
|
|
173
|
+
|
|
174
|
+
colwm = [b["C_right"] - b["C_left"] for b in boxes if "C" in b]
|
|
175
|
+
colwm = np.min(colwm) if colwm else 0
|
|
176
|
+
crosspage = len(set([b["page_number"] for b in boxes])) > 1
|
|
177
|
+
if crosspage:
|
|
178
|
+
boxes = Recognizer.sort_X_firstly(boxes, colwm / 2)
|
|
179
|
+
else:
|
|
180
|
+
boxes = Recognizer.sort_C_firstly(boxes, colwm / 2)
|
|
181
|
+
boxes[0]["cn"] = 0
|
|
182
|
+
cols = [[boxes[0]]]
|
|
183
|
+
right = boxes[0]["x1"]
|
|
184
|
+
for b in boxes[1:]:
|
|
185
|
+
b["cn"] = len(cols) - 1
|
|
186
|
+
lst_c = cols[-1]
|
|
187
|
+
if (int(b.get("C", "1")) - int(lst_c[-1].get("C", "1")) == 1 and b["page_number"] == lst_c[-1]["page_number"]) or (
|
|
188
|
+
b["x0"] >= right and lst_c[-1].get("C", "-1") != b.get("C", "-2")
|
|
189
|
+
): # new col
|
|
190
|
+
right = b["x1"]
|
|
191
|
+
b["cn"] += 1
|
|
192
|
+
cols.append([b])
|
|
193
|
+
continue
|
|
194
|
+
right = (right + b["x1"]) / 2.0
|
|
195
|
+
cols[-1].append(b)
|
|
196
|
+
|
|
197
|
+
tbl = [[[] for _ in range(len(cols))] for _ in range(len(rows))]
|
|
198
|
+
for b in boxes:
|
|
199
|
+
tbl[b["rn"]][b["cn"]].append(b)
|
|
200
|
+
|
|
201
|
+
if len(rows) >= 4:
|
|
202
|
+
# remove single in column
|
|
203
|
+
j = 0
|
|
204
|
+
while j < len(tbl[0]):
|
|
205
|
+
e, ii = 0, 0
|
|
206
|
+
for i in range(len(tbl)):
|
|
207
|
+
if tbl[i][j]:
|
|
208
|
+
e += 1
|
|
209
|
+
ii = i
|
|
210
|
+
if e > 1:
|
|
211
|
+
break
|
|
212
|
+
if e > 1:
|
|
213
|
+
j += 1
|
|
214
|
+
continue
|
|
215
|
+
f = (j > 0 and tbl[ii][j - 1] and tbl[ii][j - 1][0].get("text")) or j == 0
|
|
216
|
+
ff = (j + 1 < len(tbl[ii]) and tbl[ii][j + 1] and tbl[ii][j + 1][0].get("text")) or j + 1 >= len(tbl[ii])
|
|
217
|
+
if f and ff:
|
|
218
|
+
j += 1
|
|
219
|
+
continue
|
|
220
|
+
bx = tbl[ii][j][0]
|
|
221
|
+
logging.debug("Relocate column single: " + bx["text"])
|
|
222
|
+
# j column only has one value
|
|
223
|
+
left, right = 100000, 100000
|
|
224
|
+
if j > 0 and not f:
|
|
225
|
+
for i in range(len(tbl)):
|
|
226
|
+
if tbl[i][j - 1]:
|
|
227
|
+
left = min(left, np.min([bx["x0"] - a["x1"] for a in tbl[i][j - 1]]))
|
|
228
|
+
if j + 1 < len(tbl[0]) and not ff:
|
|
229
|
+
for i in range(len(tbl)):
|
|
230
|
+
if tbl[i][j + 1]:
|
|
231
|
+
right = min(right, np.min([a["x0"] - bx["x1"] for a in tbl[i][j + 1]]))
|
|
232
|
+
assert left < 100000 or right < 100000
|
|
233
|
+
if left < right:
|
|
234
|
+
for jj in range(j, len(tbl[0])):
|
|
235
|
+
for i in range(len(tbl)):
|
|
236
|
+
for a in tbl[i][jj]:
|
|
237
|
+
a["cn"] -= 1
|
|
238
|
+
if tbl[ii][j - 1]:
|
|
239
|
+
tbl[ii][j - 1].extend(tbl[ii][j])
|
|
240
|
+
else:
|
|
241
|
+
tbl[ii][j - 1] = tbl[ii][j]
|
|
242
|
+
for i in range(len(tbl)):
|
|
243
|
+
tbl[i].pop(j)
|
|
244
|
+
|
|
245
|
+
else:
|
|
246
|
+
for jj in range(j + 1, len(tbl[0])):
|
|
247
|
+
for i in range(len(tbl)):
|
|
248
|
+
for a in tbl[i][jj]:
|
|
249
|
+
a["cn"] -= 1
|
|
250
|
+
if tbl[ii][j + 1]:
|
|
251
|
+
tbl[ii][j + 1].extend(tbl[ii][j])
|
|
252
|
+
else:
|
|
253
|
+
tbl[ii][j + 1] = tbl[ii][j]
|
|
254
|
+
for i in range(len(tbl)):
|
|
255
|
+
tbl[i].pop(j)
|
|
256
|
+
cols.pop(j)
|
|
257
|
+
assert len(cols) == len(tbl[0]), "Column NO. miss matched: %d vs %d" % (len(cols), len(tbl[0]))
|
|
258
|
+
|
|
259
|
+
if len(cols) >= 4:
|
|
260
|
+
# remove single in row
|
|
261
|
+
i = 0
|
|
262
|
+
while i < len(tbl):
|
|
263
|
+
e, jj = 0, 0
|
|
264
|
+
for j in range(len(tbl[i])):
|
|
265
|
+
if tbl[i][j]:
|
|
266
|
+
e += 1
|
|
267
|
+
jj = j
|
|
268
|
+
if e > 1:
|
|
269
|
+
break
|
|
270
|
+
if e > 1:
|
|
271
|
+
i += 1
|
|
272
|
+
continue
|
|
273
|
+
f = (i > 0 and tbl[i - 1][jj] and tbl[i - 1][jj][0].get("text")) or i == 0
|
|
274
|
+
ff = (i + 1 < len(tbl) and tbl[i + 1][jj] and tbl[i + 1][jj][0].get("text")) or i + 1 >= len(tbl)
|
|
275
|
+
if f and ff:
|
|
276
|
+
i += 1
|
|
277
|
+
continue
|
|
278
|
+
|
|
279
|
+
bx = tbl[i][jj][0]
|
|
280
|
+
logging.debug("Relocate row single: " + bx["text"])
|
|
281
|
+
# i row only has one value
|
|
282
|
+
up, down = 100000, 100000
|
|
283
|
+
if i > 0 and not f:
|
|
284
|
+
for j in range(len(tbl[i - 1])):
|
|
285
|
+
if tbl[i - 1][j]:
|
|
286
|
+
up = min(up, np.min([bx["top"] - a["bottom"] for a in tbl[i - 1][j]]))
|
|
287
|
+
if i + 1 < len(tbl) and not ff:
|
|
288
|
+
for j in range(len(tbl[i + 1])):
|
|
289
|
+
if tbl[i + 1][j]:
|
|
290
|
+
down = min(down, np.min([a["top"] - bx["bottom"] for a in tbl[i + 1][j]]))
|
|
291
|
+
assert up < 100000 or down < 100000
|
|
292
|
+
if up < down:
|
|
293
|
+
for ii in range(i, len(tbl)):
|
|
294
|
+
for j in range(len(tbl[ii])):
|
|
295
|
+
for a in tbl[ii][j]:
|
|
296
|
+
a["rn"] -= 1
|
|
297
|
+
if tbl[i - 1][jj]:
|
|
298
|
+
tbl[i - 1][jj].extend(tbl[i][jj])
|
|
299
|
+
else:
|
|
300
|
+
tbl[i - 1][jj] = tbl[i][jj]
|
|
301
|
+
tbl.pop(i)
|
|
302
|
+
|
|
303
|
+
else:
|
|
304
|
+
for ii in range(i + 1, len(tbl)):
|
|
305
|
+
for j in range(len(tbl[ii])):
|
|
306
|
+
for a in tbl[ii][j]:
|
|
307
|
+
a["rn"] -= 1
|
|
308
|
+
if tbl[i + 1][jj]:
|
|
309
|
+
tbl[i + 1][jj].extend(tbl[i][jj])
|
|
310
|
+
else:
|
|
311
|
+
tbl[i + 1][jj] = tbl[i][jj]
|
|
312
|
+
tbl.pop(i)
|
|
313
|
+
rows.pop(i)
|
|
314
|
+
|
|
315
|
+
# which rows are headers
|
|
316
|
+
hdset = set([])
|
|
317
|
+
for i in range(len(tbl)):
|
|
318
|
+
cnt, h = 0, 0
|
|
319
|
+
for j, arr in enumerate(tbl[i]):
|
|
320
|
+
if not arr:
|
|
321
|
+
continue
|
|
322
|
+
cnt += 1
|
|
323
|
+
if max_type == "Nu" and arr[0]["btype"] == "Nu":
|
|
324
|
+
continue
|
|
325
|
+
if any([a.get("H") for a in arr]) or (max_type == "Nu" and arr[0]["btype"] != "Nu"):
|
|
326
|
+
h += 1
|
|
327
|
+
if h / cnt > 0.5:
|
|
328
|
+
hdset.add(i)
|
|
329
|
+
|
|
330
|
+
if html:
|
|
331
|
+
return TableStructureRecognizer.__html_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, True))
|
|
332
|
+
|
|
333
|
+
return TableStructureRecognizer.__desc_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, False), is_english)
|
|
334
|
+
|
|
335
|
+
@staticmethod
|
|
336
|
+
def __html_table(cap, hdset, tbl):
|
|
337
|
+
# constrcut HTML
|
|
338
|
+
html = "<table>"
|
|
339
|
+
if cap:
|
|
340
|
+
html += f"<caption>{cap}</caption>"
|
|
341
|
+
for i in range(len(tbl)):
|
|
342
|
+
row = "<tr>"
|
|
343
|
+
txts = []
|
|
344
|
+
for j, arr in enumerate(tbl[i]):
|
|
345
|
+
if arr is None:
|
|
346
|
+
continue
|
|
347
|
+
if not arr:
|
|
348
|
+
row += "<td></td>" if i not in hdset else "<th></th>"
|
|
349
|
+
continue
|
|
350
|
+
txt = ""
|
|
351
|
+
if arr:
|
|
352
|
+
h = min(np.min([c["bottom"] - c["top"] for c in arr]) / 2, 10)
|
|
353
|
+
txt = " ".join([c["text"] for c in Recognizer.sort_Y_firstly(arr, h)])
|
|
354
|
+
txts.append(txt)
|
|
355
|
+
sp = ""
|
|
356
|
+
if arr[0].get("colspan"):
|
|
357
|
+
sp = "colspan={}".format(arr[0]["colspan"])
|
|
358
|
+
if arr[0].get("rowspan"):
|
|
359
|
+
sp += " rowspan={}".format(arr[0]["rowspan"])
|
|
360
|
+
if i in hdset:
|
|
361
|
+
row += f"<th {sp} >" + txt + "</th>"
|
|
362
|
+
else:
|
|
363
|
+
row += f"<td {sp} >" + txt + "</td>"
|
|
364
|
+
|
|
365
|
+
if i in hdset:
|
|
366
|
+
if all([t in hdset for t in txts]):
|
|
367
|
+
continue
|
|
368
|
+
for t in txts:
|
|
369
|
+
hdset.add(t)
|
|
370
|
+
|
|
371
|
+
if row != "<tr>":
|
|
372
|
+
row += "</tr>"
|
|
373
|
+
else:
|
|
374
|
+
row = ""
|
|
375
|
+
html += "\n" + row
|
|
376
|
+
html += "\n</table>"
|
|
377
|
+
return html
|
|
378
|
+
|
|
379
|
+
@staticmethod
|
|
380
|
+
def __desc_table(cap, hdr_rowno, tbl, is_english):
|
|
381
|
+
# get text of every column in header row to become header text
|
|
382
|
+
clmno = len(tbl[0])
|
|
383
|
+
rowno = len(tbl)
|
|
384
|
+
headers = {}
|
|
385
|
+
hdrset = set()
|
|
386
|
+
lst_hdr = []
|
|
387
|
+
de = "的" if not is_english else " for "
|
|
388
|
+
for r in sorted(list(hdr_rowno)):
|
|
389
|
+
headers[r] = ["" for _ in range(clmno)]
|
|
390
|
+
for i in range(clmno):
|
|
391
|
+
if not tbl[r][i]:
|
|
392
|
+
continue
|
|
393
|
+
txt = " ".join([a["text"].strip() for a in tbl[r][i]])
|
|
394
|
+
headers[r][i] = txt
|
|
395
|
+
hdrset.add(txt)
|
|
396
|
+
if all([not t for t in headers[r]]):
|
|
397
|
+
del headers[r]
|
|
398
|
+
hdr_rowno.remove(r)
|
|
399
|
+
continue
|
|
400
|
+
for j in range(clmno):
|
|
401
|
+
if headers[r][j]:
|
|
402
|
+
continue
|
|
403
|
+
if j >= len(lst_hdr):
|
|
404
|
+
break
|
|
405
|
+
headers[r][j] = lst_hdr[j]
|
|
406
|
+
lst_hdr = headers[r]
|
|
407
|
+
for i in range(rowno):
|
|
408
|
+
if i not in hdr_rowno:
|
|
409
|
+
continue
|
|
410
|
+
for j in range(i + 1, rowno):
|
|
411
|
+
if j not in hdr_rowno:
|
|
412
|
+
break
|
|
413
|
+
for k in range(clmno):
|
|
414
|
+
if not headers[j - 1][k]:
|
|
415
|
+
continue
|
|
416
|
+
if headers[j][k].find(headers[j - 1][k]) >= 0:
|
|
417
|
+
continue
|
|
418
|
+
if len(headers[j][k]) > len(headers[j - 1][k]):
|
|
419
|
+
headers[j][k] += (de if headers[j][k] else "") + headers[j - 1][k]
|
|
420
|
+
else:
|
|
421
|
+
headers[j][k] = headers[j - 1][k] + (de if headers[j - 1][k] else "") + headers[j][k]
|
|
422
|
+
|
|
423
|
+
logging.debug(f">>>>>>>>>>>>>>>>>{cap}:SIZE:{rowno}X{clmno} Header: {hdr_rowno}")
|
|
424
|
+
row_txt = []
|
|
425
|
+
for i in range(rowno):
|
|
426
|
+
if i in hdr_rowno:
|
|
427
|
+
continue
|
|
428
|
+
rtxt = []
|
|
429
|
+
|
|
430
|
+
def append(delimer):
|
|
431
|
+
nonlocal rtxt, row_txt
|
|
432
|
+
rtxt = delimer.join(rtxt)
|
|
433
|
+
if row_txt and len(row_txt[-1]) + len(rtxt) < 64:
|
|
434
|
+
row_txt[-1] += "\n" + rtxt
|
|
435
|
+
else:
|
|
436
|
+
row_txt.append(rtxt)
|
|
437
|
+
|
|
438
|
+
r = 0
|
|
439
|
+
if len(headers.items()):
|
|
440
|
+
_arr = [(i - r, r) for r, _ in headers.items() if r < i]
|
|
441
|
+
if _arr:
|
|
442
|
+
_, r = min(_arr, key=lambda x: x[0])
|
|
443
|
+
|
|
444
|
+
if r not in headers and clmno <= 2:
|
|
445
|
+
for j in range(clmno):
|
|
446
|
+
if not tbl[i][j]:
|
|
447
|
+
continue
|
|
448
|
+
txt = "".join([a["text"].strip() for a in tbl[i][j]])
|
|
449
|
+
if txt:
|
|
450
|
+
rtxt.append(txt)
|
|
451
|
+
if rtxt:
|
|
452
|
+
append(":")
|
|
453
|
+
continue
|
|
454
|
+
|
|
455
|
+
for j in range(clmno):
|
|
456
|
+
if not tbl[i][j]:
|
|
457
|
+
continue
|
|
458
|
+
txt = "".join([a["text"].strip() for a in tbl[i][j]])
|
|
459
|
+
if not txt:
|
|
460
|
+
continue
|
|
461
|
+
ctt = headers[r][j] if r in headers else ""
|
|
462
|
+
if ctt:
|
|
463
|
+
ctt += ":"
|
|
464
|
+
ctt += txt
|
|
465
|
+
if ctt:
|
|
466
|
+
rtxt.append(ctt)
|
|
467
|
+
|
|
468
|
+
if rtxt:
|
|
469
|
+
row_txt.append("; ".join(rtxt))
|
|
470
|
+
|
|
471
|
+
if cap:
|
|
472
|
+
if is_english:
|
|
473
|
+
from_ = " in "
|
|
474
|
+
else:
|
|
475
|
+
from_ = "来自"
|
|
476
|
+
row_txt = [t + f"\t——{from_}“{cap}”" for t in row_txt]
|
|
477
|
+
return row_txt
|
|
478
|
+
|
|
479
|
+
@staticmethod
|
|
480
|
+
def __cal_spans(boxes, rows, cols, tbl, html=True):
|
|
481
|
+
# caculate span
|
|
482
|
+
clft = [np.mean([c.get("C_left", c["x0"]) for c in cln]) for cln in cols]
|
|
483
|
+
crgt = [np.mean([c.get("C_right", c["x1"]) for c in cln]) for cln in cols]
|
|
484
|
+
rtop = [np.mean([c.get("R_top", c["top"]) for c in row]) for row in rows]
|
|
485
|
+
rbtm = [np.mean([c.get("R_btm", c["bottom"]) for c in row]) for row in rows]
|
|
486
|
+
for b in boxes:
|
|
487
|
+
if "SP" not in b:
|
|
488
|
+
continue
|
|
489
|
+
b["colspan"] = [b["cn"]]
|
|
490
|
+
b["rowspan"] = [b["rn"]]
|
|
491
|
+
# col span
|
|
492
|
+
for j in range(0, len(clft)):
|
|
493
|
+
if j == b["cn"]:
|
|
494
|
+
continue
|
|
495
|
+
if clft[j] + (crgt[j] - clft[j]) / 2 < b["H_left"]:
|
|
496
|
+
continue
|
|
497
|
+
if crgt[j] - (crgt[j] - clft[j]) / 2 > b["H_right"]:
|
|
498
|
+
continue
|
|
499
|
+
b["colspan"].append(j)
|
|
500
|
+
# row span
|
|
501
|
+
for j in range(0, len(rtop)):
|
|
502
|
+
if j == b["rn"]:
|
|
503
|
+
continue
|
|
504
|
+
if rtop[j] + (rbtm[j] - rtop[j]) / 2 < b["H_top"]:
|
|
505
|
+
continue
|
|
506
|
+
if rbtm[j] - (rbtm[j] - rtop[j]) / 2 > b["H_bott"]:
|
|
507
|
+
continue
|
|
508
|
+
b["rowspan"].append(j)
|
|
509
|
+
|
|
510
|
+
def join(arr):
|
|
511
|
+
if not arr:
|
|
512
|
+
return ""
|
|
513
|
+
return "".join([t["text"] for t in arr])
|
|
514
|
+
|
|
515
|
+
# rm the spaning cells
|
|
516
|
+
for i in range(len(tbl)):
|
|
517
|
+
for j, arr in enumerate(tbl[i]):
|
|
518
|
+
if not arr:
|
|
519
|
+
continue
|
|
520
|
+
if all(["rowspan" not in a and "colspan" not in a for a in arr]):
|
|
521
|
+
continue
|
|
522
|
+
rowspan, colspan = [], []
|
|
523
|
+
for a in arr:
|
|
524
|
+
if isinstance(a.get("rowspan", 0), list):
|
|
525
|
+
rowspan.extend(a["rowspan"])
|
|
526
|
+
if isinstance(a.get("colspan", 0), list):
|
|
527
|
+
colspan.extend(a["colspan"])
|
|
528
|
+
rowspan, colspan = set(rowspan), set(colspan)
|
|
529
|
+
if len(rowspan) < 2 and len(colspan) < 2:
|
|
530
|
+
for a in arr:
|
|
531
|
+
if "rowspan" in a:
|
|
532
|
+
del a["rowspan"]
|
|
533
|
+
if "colspan" in a:
|
|
534
|
+
del a["colspan"]
|
|
535
|
+
continue
|
|
536
|
+
rowspan, colspan = sorted(rowspan), sorted(colspan)
|
|
537
|
+
rowspan = list(range(rowspan[0], rowspan[-1] + 1))
|
|
538
|
+
colspan = list(range(colspan[0], colspan[-1] + 1))
|
|
539
|
+
assert i in rowspan, rowspan
|
|
540
|
+
assert j in colspan, colspan
|
|
541
|
+
arr = []
|
|
542
|
+
for r in rowspan:
|
|
543
|
+
for c in colspan:
|
|
544
|
+
arr_txt = join(arr)
|
|
545
|
+
if tbl[r][c] and join(tbl[r][c]) != arr_txt:
|
|
546
|
+
arr.extend(tbl[r][c])
|
|
547
|
+
tbl[r][c] = None if html else arr
|
|
548
|
+
for a in arr:
|
|
549
|
+
if len(rowspan) > 1:
|
|
550
|
+
a["rowspan"] = len(rowspan)
|
|
551
|
+
elif "rowspan" in a:
|
|
552
|
+
del a["rowspan"]
|
|
553
|
+
if len(colspan) > 1:
|
|
554
|
+
a["colspan"] = len(colspan)
|
|
555
|
+
elif "colspan" in a:
|
|
556
|
+
del a["colspan"]
|
|
557
|
+
tbl[rowspan[0]][colspan[0]] = arr
|
|
558
|
+
|
|
559
|
+
return tbl
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Lightweight stand-in for RAGFlow's rag_tokenizer (which itself wraps a
|
|
3
|
+
tokenizer bundled inside the infinity-sdk vector-DB client). deepdoc's live
|
|
4
|
+
call sites only need coarse signals -- "is this char CJK", "how many
|
|
5
|
+
word-tokens is this text", "is this single token a person name" -- for
|
|
6
|
+
table-cell type classification, not text reconstruction, so a real
|
|
7
|
+
segmenter (jieba) is enough; we don't need infinity-sdk's tokenizer.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import jieba
|
|
13
|
+
import jieba.posseg as jieba_posseg
|
|
14
|
+
|
|
15
|
+
_CJK_RANGE = ("一", "鿿")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def is_chinese(text: str) -> bool:
|
|
19
|
+
return bool(text) and any(_CJK_RANGE[0] <= ch <= _CJK_RANGE[1] for ch in text)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def tokenize(text: str) -> str:
|
|
23
|
+
return " ".join(jieba.cut(text))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def tag(token: str) -> str:
|
|
27
|
+
if not token:
|
|
28
|
+
return ""
|
|
29
|
+
words = list(jieba_posseg.cut(token))
|
|
30
|
+
return words[0].flag if words else ""
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
#
|
|
16
|
+
from io import BytesIO
|
|
17
|
+
|
|
18
|
+
from pypdf import PdfReader as pdf2_read
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def extract_pdf_outlines(source):
|
|
22
|
+
try:
|
|
23
|
+
with pdf2_read(source if isinstance(source, str) else BytesIO(source)) as pdf:
|
|
24
|
+
outlines = []
|
|
25
|
+
|
|
26
|
+
def dfs(nodes, depth):
|
|
27
|
+
for node in nodes:
|
|
28
|
+
if isinstance(node, list):
|
|
29
|
+
dfs(node, depth + 1)
|
|
30
|
+
else:
|
|
31
|
+
outlines.append((node["/Title"], depth, pdf.get_destination_page_number(node) + 1))
|
|
32
|
+
|
|
33
|
+
dfs(pdf.outline, 0)
|
|
34
|
+
return outlines
|
|
35
|
+
except Exception:
|
|
36
|
+
return []
|