langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,559 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+ import logging
17
+ import re
18
+ from collections import Counter
19
+
20
+ import numpy as np
21
+
22
+ from .model_loader import default_model_dir
23
+ from .tokenizer import tag, tokenize
24
+
25
+ from .recognizer import Recognizer
26
+
27
+
28
+ class TableStructureRecognizer(Recognizer):
29
+ labels = [
30
+ "table",
31
+ "table column",
32
+ "table row",
33
+ "table column header",
34
+ "table projected row header",
35
+ "table spanning cell",
36
+ ]
37
+
38
+ def __init__(self, model_dir=None):
39
+ model_dir = model_dir or str(default_model_dir())
40
+ super().__init__(self.labels, "tsr", model_dir)
41
+
42
+ def __call__(self, images, thr=0.2):
43
+ tbls = super().__call__(images, thr)
44
+
45
+ res = []
46
+ # align left&right for rows, align top&bottom for columns
47
+ for tbl in tbls:
48
+ lts = [
49
+ {
50
+ "label": b["type"],
51
+ "score": b["score"],
52
+ "x0": b["bbox"][0],
53
+ "x1": b["bbox"][2],
54
+ "top": b["bbox"][1],
55
+ "bottom": b["bbox"][-1],
56
+ }
57
+ for b in tbl
58
+ ]
59
+ if not lts:
60
+ continue
61
+
62
+ left = [b["x0"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
63
+ right = [b["x1"] for b in lts if b["label"].find("row") > 0 or b["label"].find("header") > 0]
64
+ if not left:
65
+ continue
66
+ left = np.mean(left) if len(left) > 4 else np.min(left)
67
+ right = np.mean(right) if len(right) > 4 else np.max(right)
68
+ for b in lts:
69
+ if b["label"].find("row") > 0 or b["label"].find("header") > 0:
70
+ if b["x0"] > left:
71
+ b["x0"] = left
72
+ if b["x1"] < right:
73
+ b["x1"] = right
74
+
75
+ top = [b["top"] for b in lts if b["label"] == "table column"]
76
+ bottom = [b["bottom"] for b in lts if b["label"] == "table column"]
77
+ if not top:
78
+ res.append(lts)
79
+ continue
80
+ top = np.median(top) if len(top) > 4 else np.min(top)
81
+ bottom = np.median(bottom) if len(bottom) > 4 else np.max(bottom)
82
+ for b in lts:
83
+ if b["label"] == "table column":
84
+ if b["top"] > top:
85
+ b["top"] = top
86
+ if b["bottom"] < bottom:
87
+ b["bottom"] = bottom
88
+
89
+ res.append(lts)
90
+ return res
91
+
92
+ @staticmethod
93
+ def is_caption(bx):
94
+ patt = [
95
+ r"[图表]+[ 0-9::]{2,}",
96
+ r"(?i)Fig\.?\s*\d+",
97
+ r"(?i)Figure\s+\d+",
98
+ r"(?i)Table\s+\d+",
99
+ ]
100
+ if any([re.match(p, bx["text"].strip()) for p in patt]) or bx.get("layout_type", "").find("caption") >= 0:
101
+ return True
102
+ return False
103
+
104
+ @staticmethod
105
+ def blockType(b):
106
+ patt = [
107
+ ("^(20|19)[0-9]{2}[年/-][0-9]{1,2}[月/-][0-9]{1,2}日*$", "Dt"),
108
+ (r"^(20|19)[0-9]{2}年$", "Dt"),
109
+ (r"^(20|19)[0-9]{2}[年-][0-9]{1,2}月*$", "Dt"),
110
+ ("^[0-9]{1,2}[月-][0-9]{1,2}日*$", "Dt"),
111
+ (r"^第*[一二三四1-4]季度$", "Dt"),
112
+ (r"^(20|19)[0-9]{2}年*[一二三四1-4]季度$", "Dt"),
113
+ (r"^(20|19)[0-9]{2}[ABCDE]$", "Dt"),
114
+ ("^[0-9.,+%/ -]+$", "Nu"),
115
+ (r"^[0-9A-Z/\._~-]+$", "Ca"),
116
+ (r"^[A-Z]*[a-z' -]+$", "En"),
117
+ (r"^[0-9.,+-]+[0-9A-Za-z/$¥%<>()()' -]+$", "NE"),
118
+ (r"^.{1}$", "Sg"),
119
+ ]
120
+ for p, n in patt:
121
+ if re.search(p, b["text"].strip()):
122
+ return n
123
+ tks = [t for t in tokenize(b["text"]).split() if len(t) > 1]
124
+ if len(tks) > 3:
125
+ if len(tks) < 12:
126
+ return "Tx"
127
+ else:
128
+ return "Lx"
129
+
130
+ if len(tks) == 1 and tag(tks[0]) == "nr":
131
+ return "Nr"
132
+
133
+ return "Ot"
134
+
135
+ @staticmethod
136
+ def construct_table(boxes, is_english=False, html=True, **kwargs):
137
+ cap = ""
138
+ i = 0
139
+ while i < len(boxes):
140
+ if TableStructureRecognizer.is_caption(boxes[i]):
141
+ if is_english:
142
+ cap += " "
143
+ cap += boxes[i]["text"]
144
+ boxes.pop(i)
145
+ i -= 1
146
+ i += 1
147
+
148
+ if not boxes:
149
+ return []
150
+ for b in boxes:
151
+ b["btype"] = TableStructureRecognizer.blockType(b)
152
+ max_type = Counter([b["btype"] for b in boxes]).items()
153
+ max_type = max(max_type, key=lambda x: x[1])[0] if max_type else ""
154
+ logging.debug("MAXTYPE: " + max_type)
155
+
156
+ rowh = [b["R_bott"] - b["R_top"] for b in boxes if "R" in b]
157
+ rowh = np.min(rowh) if rowh else 0
158
+ boxes = Recognizer.sort_R_firstly(boxes, rowh / 2)
159
+ # for b in boxes:print(b)
160
+ boxes[0]["rn"] = 0
161
+ rows = [[boxes[0]]]
162
+ btm = boxes[0]["bottom"]
163
+ for b in boxes[1:]:
164
+ b["rn"] = len(rows) - 1
165
+ lst_r = rows[-1]
166
+ if lst_r[-1].get("R", "") != b.get("R", "") or (b["top"] >= btm - 3 and lst_r[-1].get("R", "-1") != b.get("R", "-2")): # new row
167
+ btm = b["bottom"]
168
+ b["rn"] += 1
169
+ rows.append([b])
170
+ continue
171
+ btm = (btm + b["bottom"]) / 2.0
172
+ rows[-1].append(b)
173
+
174
+ colwm = [b["C_right"] - b["C_left"] for b in boxes if "C" in b]
175
+ colwm = np.min(colwm) if colwm else 0
176
+ crosspage = len(set([b["page_number"] for b in boxes])) > 1
177
+ if crosspage:
178
+ boxes = Recognizer.sort_X_firstly(boxes, colwm / 2)
179
+ else:
180
+ boxes = Recognizer.sort_C_firstly(boxes, colwm / 2)
181
+ boxes[0]["cn"] = 0
182
+ cols = [[boxes[0]]]
183
+ right = boxes[0]["x1"]
184
+ for b in boxes[1:]:
185
+ b["cn"] = len(cols) - 1
186
+ lst_c = cols[-1]
187
+ if (int(b.get("C", "1")) - int(lst_c[-1].get("C", "1")) == 1 and b["page_number"] == lst_c[-1]["page_number"]) or (
188
+ b["x0"] >= right and lst_c[-1].get("C", "-1") != b.get("C", "-2")
189
+ ): # new col
190
+ right = b["x1"]
191
+ b["cn"] += 1
192
+ cols.append([b])
193
+ continue
194
+ right = (right + b["x1"]) / 2.0
195
+ cols[-1].append(b)
196
+
197
+ tbl = [[[] for _ in range(len(cols))] for _ in range(len(rows))]
198
+ for b in boxes:
199
+ tbl[b["rn"]][b["cn"]].append(b)
200
+
201
+ if len(rows) >= 4:
202
+ # remove single in column
203
+ j = 0
204
+ while j < len(tbl[0]):
205
+ e, ii = 0, 0
206
+ for i in range(len(tbl)):
207
+ if tbl[i][j]:
208
+ e += 1
209
+ ii = i
210
+ if e > 1:
211
+ break
212
+ if e > 1:
213
+ j += 1
214
+ continue
215
+ f = (j > 0 and tbl[ii][j - 1] and tbl[ii][j - 1][0].get("text")) or j == 0
216
+ ff = (j + 1 < len(tbl[ii]) and tbl[ii][j + 1] and tbl[ii][j + 1][0].get("text")) or j + 1 >= len(tbl[ii])
217
+ if f and ff:
218
+ j += 1
219
+ continue
220
+ bx = tbl[ii][j][0]
221
+ logging.debug("Relocate column single: " + bx["text"])
222
+ # j column only has one value
223
+ left, right = 100000, 100000
224
+ if j > 0 and not f:
225
+ for i in range(len(tbl)):
226
+ if tbl[i][j - 1]:
227
+ left = min(left, np.min([bx["x0"] - a["x1"] for a in tbl[i][j - 1]]))
228
+ if j + 1 < len(tbl[0]) and not ff:
229
+ for i in range(len(tbl)):
230
+ if tbl[i][j + 1]:
231
+ right = min(right, np.min([a["x0"] - bx["x1"] for a in tbl[i][j + 1]]))
232
+ assert left < 100000 or right < 100000
233
+ if left < right:
234
+ for jj in range(j, len(tbl[0])):
235
+ for i in range(len(tbl)):
236
+ for a in tbl[i][jj]:
237
+ a["cn"] -= 1
238
+ if tbl[ii][j - 1]:
239
+ tbl[ii][j - 1].extend(tbl[ii][j])
240
+ else:
241
+ tbl[ii][j - 1] = tbl[ii][j]
242
+ for i in range(len(tbl)):
243
+ tbl[i].pop(j)
244
+
245
+ else:
246
+ for jj in range(j + 1, len(tbl[0])):
247
+ for i in range(len(tbl)):
248
+ for a in tbl[i][jj]:
249
+ a["cn"] -= 1
250
+ if tbl[ii][j + 1]:
251
+ tbl[ii][j + 1].extend(tbl[ii][j])
252
+ else:
253
+ tbl[ii][j + 1] = tbl[ii][j]
254
+ for i in range(len(tbl)):
255
+ tbl[i].pop(j)
256
+ cols.pop(j)
257
+ assert len(cols) == len(tbl[0]), "Column NO. miss matched: %d vs %d" % (len(cols), len(tbl[0]))
258
+
259
+ if len(cols) >= 4:
260
+ # remove single in row
261
+ i = 0
262
+ while i < len(tbl):
263
+ e, jj = 0, 0
264
+ for j in range(len(tbl[i])):
265
+ if tbl[i][j]:
266
+ e += 1
267
+ jj = j
268
+ if e > 1:
269
+ break
270
+ if e > 1:
271
+ i += 1
272
+ continue
273
+ f = (i > 0 and tbl[i - 1][jj] and tbl[i - 1][jj][0].get("text")) or i == 0
274
+ ff = (i + 1 < len(tbl) and tbl[i + 1][jj] and tbl[i + 1][jj][0].get("text")) or i + 1 >= len(tbl)
275
+ if f and ff:
276
+ i += 1
277
+ continue
278
+
279
+ bx = tbl[i][jj][0]
280
+ logging.debug("Relocate row single: " + bx["text"])
281
+ # i row only has one value
282
+ up, down = 100000, 100000
283
+ if i > 0 and not f:
284
+ for j in range(len(tbl[i - 1])):
285
+ if tbl[i - 1][j]:
286
+ up = min(up, np.min([bx["top"] - a["bottom"] for a in tbl[i - 1][j]]))
287
+ if i + 1 < len(tbl) and not ff:
288
+ for j in range(len(tbl[i + 1])):
289
+ if tbl[i + 1][j]:
290
+ down = min(down, np.min([a["top"] - bx["bottom"] for a in tbl[i + 1][j]]))
291
+ assert up < 100000 or down < 100000
292
+ if up < down:
293
+ for ii in range(i, len(tbl)):
294
+ for j in range(len(tbl[ii])):
295
+ for a in tbl[ii][j]:
296
+ a["rn"] -= 1
297
+ if tbl[i - 1][jj]:
298
+ tbl[i - 1][jj].extend(tbl[i][jj])
299
+ else:
300
+ tbl[i - 1][jj] = tbl[i][jj]
301
+ tbl.pop(i)
302
+
303
+ else:
304
+ for ii in range(i + 1, len(tbl)):
305
+ for j in range(len(tbl[ii])):
306
+ for a in tbl[ii][j]:
307
+ a["rn"] -= 1
308
+ if tbl[i + 1][jj]:
309
+ tbl[i + 1][jj].extend(tbl[i][jj])
310
+ else:
311
+ tbl[i + 1][jj] = tbl[i][jj]
312
+ tbl.pop(i)
313
+ rows.pop(i)
314
+
315
+ # which rows are headers
316
+ hdset = set([])
317
+ for i in range(len(tbl)):
318
+ cnt, h = 0, 0
319
+ for j, arr in enumerate(tbl[i]):
320
+ if not arr:
321
+ continue
322
+ cnt += 1
323
+ if max_type == "Nu" and arr[0]["btype"] == "Nu":
324
+ continue
325
+ if any([a.get("H") for a in arr]) or (max_type == "Nu" and arr[0]["btype"] != "Nu"):
326
+ h += 1
327
+ if h / cnt > 0.5:
328
+ hdset.add(i)
329
+
330
+ if html:
331
+ return TableStructureRecognizer.__html_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, True))
332
+
333
+ return TableStructureRecognizer.__desc_table(cap, hdset, TableStructureRecognizer.__cal_spans(boxes, rows, cols, tbl, False), is_english)
334
+
335
+ @staticmethod
336
+ def __html_table(cap, hdset, tbl):
337
+ # constrcut HTML
338
+ html = "<table>"
339
+ if cap:
340
+ html += f"<caption>{cap}</caption>"
341
+ for i in range(len(tbl)):
342
+ row = "<tr>"
343
+ txts = []
344
+ for j, arr in enumerate(tbl[i]):
345
+ if arr is None:
346
+ continue
347
+ if not arr:
348
+ row += "<td></td>" if i not in hdset else "<th></th>"
349
+ continue
350
+ txt = ""
351
+ if arr:
352
+ h = min(np.min([c["bottom"] - c["top"] for c in arr]) / 2, 10)
353
+ txt = " ".join([c["text"] for c in Recognizer.sort_Y_firstly(arr, h)])
354
+ txts.append(txt)
355
+ sp = ""
356
+ if arr[0].get("colspan"):
357
+ sp = "colspan={}".format(arr[0]["colspan"])
358
+ if arr[0].get("rowspan"):
359
+ sp += " rowspan={}".format(arr[0]["rowspan"])
360
+ if i in hdset:
361
+ row += f"<th {sp} >" + txt + "</th>"
362
+ else:
363
+ row += f"<td {sp} >" + txt + "</td>"
364
+
365
+ if i in hdset:
366
+ if all([t in hdset for t in txts]):
367
+ continue
368
+ for t in txts:
369
+ hdset.add(t)
370
+
371
+ if row != "<tr>":
372
+ row += "</tr>"
373
+ else:
374
+ row = ""
375
+ html += "\n" + row
376
+ html += "\n</table>"
377
+ return html
378
+
379
+ @staticmethod
380
+ def __desc_table(cap, hdr_rowno, tbl, is_english):
381
+ # get text of every column in header row to become header text
382
+ clmno = len(tbl[0])
383
+ rowno = len(tbl)
384
+ headers = {}
385
+ hdrset = set()
386
+ lst_hdr = []
387
+ de = "的" if not is_english else " for "
388
+ for r in sorted(list(hdr_rowno)):
389
+ headers[r] = ["" for _ in range(clmno)]
390
+ for i in range(clmno):
391
+ if not tbl[r][i]:
392
+ continue
393
+ txt = " ".join([a["text"].strip() for a in tbl[r][i]])
394
+ headers[r][i] = txt
395
+ hdrset.add(txt)
396
+ if all([not t for t in headers[r]]):
397
+ del headers[r]
398
+ hdr_rowno.remove(r)
399
+ continue
400
+ for j in range(clmno):
401
+ if headers[r][j]:
402
+ continue
403
+ if j >= len(lst_hdr):
404
+ break
405
+ headers[r][j] = lst_hdr[j]
406
+ lst_hdr = headers[r]
407
+ for i in range(rowno):
408
+ if i not in hdr_rowno:
409
+ continue
410
+ for j in range(i + 1, rowno):
411
+ if j not in hdr_rowno:
412
+ break
413
+ for k in range(clmno):
414
+ if not headers[j - 1][k]:
415
+ continue
416
+ if headers[j][k].find(headers[j - 1][k]) >= 0:
417
+ continue
418
+ if len(headers[j][k]) > len(headers[j - 1][k]):
419
+ headers[j][k] += (de if headers[j][k] else "") + headers[j - 1][k]
420
+ else:
421
+ headers[j][k] = headers[j - 1][k] + (de if headers[j - 1][k] else "") + headers[j][k]
422
+
423
+ logging.debug(f">>>>>>>>>>>>>>>>>{cap}:SIZE:{rowno}X{clmno} Header: {hdr_rowno}")
424
+ row_txt = []
425
+ for i in range(rowno):
426
+ if i in hdr_rowno:
427
+ continue
428
+ rtxt = []
429
+
430
+ def append(delimer):
431
+ nonlocal rtxt, row_txt
432
+ rtxt = delimer.join(rtxt)
433
+ if row_txt and len(row_txt[-1]) + len(rtxt) < 64:
434
+ row_txt[-1] += "\n" + rtxt
435
+ else:
436
+ row_txt.append(rtxt)
437
+
438
+ r = 0
439
+ if len(headers.items()):
440
+ _arr = [(i - r, r) for r, _ in headers.items() if r < i]
441
+ if _arr:
442
+ _, r = min(_arr, key=lambda x: x[0])
443
+
444
+ if r not in headers and clmno <= 2:
445
+ for j in range(clmno):
446
+ if not tbl[i][j]:
447
+ continue
448
+ txt = "".join([a["text"].strip() for a in tbl[i][j]])
449
+ if txt:
450
+ rtxt.append(txt)
451
+ if rtxt:
452
+ append(":")
453
+ continue
454
+
455
+ for j in range(clmno):
456
+ if not tbl[i][j]:
457
+ continue
458
+ txt = "".join([a["text"].strip() for a in tbl[i][j]])
459
+ if not txt:
460
+ continue
461
+ ctt = headers[r][j] if r in headers else ""
462
+ if ctt:
463
+ ctt += ":"
464
+ ctt += txt
465
+ if ctt:
466
+ rtxt.append(ctt)
467
+
468
+ if rtxt:
469
+ row_txt.append("; ".join(rtxt))
470
+
471
+ if cap:
472
+ if is_english:
473
+ from_ = " in "
474
+ else:
475
+ from_ = "来自"
476
+ row_txt = [t + f"\t——{from_}“{cap}”" for t in row_txt]
477
+ return row_txt
478
+
479
+ @staticmethod
480
+ def __cal_spans(boxes, rows, cols, tbl, html=True):
481
+ # caculate span
482
+ clft = [np.mean([c.get("C_left", c["x0"]) for c in cln]) for cln in cols]
483
+ crgt = [np.mean([c.get("C_right", c["x1"]) for c in cln]) for cln in cols]
484
+ rtop = [np.mean([c.get("R_top", c["top"]) for c in row]) for row in rows]
485
+ rbtm = [np.mean([c.get("R_btm", c["bottom"]) for c in row]) for row in rows]
486
+ for b in boxes:
487
+ if "SP" not in b:
488
+ continue
489
+ b["colspan"] = [b["cn"]]
490
+ b["rowspan"] = [b["rn"]]
491
+ # col span
492
+ for j in range(0, len(clft)):
493
+ if j == b["cn"]:
494
+ continue
495
+ if clft[j] + (crgt[j] - clft[j]) / 2 < b["H_left"]:
496
+ continue
497
+ if crgt[j] - (crgt[j] - clft[j]) / 2 > b["H_right"]:
498
+ continue
499
+ b["colspan"].append(j)
500
+ # row span
501
+ for j in range(0, len(rtop)):
502
+ if j == b["rn"]:
503
+ continue
504
+ if rtop[j] + (rbtm[j] - rtop[j]) / 2 < b["H_top"]:
505
+ continue
506
+ if rbtm[j] - (rbtm[j] - rtop[j]) / 2 > b["H_bott"]:
507
+ continue
508
+ b["rowspan"].append(j)
509
+
510
+ def join(arr):
511
+ if not arr:
512
+ return ""
513
+ return "".join([t["text"] for t in arr])
514
+
515
+ # rm the spaning cells
516
+ for i in range(len(tbl)):
517
+ for j, arr in enumerate(tbl[i]):
518
+ if not arr:
519
+ continue
520
+ if all(["rowspan" not in a and "colspan" not in a for a in arr]):
521
+ continue
522
+ rowspan, colspan = [], []
523
+ for a in arr:
524
+ if isinstance(a.get("rowspan", 0), list):
525
+ rowspan.extend(a["rowspan"])
526
+ if isinstance(a.get("colspan", 0), list):
527
+ colspan.extend(a["colspan"])
528
+ rowspan, colspan = set(rowspan), set(colspan)
529
+ if len(rowspan) < 2 and len(colspan) < 2:
530
+ for a in arr:
531
+ if "rowspan" in a:
532
+ del a["rowspan"]
533
+ if "colspan" in a:
534
+ del a["colspan"]
535
+ continue
536
+ rowspan, colspan = sorted(rowspan), sorted(colspan)
537
+ rowspan = list(range(rowspan[0], rowspan[-1] + 1))
538
+ colspan = list(range(colspan[0], colspan[-1] + 1))
539
+ assert i in rowspan, rowspan
540
+ assert j in colspan, colspan
541
+ arr = []
542
+ for r in rowspan:
543
+ for c in colspan:
544
+ arr_txt = join(arr)
545
+ if tbl[r][c] and join(tbl[r][c]) != arr_txt:
546
+ arr.extend(tbl[r][c])
547
+ tbl[r][c] = None if html else arr
548
+ for a in arr:
549
+ if len(rowspan) > 1:
550
+ a["rowspan"] = len(rowspan)
551
+ elif "rowspan" in a:
552
+ del a["rowspan"]
553
+ if len(colspan) > 1:
554
+ a["colspan"] = len(colspan)
555
+ elif "colspan" in a:
556
+ del a["colspan"]
557
+ tbl[rowspan[0]][colspan[0]] = arr
558
+
559
+ return tbl
@@ -0,0 +1,30 @@
1
+ """
2
+ Lightweight stand-in for RAGFlow's rag_tokenizer (which itself wraps a
3
+ tokenizer bundled inside the infinity-sdk vector-DB client). deepdoc's live
4
+ call sites only need coarse signals -- "is this char CJK", "how many
5
+ word-tokens is this text", "is this single token a person name" -- for
6
+ table-cell type classification, not text reconstruction, so a real
7
+ segmenter (jieba) is enough; we don't need infinity-sdk's tokenizer.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import jieba
13
+ import jieba.posseg as jieba_posseg
14
+
15
+ _CJK_RANGE = ("一", "鿿")
16
+
17
+
18
+ def is_chinese(text: str) -> bool:
19
+ return bool(text) and any(_CJK_RANGE[0] <= ch <= _CJK_RANGE[1] for ch in text)
20
+
21
+
22
+ def tokenize(text: str) -> str:
23
+ return " ".join(jieba.cut(text))
24
+
25
+
26
+ def tag(token: str) -> str:
27
+ if not token:
28
+ return ""
29
+ words = list(jieba_posseg.cut(token))
30
+ return words[0].flag if words else ""
@@ -0,0 +1,36 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+ from io import BytesIO
17
+
18
+ from pypdf import PdfReader as pdf2_read
19
+
20
+
21
+ def extract_pdf_outlines(source):
22
+ try:
23
+ with pdf2_read(source if isinstance(source, str) else BytesIO(source)) as pdf:
24
+ outlines = []
25
+
26
+ def dfs(nodes, depth):
27
+ for node in nodes:
28
+ if isinstance(node, list):
29
+ dfs(node, depth + 1)
30
+ else:
31
+ outlines.append((node["/Title"], depth, pdf.get_destination_page_number(node) + 1))
32
+
33
+ dfs(pdf.outline, 0)
34
+ return outlines
35
+ except Exception:
36
+ return []