langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,418 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+ import gc
17
+ import logging
18
+ import math
19
+ import numpy as np
20
+ import cv2
21
+ from functools import cmp_to_key
22
+
23
+
24
+ from .model_loader import default_model_dir
25
+ from .operators import * # noqa: F403
26
+ from .operators import preprocess
27
+ from . import operators
28
+ from .ocr import load_model
29
+
30
+
31
+ class Recognizer:
32
+ def __init__(self, label_list, task_name, model_dir=None):
33
+ """
34
+ If you have trouble downloading HuggingFace models, -_^ this might help!!
35
+
36
+ For Linux:
37
+ export HF_ENDPOINT=https://hf-mirror.com
38
+
39
+ For Windows:
40
+ Good luck
41
+ ^_-
42
+
43
+ """
44
+ if not model_dir:
45
+ model_dir = str(default_model_dir())
46
+ self.ort_sess, self.run_options = load_model(model_dir, task_name)
47
+ self.input_names = [node.name for node in self.ort_sess.get_inputs()]
48
+ self.output_names = [node.name for node in self.ort_sess.get_outputs()]
49
+ self.input_shape = self.ort_sess.get_inputs()[0].shape[2:4]
50
+ self.label_list = label_list
51
+
52
+ @staticmethod
53
+ def sort_Y_firstly(arr, threshold):
54
+ def cmp(c1, c2):
55
+ diff = c1["top"] - c2["top"]
56
+ if abs(diff) < threshold:
57
+ diff = c1["x0"] - c2["x0"]
58
+ return diff
59
+
60
+ arr = sorted(arr, key=cmp_to_key(cmp))
61
+ return arr
62
+
63
+ @staticmethod
64
+ def sort_X_firstly(arr, threshold):
65
+ def cmp(c1, c2):
66
+ diff = c1["x0"] - c2["x0"]
67
+ if abs(diff) < threshold:
68
+ diff = c1["top"] - c2["top"]
69
+ return diff
70
+
71
+ arr = sorted(arr, key=cmp_to_key(cmp))
72
+ return arr
73
+
74
+ @staticmethod
75
+ def sort_C_firstly(arr, thr=0):
76
+ # sort using y1 first and then x1
77
+ # sorted(arr, key=lambda r: (r["x0"], r["top"]))
78
+ arr = Recognizer.sort_X_firstly(arr, thr)
79
+ for i in range(len(arr) - 1):
80
+ for j in range(i, -1, -1):
81
+ # restore the order using th
82
+ if "C" not in arr[j] or "C" not in arr[j + 1]:
83
+ continue
84
+ if arr[j + 1]["C"] < arr[j]["C"] or (arr[j + 1]["C"] == arr[j]["C"] and arr[j + 1]["top"] < arr[j]["top"]):
85
+ tmp = arr[j]
86
+ arr[j] = arr[j + 1]
87
+ arr[j + 1] = tmp
88
+ return arr
89
+
90
+ @staticmethod
91
+ def sort_R_firstly(arr, thr=0):
92
+ # sort using y1 first and then x1
93
+ # sorted(arr, key=lambda r: (r["top"], r["x0"]))
94
+ arr = Recognizer.sort_Y_firstly(arr, thr)
95
+ for i in range(len(arr) - 1):
96
+ for j in range(i, -1, -1):
97
+ if "R" not in arr[j] or "R" not in arr[j + 1]:
98
+ continue
99
+ if arr[j + 1]["R"] < arr[j]["R"] or (arr[j + 1]["R"] == arr[j]["R"] and arr[j + 1]["x0"] < arr[j]["x0"]):
100
+ tmp = arr[j]
101
+ arr[j] = arr[j + 1]
102
+ arr[j + 1] = tmp
103
+ return arr
104
+
105
+ @staticmethod
106
+ def overlapped_area(a, b, ratio=True):
107
+ tp, btm, x0, x1 = a["top"], a["bottom"], a["x0"], a["x1"]
108
+ if b["x0"] > x1 or b["x1"] < x0:
109
+ return 0
110
+ if b["bottom"] < tp or b["top"] > btm:
111
+ return 0
112
+ x0_ = max(b["x0"], x0)
113
+ x1_ = min(b["x1"], x1)
114
+ assert x0_ <= x1_, "Bbox mismatch! T:{},B:{},X0:{},X1:{} ==> {}".format(tp, btm, x0, x1, b)
115
+ tp_ = max(b["top"], tp)
116
+ btm_ = min(b["bottom"], btm)
117
+ assert tp_ <= btm_, "Bbox mismatch! T:{},B:{},X0:{},X1:{} => {}".format(tp, btm, x0, x1, b)
118
+ ov = (btm_ - tp_) * (x1_ - x0_) if x1 - x0 != 0 and btm - tp != 0 else 0
119
+ if ov > 0 and ratio:
120
+ ov /= (x1 - x0) * (btm - tp)
121
+ return ov
122
+
123
+ @staticmethod
124
+ def layouts_cleanup(boxes, layouts, far=2, thr=0.7):
125
+ def not_overlapped(a, b):
126
+ return any([a["x1"] < b["x0"], a["x0"] > b["x1"], a["bottom"] < b["top"], a["top"] > b["bottom"]])
127
+
128
+ i = 0
129
+ while i + 1 < len(layouts):
130
+ j = i + 1
131
+ while j < min(i + far, len(layouts)) and (layouts[i].get("type", "") != layouts[j].get("type", "") or not_overlapped(layouts[i], layouts[j])):
132
+ j += 1
133
+ if j >= min(i + far, len(layouts)):
134
+ i += 1
135
+ continue
136
+ if Recognizer.overlapped_area(layouts[i], layouts[j]) < thr and Recognizer.overlapped_area(layouts[j], layouts[i]) < thr:
137
+ i += 1
138
+ continue
139
+
140
+ if layouts[i].get("score") and layouts[j].get("score"):
141
+ if layouts[i]["score"] > layouts[j]["score"]:
142
+ layouts.pop(j)
143
+ else:
144
+ layouts.pop(i)
145
+ continue
146
+
147
+ area_i, area_i_1 = 0, 0
148
+ for b in boxes:
149
+ if not not_overlapped(b, layouts[i]):
150
+ area_i += Recognizer.overlapped_area(b, layouts[i], False)
151
+ if not not_overlapped(b, layouts[j]):
152
+ area_i_1 += Recognizer.overlapped_area(b, layouts[j], False)
153
+
154
+ if area_i > area_i_1:
155
+ layouts.pop(j)
156
+ else:
157
+ layouts.pop(i)
158
+
159
+ return layouts
160
+
161
+ def create_inputs(self, imgs, im_info):
162
+ """generate input for different model type
163
+ Args:
164
+ imgs (list(numpy)): list of images (np.ndarray)
165
+ im_info (list(dict)): list of image info
166
+ Returns:
167
+ inputs (dict): input of model
168
+ """
169
+ inputs = {}
170
+
171
+ im_shape = []
172
+ scale_factor = []
173
+ if len(imgs) == 1:
174
+ inputs["image"] = np.array((imgs[0],)).astype("float32")
175
+ inputs["im_shape"] = np.array((im_info[0]["im_shape"],)).astype("float32")
176
+ inputs["scale_factor"] = np.array((im_info[0]["scale_factor"],)).astype("float32")
177
+ return inputs
178
+
179
+ im_shape = np.array([info["im_shape"] for info in im_info], dtype="float32")
180
+ scale_factor = np.array([info["scale_factor"] for info in im_info], dtype="float32")
181
+
182
+ inputs["im_shape"] = np.concatenate(im_shape, axis=0)
183
+ inputs["scale_factor"] = np.concatenate(scale_factor, axis=0)
184
+
185
+ imgs_shape = [[e.shape[1], e.shape[2]] for e in imgs]
186
+ max_shape_h = max([e[0] for e in imgs_shape])
187
+ max_shape_w = max([e[1] for e in imgs_shape])
188
+ padding_imgs = []
189
+ for img in imgs:
190
+ im_c, im_h, im_w = img.shape[:]
191
+ padding_im = np.zeros((im_c, max_shape_h, max_shape_w), dtype=np.float32)
192
+ padding_im[:, :im_h, :im_w] = img
193
+ padding_imgs.append(padding_im)
194
+ inputs["image"] = np.stack(padding_imgs, axis=0)
195
+ return inputs
196
+
197
+ @staticmethod
198
+ def find_overlapped(box, boxes_sorted_by_y, naive=False):
199
+ if not boxes_sorted_by_y:
200
+ return
201
+ bxs = boxes_sorted_by_y
202
+ s, e, ii = 0, len(bxs), 0
203
+ while s < e and not naive:
204
+ ii = (e + s) // 2
205
+ pv = bxs[ii]
206
+ if box["bottom"] < pv["top"]:
207
+ e = ii
208
+ continue
209
+ if box["top"] > pv["bottom"]:
210
+ s = ii + 1
211
+ continue
212
+ break
213
+ while s < ii:
214
+ if box["top"] > bxs[s]["bottom"]:
215
+ s += 1
216
+ break
217
+ while e - 1 > ii:
218
+ if box["bottom"] < bxs[e - 1]["top"]:
219
+ e -= 1
220
+ break
221
+
222
+ max_overlapped_i, max_overlapped = None, 0
223
+ for i in range(s, e):
224
+ ov = Recognizer.overlapped_area(bxs[i], box)
225
+ if ov <= max_overlapped:
226
+ continue
227
+ max_overlapped_i = i
228
+ max_overlapped = ov
229
+
230
+ return max_overlapped_i
231
+
232
+ @staticmethod
233
+ def find_horizontally_tightest_fit(box, boxes):
234
+ if not boxes:
235
+ return
236
+ min_dis, min_i = 1000000, None
237
+ for i, b in enumerate(boxes):
238
+ if box.get("layoutno", "0") != b.get("layoutno", "0"):
239
+ continue
240
+ # layoutno is f"table-{index}" and resets per page, so the same string
241
+ # names a different table on every page. Only accept a candidate that
242
+ # shares vertical extent with the box; top/bottom are page-cumulative,
243
+ # so this rejects a same-layoutno column from another page whose x range
244
+ # happens to be closer.
245
+ if min(box["bottom"], b["bottom"]) <= max(box["top"], b["top"]):
246
+ continue
247
+ dis = min(abs(box["x0"] - b["x0"]), abs(box["x1"] - b["x1"]), abs(box["x0"] + box["x1"] - b["x1"] - b["x0"]) / 2)
248
+ if dis < min_dis:
249
+ min_i = i
250
+ min_dis = dis
251
+ return min_i
252
+
253
+ @staticmethod
254
+ def find_overlapped_with_threshold(box, boxes, thr=0.3):
255
+ if not boxes:
256
+ return
257
+ max_overlapped_i, max_overlapped, _max_overlapped = None, thr, 0
258
+ s, e = 0, len(boxes)
259
+ for i in range(s, e):
260
+ ov = Recognizer.overlapped_area(box, boxes[i])
261
+ _ov = Recognizer.overlapped_area(boxes[i], box)
262
+ if (ov, _ov) < (max_overlapped, _max_overlapped):
263
+ continue
264
+ max_overlapped_i = i
265
+ max_overlapped = ov
266
+ _max_overlapped = _ov
267
+
268
+ return max_overlapped_i
269
+
270
+ def preprocess(self, image_list):
271
+ inputs = []
272
+ if "scale_factor" in self.input_names:
273
+ preprocess_ops = []
274
+ for op_info in [
275
+ {"interp": 2, "keep_ratio": False, "target_size": [800, 608], "type": "LinearResize"},
276
+ {"is_scale": True, "mean": [0.485, 0.456, 0.406], "std": [0.229, 0.224, 0.225], "type": "StandardizeImage"},
277
+ {"type": "Permute"},
278
+ {"stride": 32, "type": "PadStride"},
279
+ ]:
280
+ new_op_info = op_info.copy()
281
+ op_type = new_op_info.pop("type")
282
+ preprocess_ops.append(getattr(operators, op_type)(**new_op_info))
283
+
284
+ for im_path in image_list:
285
+ im, im_info = preprocess(im_path, preprocess_ops)
286
+ inputs.append({"image": np.array((im,)).astype("float32"), "scale_factor": np.array((im_info["scale_factor"],)).astype("float32")})
287
+ else:
288
+ hh, ww = self.input_shape
289
+ for img in image_list:
290
+ h, w = img.shape[:2]
291
+ img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
292
+ img = cv2.resize(np.array(img).astype("float32"), (ww, hh))
293
+ # Scale input pixel values to 0 to 1
294
+ img /= 255.0
295
+ img = img.transpose(2, 0, 1)
296
+ img = img[np.newaxis, :, :, :].astype(np.float32)
297
+ inputs.append({self.input_names[0]: img, "scale_factor": [w / ww, h / hh]})
298
+ return inputs
299
+
300
+ def postprocess(self, boxes, inputs, thr):
301
+ if "scale_factor" in self.input_names:
302
+ bb = []
303
+ for b in boxes:
304
+ clsid, bbox, score = int(b[0]), b[2:], b[1]
305
+ if score < thr:
306
+ continue
307
+ if clsid >= len(self.label_list):
308
+ continue
309
+ bb.append({"type": self.label_list[clsid].lower(), "bbox": [float(t) for t in bbox.tolist()], "score": float(score)})
310
+ return bb
311
+
312
+ def xywh2xyxy(x):
313
+ # [x, y, w, h] to [x1, y1, x2, y2]
314
+ y = np.copy(x)
315
+ y[:, 0] = x[:, 0] - x[:, 2] / 2
316
+ y[:, 1] = x[:, 1] - x[:, 3] / 2
317
+ y[:, 2] = x[:, 0] + x[:, 2] / 2
318
+ y[:, 3] = x[:, 1] + x[:, 3] / 2
319
+ return y
320
+
321
+ def compute_iou(box, boxes):
322
+ # Compute xmin, ymin, xmax, ymax for both boxes
323
+ xmin = np.maximum(box[0], boxes[:, 0])
324
+ ymin = np.maximum(box[1], boxes[:, 1])
325
+ xmax = np.minimum(box[2], boxes[:, 2])
326
+ ymax = np.minimum(box[3], boxes[:, 3])
327
+
328
+ # Compute intersection area
329
+ intersection_area = np.maximum(0, xmax - xmin) * np.maximum(0, ymax - ymin)
330
+
331
+ # Compute union area
332
+ box_area = (box[2] - box[0]) * (box[3] - box[1])
333
+ boxes_area = (boxes[:, 2] - boxes[:, 0]) * (boxes[:, 3] - boxes[:, 1])
334
+ union_area = box_area + boxes_area - intersection_area
335
+
336
+ # Compute IoU
337
+ iou = intersection_area / union_area
338
+
339
+ return iou
340
+
341
+ def iou_filter(boxes, scores, iou_threshold):
342
+ sorted_indices = np.argsort(scores)[::-1]
343
+
344
+ keep_boxes = []
345
+ while sorted_indices.size > 0:
346
+ # Pick the last box
347
+ box_id = sorted_indices[0]
348
+ keep_boxes.append(box_id)
349
+
350
+ # Compute IoU of the picked box with the rest
351
+ ious = compute_iou(boxes[box_id, :], boxes[sorted_indices[1:], :])
352
+
353
+ # Remove boxes with IoU over the threshold
354
+ keep_indices = np.where(ious < iou_threshold)[0]
355
+
356
+ # print(keep_indices.shape, sorted_indices.shape)
357
+ sorted_indices = sorted_indices[keep_indices + 1]
358
+
359
+ return keep_boxes
360
+
361
+ boxes = np.squeeze(boxes).T
362
+ # Filter out object confidence scores below threshold
363
+ scores = np.max(boxes[:, 4:], axis=1)
364
+ boxes = boxes[scores > thr, :]
365
+ scores = scores[scores > thr]
366
+ if len(boxes) == 0:
367
+ return []
368
+
369
+ # Get the class with the highest confidence
370
+ class_ids = np.argmax(boxes[:, 4:], axis=1)
371
+ boxes = boxes[:, :4]
372
+ input_shape = np.array([inputs["scale_factor"][0], inputs["scale_factor"][1], inputs["scale_factor"][0], inputs["scale_factor"][1]])
373
+ boxes = np.multiply(boxes, input_shape, dtype=np.float32)
374
+ boxes = xywh2xyxy(boxes)
375
+
376
+ unique_class_ids = np.unique(class_ids)
377
+ indices = []
378
+ for class_id in unique_class_ids:
379
+ class_indices = np.where(class_ids == class_id)[0]
380
+ class_boxes = boxes[class_indices, :]
381
+ class_scores = scores[class_indices]
382
+ class_keep_boxes = iou_filter(class_boxes, class_scores, 0.2)
383
+ indices.extend(class_indices[class_keep_boxes])
384
+
385
+ return [{"type": self.label_list[class_ids[i]].lower(), "bbox": [float(t) for t in boxes[i].tolist()], "score": float(scores[i])} for i in indices]
386
+
387
+ def close(self):
388
+ logging.info("Close recognizer.")
389
+ if hasattr(self, "ort_sess"):
390
+ del self.ort_sess
391
+ gc.collect()
392
+
393
+ def __call__(self, image_list, thr=0.7, batch_size=16):
394
+ res = []
395
+ images = []
396
+ for i in range(len(image_list)):
397
+ if not isinstance(image_list[i], np.ndarray):
398
+ images.append(np.array(image_list[i]))
399
+ else:
400
+ images.append(image_list[i])
401
+
402
+ batch_loop_cnt = math.ceil(float(len(images)) / batch_size)
403
+ for i in range(batch_loop_cnt):
404
+ start_index = i * batch_size
405
+ end_index = min((i + 1) * batch_size, len(images))
406
+ batch_image_list = images[start_index:end_index]
407
+ inputs = self.preprocess(batch_image_list)
408
+ logging.debug("preprocess")
409
+ for ins in inputs:
410
+ bb = self.postprocess(self.ort_sess.run(None, {k: v for k, v in ins.items() if k in self.input_names}, self.run_options)[0], ins, thr)
411
+ res.append(bb)
412
+
413
+ # seeit.save_results(image_list, res, self.label_list, threshold=thr)
414
+
415
+ return res
416
+
417
+ def __del__(self):
418
+ self.close()
@@ -0,0 +1,210 @@
1
+ """
2
+ New code, not ported from upstream: deepdoc's TableStructureRecognizer emits
3
+ tables as HTML (colspan/rowspan) or as Chinese natural-language sentences --
4
+ neither matches langparse's cross-engine table shape. This module renders
5
+ deepdoc's box list into ParsedPageResult, normalizing tables to
6
+ {"rows": list[list[str]]} to match SimplePDFEngine/MinerUEngine and to keep
7
+ services/fidelity.py's TEDS scoring working across engines.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections import defaultdict
13
+ from html.parser import HTMLParser
14
+
15
+ from langparse.types import ParsedElement, ParsedPageResult
16
+
17
+
18
+ class _RawTableParser(HTMLParser):
19
+ def __init__(self) -> None:
20
+ super().__init__()
21
+ self.rows: list[list[dict]] = []
22
+ self._row: list[dict] | None = None
23
+ self._cell: list[str] | None = None
24
+ self._cell_attrs: dict[str, str] = {}
25
+
26
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
27
+ attrs_dict = dict(attrs)
28
+ if tag == "tr":
29
+ self._row = []
30
+ elif tag in ("td", "th"):
31
+ self._cell = []
32
+ self._cell_attrs = attrs_dict
33
+
34
+ def handle_data(self, data: str) -> None:
35
+ if self._cell is not None:
36
+ self._cell.append(data)
37
+
38
+ def handle_endtag(self, tag: str) -> None:
39
+ if tag in ("td", "th") and self._row is not None and self._cell is not None:
40
+ text = "".join(self._cell).strip()
41
+ colspan = int(self._cell_attrs.get("colspan", 1) or 1)
42
+ rowspan = int(self._cell_attrs.get("rowspan", 1) or 1)
43
+ self._row.append({"text": text, "colspan": colspan, "rowspan": rowspan})
44
+ self._cell = None
45
+ elif tag == "tr" and self._row is not None:
46
+ self.rows.append(self._row)
47
+ self._row = None
48
+
49
+
50
+ def _extract_raw_rows(html: str) -> list[list[dict]]:
51
+ parser = _RawTableParser()
52
+ parser.feed(html)
53
+ return parser.rows
54
+
55
+
56
+ def html_table_to_rows(html: str) -> list[list[str]]:
57
+ """Flatten an HTML table (with colspan/rowspan) into a plain row grid."""
58
+ raw_rows = _extract_raw_rows(html)
59
+ grid: list[list[str]] = []
60
+ carry: dict[int, tuple[str, int]] = {} # column -> (text, additional rows remaining)
61
+
62
+ for raw_row in raw_rows:
63
+ row: list[str] = []
64
+ col = 0
65
+ cells = list(raw_row)
66
+ cell_index = 0
67
+ while cell_index < len(cells) or col in carry or any(c > col for c in carry):
68
+ if col in carry:
69
+ text, remaining = carry.pop(col)
70
+ row.append(text)
71
+ if remaining > 1:
72
+ carry[col] = (text, remaining - 1)
73
+ col += 1
74
+ continue
75
+ if cell_index >= len(cells):
76
+ col += 1
77
+ continue
78
+ cell = cells[cell_index]
79
+ cell_index += 1
80
+ colspan = max(1, cell["colspan"])
81
+ rowspan = max(1, cell["rowspan"])
82
+ for _ in range(colspan):
83
+ row.append(cell["text"])
84
+ if rowspan > 1:
85
+ carry[col] = (cell["text"], rowspan - 1)
86
+ col += 1
87
+ grid.append(row)
88
+ return grid
89
+
90
+
91
+ def _bbox(box: dict) -> list[float]:
92
+ # box["top"]/box["bottom"] are in document-cumulative Y space (offset by
93
+ # the running sum of prior pages' heights, see _layouts_rec in
94
+ # pdf_parser.py) so boxes sort correctly across a multi-page document
95
+ # internally. box["positions"] -- a list of
96
+ # [page_number, left, right, top, bottom] tuples -- carries the same
97
+ # rectangle in page-local coordinates instead, which is what a
98
+ # ParsedElement.bbox must report (MinerUEngine's bbox is page-local too).
99
+ # Fall back to x0/top/x1/bottom when positions is absent, e.g. for
100
+ # hand-built test fixtures.
101
+ positions = box.get("positions")
102
+ if positions:
103
+ _page_number, left, right, top, bottom = positions[0]
104
+ return [float(left), float(top), float(right), float(bottom)]
105
+ return [
106
+ float(box.get("x0", 0.0)),
107
+ float(box.get("top", 0.0)),
108
+ float(box.get("x1", 0.0)),
109
+ float(box.get("bottom", 0.0)),
110
+ ]
111
+
112
+
113
+ def _rows_to_markdown_table(rows: list[list[str]]) -> str:
114
+ if not rows:
115
+ return ""
116
+ lines = [f"| {' | '.join(rows[0])} |", f"| {' | '.join(['---'] * len(rows[0]))} |"]
117
+ for row in rows[1:]:
118
+ lines.append(f"| {' | '.join(row)} |")
119
+ return "\n".join(lines)
120
+
121
+
122
+ def render_pages(
123
+ boxes: list[dict], ocr_pages: dict[int, bool] | None = None
124
+ ) -> list[ParsedPageResult]:
125
+ """Render deepdoc's flat box list (from RAGFlowPdfParser.parse_into_bboxes) into pages.
126
+
127
+ ocr_pages is an optional {page_number: bool} map (1-indexed, matching
128
+ box["page_number"]) saying whether a page's text should be credited to
129
+ OCR rather than the PDF's native text layer -- see
130
+ DeepDocEngine._classify_ocr_pages, which derives it via the same
131
+ needs_ocr() heuristic simple/ocr.py already uses. A page absent from the
132
+ map, or the map itself omitted, defaults to False/0.
133
+ """
134
+ boxes_by_page: dict[int, list[dict]] = defaultdict(list)
135
+ for box in boxes:
136
+ boxes_by_page[box["page_number"]].append(box)
137
+
138
+ ocr_pages = ocr_pages or {}
139
+ pages = []
140
+ for page_number in sorted(boxes_by_page):
141
+ markdown_parts: list[str] = []
142
+ plain_parts: list[str] = []
143
+ elements: list[ParsedElement] = []
144
+ tables: list[dict] = []
145
+ images: list[dict] = []
146
+
147
+ for box in boxes_by_page[page_number]:
148
+ layout_type = box.get("layout_type") or "text"
149
+ text = (box.get("text") or "").strip()
150
+ bbox = _bbox(box)
151
+
152
+ if layout_type == "table":
153
+ rows = html_table_to_rows(text)
154
+ tables.append({"rows": rows})
155
+ # Use original cells, not the span-expanded grid, so merged
156
+ # text appears once in search text and OCR character counts.
157
+ table_text = "\n".join(
158
+ "\t".join(cell["text"] for cell in row if cell["text"])
159
+ for row in _extract_raw_rows(text)
160
+ ).strip()
161
+ if table_text:
162
+ plain_parts.append(table_text)
163
+ markdown_parts.append(_rows_to_markdown_table(rows))
164
+ elements.append(
165
+ ParsedElement(
166
+ kind="table", text=text, bbox=bbox, metadata={"layout_type": layout_type}
167
+ )
168
+ )
169
+ continue
170
+
171
+ if layout_type == "figure":
172
+ images.append({"caption": text, "bbox": bbox})
173
+ if text:
174
+ markdown_parts.append(f"*{text}*")
175
+ elements.append(
176
+ ParsedElement(
177
+ kind="figure", text=text, bbox=bbox, metadata={"layout_type": layout_type}
178
+ )
179
+ )
180
+ continue
181
+
182
+ if not text:
183
+ continue
184
+
185
+ markdown_parts.append(f"# {text}" if layout_type == "title" else text)
186
+ plain_parts.append(text)
187
+ elements.append(
188
+ ParsedElement(
189
+ kind=layout_type, text=text, bbox=bbox, metadata={"layout_type": layout_type}
190
+ )
191
+ )
192
+
193
+ plain_text = "\n".join(plain_parts)
194
+ page_ocr_applied = bool(ocr_pages.get(page_number, False))
195
+ pages.append(
196
+ ParsedPageResult(
197
+ page_number=page_number,
198
+ markdown_content="\n\n".join(part for part in markdown_parts if part),
199
+ plain_text=plain_text,
200
+ elements=elements,
201
+ tables=tables,
202
+ images=images,
203
+ metadata={
204
+ "engine_name": "deepdoc",
205
+ "ocr_applied": page_ocr_applied,
206
+ "ocr_text_chars": len(plain_text) if page_ocr_applied else 0,
207
+ },
208
+ )
209
+ )
210
+ return pages