langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,339 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+
17
+ import copy
18
+ import re
19
+ import numpy as np
20
+ import cv2
21
+ from shapely.geometry import Polygon
22
+ import pyclipper
23
+
24
+
25
+ def build_post_process(config, global_config=None):
26
+ support_dict = {"DBPostProcess": DBPostProcess, "CTCLabelDecode": CTCLabelDecode}
27
+
28
+ config = copy.deepcopy(config)
29
+ module_name = config.pop("name")
30
+ if module_name == "None":
31
+ return
32
+ if global_config is not None:
33
+ config.update(global_config)
34
+ module_class = support_dict.get(module_name)
35
+ if module_class is None:
36
+ raise ValueError("post process only support {}".format(list(support_dict)))
37
+ return module_class(**config)
38
+
39
+
40
+ class DBPostProcess:
41
+ """
42
+ The post process for Differentiable Binarization (DB).
43
+ """
44
+
45
+ def __init__(self, thresh=0.3, box_thresh=0.7, max_candidates=1000, unclip_ratio=2.0, use_dilation=False, score_mode="fast", box_type="quad", **kwargs):
46
+ self.thresh = thresh
47
+ self.box_thresh = box_thresh
48
+ self.max_candidates = max_candidates
49
+ self.unclip_ratio = unclip_ratio
50
+ self.min_size = 3
51
+ self.score_mode = score_mode
52
+ self.box_type = box_type
53
+ assert score_mode in ["slow", "fast"], "Score mode must be in [slow, fast] but got: {}".format(score_mode)
54
+
55
+ self.dilation_kernel = None if not use_dilation else np.array([[1, 1], [1, 1]])
56
+
57
+ def polygons_from_bitmap(self, pred, _bitmap, dest_width, dest_height):
58
+ """
59
+ _bitmap: single map with shape (1, H, W),
60
+ whose values are binarized as {0, 1}
61
+ """
62
+
63
+ bitmap = _bitmap
64
+ height, width = bitmap.shape
65
+
66
+ boxes = []
67
+ scores = []
68
+
69
+ contours, _ = cv2.findContours((bitmap * 255).astype(np.uint8), cv2.RETR_LIST, cv2.CHAIN_APPROX_SIMPLE)
70
+
71
+ for contour in contours[: self.max_candidates]:
72
+ epsilon = 0.002 * cv2.arcLength(contour, True)
73
+ approx = cv2.approxPolyDP(contour, epsilon, True)
74
+ points = approx.reshape((-1, 2))
75
+ if points.shape[0] < 4:
76
+ continue
77
+
78
+ score = self.box_score_fast(pred, points.reshape(-1, 2))
79
+ if self.box_thresh > score:
80
+ continue
81
+
82
+ if points.shape[0] > 2:
83
+ box = self.unclip(points, self.unclip_ratio)
84
+ if len(box) > 1:
85
+ continue
86
+ else:
87
+ continue
88
+ box = box.reshape(-1, 2)
89
+
90
+ _, sside = self.get_mini_boxes(box.reshape((-1, 1, 2)))
91
+ if sside < self.min_size + 2:
92
+ continue
93
+
94
+ box = np.array(box)
95
+ box[:, 0] = np.clip(np.round(box[:, 0] / width * dest_width), 0, dest_width)
96
+ box[:, 1] = np.clip(np.round(box[:, 1] / height * dest_height), 0, dest_height)
97
+ boxes.append(box.tolist())
98
+ scores.append(score)
99
+ return boxes, scores
100
+
101
+ def boxes_from_bitmap(self, pred, _bitmap, dest_width, dest_height):
102
+ """
103
+ _bitmap: single map with shape (1, H, W),
104
+ whose values are binarized as {0, 1}
105
+ """
106
+
107
+ bitmap = _bitmap
108
+ height, width = bitmap.shape
109
+
110
+ outs = cv2.findContours((bitmap * 255).astype(np.uint8), cv2.RETR_LIST, cv2.CHAIN_APPROX_SIMPLE)
111
+ if len(outs) == 3:
112
+ _img, contours, _ = outs[0], outs[1], outs[2]
113
+ elif len(outs) == 2:
114
+ contours, _ = outs[0], outs[1]
115
+
116
+ num_contours = min(len(contours), self.max_candidates)
117
+
118
+ boxes = []
119
+ scores = []
120
+ for index in range(num_contours):
121
+ contour = contours[index]
122
+ points, sside = self.get_mini_boxes(contour)
123
+ if sside < self.min_size:
124
+ continue
125
+ points = np.array(points)
126
+ if self.score_mode == "fast":
127
+ score = self.box_score_fast(pred, points.reshape(-1, 2))
128
+ else:
129
+ score = self.box_score_slow(pred, contour)
130
+ if self.box_thresh > score:
131
+ continue
132
+
133
+ box = self.unclip(points, self.unclip_ratio).reshape(-1, 1, 2)
134
+ box, sside = self.get_mini_boxes(box)
135
+ if sside < self.min_size + 2:
136
+ continue
137
+ box = np.array(box)
138
+
139
+ box[:, 0] = np.clip(np.round(box[:, 0] / width * dest_width), 0, dest_width)
140
+ box[:, 1] = np.clip(np.round(box[:, 1] / height * dest_height), 0, dest_height)
141
+ boxes.append(box.astype("int32"))
142
+ scores.append(score)
143
+ return np.array(boxes, dtype="int32"), scores
144
+
145
+ def unclip(self, box, unclip_ratio):
146
+ poly = Polygon(box)
147
+ distance = poly.area * unclip_ratio / poly.length
148
+ offset = pyclipper.PyclipperOffset()
149
+ offset.AddPath(box, pyclipper.JT_ROUND, pyclipper.ET_CLOSEDPOLYGON)
150
+ expanded = np.array(offset.Execute(distance))
151
+ return expanded
152
+
153
+ def get_mini_boxes(self, contour):
154
+ bounding_box = cv2.minAreaRect(contour)
155
+ points = sorted(list(cv2.boxPoints(bounding_box)), key=lambda x: x[0])
156
+
157
+ index_1, index_2, index_3, index_4 = 0, 1, 2, 3
158
+ if points[1][1] > points[0][1]:
159
+ index_1 = 0
160
+ index_4 = 1
161
+ else:
162
+ index_1 = 1
163
+ index_4 = 0
164
+ if points[3][1] > points[2][1]:
165
+ index_2 = 2
166
+ index_3 = 3
167
+ else:
168
+ index_2 = 3
169
+ index_3 = 2
170
+
171
+ box = [points[index_1], points[index_2], points[index_3], points[index_4]]
172
+ return box, min(bounding_box[1])
173
+
174
+ def box_score_fast(self, bitmap, _box):
175
+ """
176
+ box_score_fast: use bbox mean score as the mean score
177
+ """
178
+ h, w = bitmap.shape[:2]
179
+ box = _box.copy()
180
+ xmin = np.clip(np.floor(box[:, 0].min()).astype("int32"), 0, w - 1)
181
+ xmax = np.clip(np.ceil(box[:, 0].max()).astype("int32"), 0, w - 1)
182
+ ymin = np.clip(np.floor(box[:, 1].min()).astype("int32"), 0, h - 1)
183
+ ymax = np.clip(np.ceil(box[:, 1].max()).astype("int32"), 0, h - 1)
184
+
185
+ mask = np.zeros((ymax - ymin + 1, xmax - xmin + 1), dtype=np.uint8)
186
+ box[:, 0] = box[:, 0] - xmin
187
+ box[:, 1] = box[:, 1] - ymin
188
+ cv2.fillPoly(mask, box.reshape(1, -1, 2).astype("int32"), 1)
189
+ return cv2.mean(bitmap[ymin : ymax + 1, xmin : xmax + 1], mask)[0]
190
+
191
+ def box_score_slow(self, bitmap, contour):
192
+ """
193
+ box_score_slow: use polygon mean score as the mean score
194
+ """
195
+ h, w = bitmap.shape[:2]
196
+ contour = contour.copy()
197
+ contour = np.reshape(contour, (-1, 2))
198
+
199
+ xmin = np.clip(np.min(contour[:, 0]), 0, w - 1)
200
+ xmax = np.clip(np.max(contour[:, 0]), 0, w - 1)
201
+ ymin = np.clip(np.min(contour[:, 1]), 0, h - 1)
202
+ ymax = np.clip(np.max(contour[:, 1]), 0, h - 1)
203
+
204
+ mask = np.zeros((ymax - ymin + 1, xmax - xmin + 1), dtype=np.uint8)
205
+
206
+ contour[:, 0] = contour[:, 0] - xmin
207
+ contour[:, 1] = contour[:, 1] - ymin
208
+
209
+ cv2.fillPoly(mask, contour.reshape(1, -1, 2).astype("int32"), 1)
210
+ return cv2.mean(bitmap[ymin : ymax + 1, xmin : xmax + 1], mask)[0]
211
+
212
+ def __call__(self, outs_dict, shape_list):
213
+ pred = outs_dict["maps"]
214
+ if not isinstance(pred, np.ndarray):
215
+ pred = pred.numpy()
216
+ pred = pred[:, 0, :, :]
217
+ segmentation = pred > self.thresh
218
+
219
+ boxes_batch = []
220
+ for batch_index in range(pred.shape[0]):
221
+ src_h, src_w, ratio_h, ratio_w = shape_list[batch_index]
222
+ if self.dilation_kernel is not None:
223
+ mask = cv2.dilate(np.array(segmentation[batch_index]).astype(np.uint8), self.dilation_kernel)
224
+ else:
225
+ mask = segmentation[batch_index]
226
+ if self.box_type == "poly":
227
+ boxes, scores = self.polygons_from_bitmap(pred[batch_index], mask, src_w, src_h)
228
+ elif self.box_type == "quad":
229
+ boxes, scores = self.boxes_from_bitmap(pred[batch_index], mask, src_w, src_h)
230
+ else:
231
+ raise ValueError("box_type can only be one of ['quad', 'poly']")
232
+
233
+ boxes_batch.append({"points": boxes})
234
+ return boxes_batch
235
+
236
+
237
+ class BaseRecLabelDecode:
238
+ """Convert between text-label and text-index"""
239
+
240
+ def __init__(self, character_dict_path=None, use_space_char=False):
241
+ self.beg_str = "sos"
242
+ self.end_str = "eos"
243
+ self.reverse = False
244
+ self.character_str = []
245
+
246
+ if character_dict_path is None:
247
+ self.character_str = "0123456789abcdefghijklmnopqrstuvwxyz"
248
+ dict_character = list(self.character_str)
249
+ else:
250
+ with open(character_dict_path, "rb") as fin:
251
+ lines = fin.readlines()
252
+ for line in lines:
253
+ line = line.decode("utf-8").strip("\n").strip("\r\n")
254
+ self.character_str.append(line)
255
+ if use_space_char:
256
+ self.character_str.append(" ")
257
+ dict_character = list(self.character_str)
258
+ if "arabic" in character_dict_path:
259
+ self.reverse = True
260
+
261
+ dict_character = self.add_special_char(dict_character)
262
+ self.dict = {}
263
+ for i, char in enumerate(dict_character):
264
+ self.dict[char] = i
265
+ self.character = dict_character
266
+
267
+ def pred_reverse(self, pred):
268
+ pred_re = []
269
+ c_current = ""
270
+ for c in pred:
271
+ if not bool(re.search("[a-zA-Z0-9 :*./%+-]", c)):
272
+ if c_current != "":
273
+ pred_re.append(c_current)
274
+ pred_re.append(c)
275
+ c_current = ""
276
+ else:
277
+ c_current += c
278
+ if c_current != "":
279
+ pred_re.append(c_current)
280
+
281
+ return "".join(pred_re[::-1])
282
+
283
+ def add_special_char(self, dict_character):
284
+ return dict_character
285
+
286
+ def decode(self, text_index, text_prob=None, is_remove_duplicate=False):
287
+ """convert text-index into text-label."""
288
+ result_list = []
289
+ ignored_tokens = self.get_ignored_tokens()
290
+ batch_size = len(text_index)
291
+ for batch_idx in range(batch_size):
292
+ selection = np.ones(len(text_index[batch_idx]), dtype=bool)
293
+ if is_remove_duplicate:
294
+ selection[1:] = text_index[batch_idx][1:] != text_index[batch_idx][:-1]
295
+ for ignored_token in ignored_tokens:
296
+ selection &= text_index[batch_idx] != ignored_token
297
+
298
+ char_list = [self.character[text_id] for text_id in text_index[batch_idx][selection]]
299
+ if text_prob is not None:
300
+ conf_list = text_prob[batch_idx][selection]
301
+ else:
302
+ conf_list = [1] * len(selection)
303
+ if len(conf_list) == 0:
304
+ conf_list = [0]
305
+
306
+ text = "".join(char_list)
307
+
308
+ if self.reverse: # for arabic rec
309
+ text = self.pred_reverse(text)
310
+
311
+ result_list.append((text, np.mean(conf_list).tolist()))
312
+ return result_list
313
+
314
+ def get_ignored_tokens(self):
315
+ return [0] # for ctc blank
316
+
317
+
318
+ class CTCLabelDecode(BaseRecLabelDecode):
319
+ """Convert between text-label and text-index"""
320
+
321
+ def __init__(self, character_dict_path=None, use_space_char=False, **kwargs):
322
+ super(CTCLabelDecode, self).__init__(character_dict_path, use_space_char)
323
+
324
+ def __call__(self, preds, label=None, *args, **kwargs):
325
+ if isinstance(preds, tuple) or isinstance(preds, list):
326
+ preds = preds[-1]
327
+ if not isinstance(preds, np.ndarray):
328
+ preds = preds.numpy()
329
+ preds_idx = preds.argmax(axis=2)
330
+ preds_prob = preds.max(axis=2)
331
+ text = self.decode(preds_idx, preds_prob, is_remove_duplicate=True)
332
+ if label is None:
333
+ return text
334
+ label = self.decode(label)
335
+ return text, label
336
+
337
+ def add_special_char(self, dict_character):
338
+ dict_character = ["blank"] + dict_character
339
+ return dict_character