langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,55 @@
1
+ """
2
+ Ported from RAGFlow's deepdoc module (Apache-2.0):
3
+ https://github.com/infiniflow/ragflow/tree/main/deepdoc
4
+ Copyright 2025 The InfiniFlow Authors.
5
+
6
+ Ported near-verbatim: geometry, OCR, layout-recognition, and
7
+ table-structure-recognition logic is unchanged. Removed or replaced:
8
+ - All non-PDF format parsers, the resume parser, and the deepdoc_server
9
+ FastAPI service (out of scope -- langparse has its own docx/excel/
10
+ markdown parsers).
11
+ - VisionParser and PlainParser (VisionParser needs an LLM and RAGFlow's
12
+ DB/service stack; PlainParser is a no-OCR pypdf fallback, redundant with
13
+ langparse's own `simple` engine).
14
+ - Ascend NPU code paths and the remote DLA HTTP client branch (this port is
15
+ CPU/ONNX-only).
16
+ - The XGBoost up/down line-merge classifier (updown_cnt_mdl /
17
+ _updown_concat_features) -- confirmed dead code on the live call path in
18
+ the source revision this was ported from: _concat_downward() returns
19
+ immediately after its first two lines.
20
+ - rag_tokenizer (a thin wrapper around a tokenizer bundled in the
21
+ infinity-sdk vector-DB client) -- replaced with tokenizer.py, a small
22
+ jieba-backed shim covering the same call sites (is_chinese/tokenize/tag).
23
+ - common.*/rag.* cross-package imports -- replaced with local equivalents
24
+ (see model_loader.py for model directory resolution).
25
+
26
+ operators.py and postprocess.py are themselves derived from PaddleOCR
27
+ (Apache-2.0) upstream in RAGFlow; that attribution carries through here too.
28
+ """
29
+
30
+ __all__ = ["OCR", "LayoutRecognizer", "Recognizer", "TableStructureRecognizer", "RAGFlowPdfParser"]
31
+
32
+ # Lazy (PEP 562) exports: importing this package must stay cheap so that
33
+ # lightweight submodules (model_loader.py, rendering.py, tokenizer.py) can be
34
+ # imported on their own -- e.g. under a `pip install -e ".[dev]"`-only
35
+ # environment -- without transitively pulling in sklearn/cv2/onnxruntime via
36
+ # pdf_parser.py's own heavy dependency chain. Python always runs a package's
37
+ # __init__.py before any of its submodules, so eager `from .x import Y` here
38
+ # would make every submodule import pay that cost.
39
+ _EXPORTS = {
40
+ "OCR": (".ocr", "OCR"),
41
+ "LayoutRecognizer": (".layout_recognizer", "LayoutRecognizer4YOLOv10"),
42
+ "Recognizer": (".recognizer", "Recognizer"),
43
+ "TableStructureRecognizer": (".table_structure_recognizer", "TableStructureRecognizer"),
44
+ "RAGFlowPdfParser": (".pdf_parser", "RAGFlowPdfParser"),
45
+ }
46
+
47
+
48
+ def __getattr__(name):
49
+ if name in _EXPORTS:
50
+ import importlib
51
+
52
+ module_name, attr_name = _EXPORTS[name]
53
+ module = importlib.import_module(module_name, __name__)
54
+ return getattr(module, attr_name)
55
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,235 @@
1
+ #
2
+ # Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ #
16
+
17
+ import logging
18
+ import re
19
+ from collections import Counter
20
+ from copy import deepcopy
21
+
22
+ import cv2
23
+ import numpy as np
24
+
25
+ from .model_loader import default_model_dir
26
+ from .recognizer import Recognizer
27
+ from .operators import nms
28
+
29
+
30
+ class LayoutRecognizer(Recognizer):
31
+ labels = [
32
+ "_background_",
33
+ "Text",
34
+ "Title",
35
+ "Figure",
36
+ "Figure caption",
37
+ "Table",
38
+ "Table caption",
39
+ "Header",
40
+ "Footer",
41
+ "Reference",
42
+ "Equation",
43
+ ]
44
+
45
+ def __init__(self, domain, model_dir=None):
46
+ self.garbage_layouts = ["footer", "header", "reference"]
47
+ self.client = None
48
+
49
+ model_dir = model_dir or str(default_model_dir())
50
+ super().__init__(self.labels, domain, model_dir)
51
+
52
+ def __call__(self, image_list, ocr_res, scale_factor=3, thr=0.2, batch_size=16, drop=True):
53
+ def __is_garbage(b):
54
+ patt = [r"\(cid\s*:\s*\d+\s*\)"]
55
+ return any([re.search(p, b.get("text", "")) for p in patt])
56
+
57
+ if self.client:
58
+ layouts = self.client.predict(image_list)
59
+ else:
60
+ layouts = super().__call__(image_list, thr, batch_size)
61
+ # save_results(image_list, layouts, self.labels, output_dir='output/', threshold=0.7)
62
+ assert len(image_list) == len(ocr_res)
63
+ # Tag layout type
64
+ boxes = []
65
+ assert len(image_list) == len(layouts)
66
+ garbages = {}
67
+ page_layout = []
68
+ for pn, lts in enumerate(layouts):
69
+ bxs = ocr_res[pn]
70
+ lts = [
71
+ {
72
+ "type": b["type"],
73
+ "score": float(b["score"]),
74
+ "x0": b["bbox"][0] / scale_factor,
75
+ "x1": b["bbox"][2] / scale_factor,
76
+ "top": b["bbox"][1] / scale_factor,
77
+ "bottom": b["bbox"][-1] / scale_factor,
78
+ "page_number": pn,
79
+ }
80
+ for b in lts
81
+ if float(b["score"]) >= 0.4 or b["type"] not in self.garbage_layouts
82
+ ]
83
+ lts = self.sort_Y_firstly(lts, np.mean([lt["bottom"] - lt["top"] for lt in lts]) / 2)
84
+ lts = self.layouts_cleanup(bxs, lts)
85
+ page_layout.append(lts)
86
+
87
+ def findLayout(ty):
88
+ nonlocal bxs, lts, self
89
+ lts_ = [lt for lt in lts if lt["type"] == ty]
90
+ i = 0
91
+ while i < len(bxs):
92
+ if bxs[i].get("layout_type"):
93
+ i += 1
94
+ continue
95
+ if __is_garbage(bxs[i]):
96
+ bxs.pop(i)
97
+ continue
98
+
99
+ ii = self.find_overlapped_with_threshold(bxs[i], lts_, thr=0.4)
100
+ if ii is None:
101
+ bxs[i]["layout_type"] = ""
102
+ i += 1
103
+ continue
104
+ lts_[ii]["visited"] = True
105
+ keep_feats = [
106
+ lts_[ii]["type"] == "footer" and bxs[i]["bottom"] < image_list[pn].size[1] * 0.9 / scale_factor,
107
+ lts_[ii]["type"] == "header" and bxs[i]["top"] > image_list[pn].size[1] * 0.1 / scale_factor,
108
+ ]
109
+ if drop and lts_[ii]["type"] in self.garbage_layouts and not any(keep_feats):
110
+ if lts_[ii]["type"] not in garbages:
111
+ garbages[lts_[ii]["type"]] = []
112
+ garbages[lts_[ii]["type"]].append(bxs[i]["text"])
113
+ bxs.pop(i)
114
+ continue
115
+
116
+ bxs[i]["layoutno"] = f"{ty}-{ii}"
117
+ bxs[i]["layout_type"] = lts_[ii]["type"] if lts_[ii]["type"] != "equation" else "figure"
118
+ i += 1
119
+
120
+ for lt in ["footer", "header", "reference", "figure caption", "table caption", "title", "table", "text", "figure", "equation"]:
121
+ findLayout(lt)
122
+
123
+ # add box to figure/equation layouts which have no text box.
124
+ # Index within each type's own list and keep the type as the layoutno
125
+ # prefix so these match the namespace findLayout() assigns to
126
+ # text-overlapping boxes (figure-N vs equation-N). Using a combined
127
+ # figure+equation index with a fixed "figure" prefix would collide
128
+ # with figure-N tags from findLayout and merge unrelated regions.
129
+ for ty in ["figure", "equation"]:
130
+ for i, lt in enumerate([lt for lt in lts if lt["type"] == ty]):
131
+ if lt.get("visited"):
132
+ continue
133
+ lt = deepcopy(lt)
134
+ lt.pop("type", None)
135
+ lt["text"] = ""
136
+ lt["layout_type"] = "figure"
137
+ lt["layoutno"] = f"{ty}-{i}"
138
+ logging.debug(f"Created placeholder box {lt['layoutno']} for textless {ty} region")
139
+ bxs.append(lt)
140
+
141
+ boxes.extend(bxs)
142
+
143
+ ocr_res = boxes
144
+
145
+ garbag_set = set()
146
+ for k in garbages.keys():
147
+ garbages[k] = Counter(garbages[k])
148
+ for g, c in garbages[k].items():
149
+ if c > 1:
150
+ garbag_set.add(g)
151
+
152
+ ocr_res = [b for b in ocr_res if b["text"].strip() not in garbag_set]
153
+ return ocr_res, page_layout
154
+
155
+ def forward(self, image_list, thr=0.7, batch_size=16):
156
+ return super().__call__(image_list, thr, batch_size)
157
+
158
+
159
+ class LayoutRecognizer4YOLOv10(LayoutRecognizer):
160
+ labels = [
161
+ "title",
162
+ "Text",
163
+ "Reference",
164
+ "Figure",
165
+ "Figure caption",
166
+ "Table",
167
+ "Table caption",
168
+ "Table caption",
169
+ "Equation",
170
+ "Figure caption",
171
+ ]
172
+
173
+ def __init__(self, domain, model_dir=None):
174
+ domain = "layout"
175
+ super().__init__(domain, model_dir)
176
+ self.auto = False
177
+ self.scaleFill = False
178
+ self.scaleup = True
179
+ self.stride = 32
180
+ self.center = True
181
+
182
+ def preprocess(self, image_list):
183
+ inputs = []
184
+ new_shape = self.input_shape # height, width
185
+ for img in image_list:
186
+ shape = img.shape[:2] # current shape [height, width]
187
+ # Scale ratio (new / old)
188
+ r = min(new_shape[0] / shape[0], new_shape[1] / shape[1])
189
+ # Compute padding
190
+ new_unpad = int(round(shape[1] * r)), int(round(shape[0] * r))
191
+ dw, dh = new_shape[1] - new_unpad[0], new_shape[0] - new_unpad[1] # wh padding
192
+ dw /= 2 # divide padding into 2 sides
193
+ dh /= 2
194
+ ww, hh = new_unpad
195
+ img = np.array(cv2.cvtColor(img, cv2.COLOR_BGR2RGB)).astype(np.float32)
196
+ img = cv2.resize(img, new_unpad, interpolation=cv2.INTER_LINEAR)
197
+ top, bottom = int(round(dh - 0.1)) if self.center else 0, int(round(dh + 0.1))
198
+ left, right = int(round(dw - 0.1)) if self.center else 0, int(round(dw + 0.1))
199
+ img = cv2.copyMakeBorder(img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(114, 114, 114)) # add border
200
+ img /= 255.0
201
+ img = img.transpose(2, 0, 1)
202
+ img = img[np.newaxis, :, :, :].astype(np.float32)
203
+ inputs.append({self.input_names[0]: img, "scale_factor": [shape[1] / ww, shape[0] / hh, dw, dh]})
204
+
205
+ return inputs
206
+
207
+ def postprocess(self, boxes, inputs, thr):
208
+ thr = 0.08
209
+ boxes = np.squeeze(boxes)
210
+ scores = boxes[:, 4]
211
+ boxes = boxes[scores > thr, :]
212
+ scores = scores[scores > thr]
213
+ if len(boxes) == 0:
214
+ return []
215
+ class_ids = boxes[:, -1].astype(int)
216
+ boxes = boxes[:, :4]
217
+ boxes[:, 0] -= inputs["scale_factor"][2]
218
+ boxes[:, 2] -= inputs["scale_factor"][2]
219
+ boxes[:, 1] -= inputs["scale_factor"][3]
220
+ boxes[:, 3] -= inputs["scale_factor"][3]
221
+ input_shape = np.array([inputs["scale_factor"][0], inputs["scale_factor"][1], inputs["scale_factor"][0], inputs["scale_factor"][1]])
222
+ boxes = np.multiply(boxes, input_shape, dtype=np.float32)
223
+
224
+ unique_class_ids = np.unique(class_ids)
225
+ indices = []
226
+ for class_id in unique_class_ids:
227
+ class_indices = np.where(class_ids == class_id)[0]
228
+ class_boxes = boxes[class_indices, :]
229
+ class_scores = scores[class_indices]
230
+ class_keep_boxes = nms(class_boxes, class_scores, 0.45)
231
+ indices.extend(class_indices[class_keep_boxes])
232
+
233
+ return [{"type": self.label_list[class_ids[i]].lower(), "bbox": [float(t) for t in boxes[i].tolist()], "score": float(scores[i])} for i in indices]
234
+
235
+
@@ -0,0 +1,101 @@
1
+ """
2
+ Model directory resolution for the deepdoc port, mirroring MinerUEngine's
3
+ model_dir/download_dir/model_policy semantics (see
4
+ langparse/engines/pdf/mineru_service.py) instead of upstream deepdoc's
5
+ per-class try/except-then-snapshot_download pattern repeated four times.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ from pathlib import Path
12
+
13
+ DEEPDOC_REPO_ID = "InfiniFlow/deepdoc"
14
+ # Immutable upstream commit resolved from the Hugging Face model metadata.
15
+ # Updating model weights is a deliberate release change, never an implicit pull.
16
+ DEEPDOC_MODEL_REVISION = "de0e793dc6d744406c96dabd688ccc969f41b443"
17
+ REQUIRED_MODEL_FILES = ("det.onnx", "rec.onnx", "layout.onnx", "tsr.onnx", "ocr.res")
18
+ DEEPDOC_MODEL_SHA256 = {
19
+ "det.onnx": "30a86f5731181461d08021402766601e4302a9b9b9666be8aff402696339cdff",
20
+ "rec.onnx": "1c7cf60de2afd728d512f4190cf37455092b45f06175365c6fc58d8cd7e2a68b",
21
+ "layout.onnx": "de401c03ee30b1c120416dc06f0705237f0c36d3cdb692c9bfefe8a8f98a4b70",
22
+ "tsr.onnx": "1585f88015c60209f16a079a26d944afca790ab7022fe7d0574113ccb9a6f9b4",
23
+ "ocr.res": "28b2362ad4ab2dc38769aa72feb535e3a9ddb3fd2a7585a05920e6393b1dc7f7",
24
+ }
25
+
26
+
27
+ def default_model_dir() -> Path:
28
+ return Path.home() / ".langparse" / "models" / "deepdoc"
29
+
30
+
31
+ def _has_required_files(model_dir: Path) -> bool:
32
+ return _model_validation_error(model_dir) is None
33
+
34
+
35
+ def _sha256(file_path: Path) -> str:
36
+ digest = hashlib.sha256()
37
+ with file_path.open("rb") as model_file:
38
+ for chunk in iter(lambda: model_file.read(1024 * 1024), b""):
39
+ digest.update(chunk)
40
+ return digest.hexdigest()
41
+
42
+
43
+ def _model_validation_error(model_dir: Path) -> str | None:
44
+ missing = [name for name in REQUIRED_MODEL_FILES if not (model_dir / name).is_file()]
45
+ if missing:
46
+ return f"missing required files: {tuple(missing)}"
47
+
48
+ for name, expected in DEEPDOC_MODEL_SHA256.items():
49
+ actual = _sha256(model_dir / name)
50
+ if actual != expected:
51
+ return f"checksum mismatch for {name}: expected {expected}, got {actual}"
52
+ return None
53
+
54
+
55
+ def download_models(local_dir: Path) -> Path:
56
+ from huggingface_hub import snapshot_download
57
+
58
+ downloaded = snapshot_download(
59
+ repo_id=DEEPDOC_REPO_ID,
60
+ revision=DEEPDOC_MODEL_REVISION,
61
+ local_dir=str(local_dir),
62
+ )
63
+ return Path(downloaded)
64
+
65
+
66
+ def ensure_deepdoc_models(
67
+ model_dir: str | None = None,
68
+ download_dir: str | None = None,
69
+ model_policy: str = "download_if_missing",
70
+ ) -> str:
71
+ if model_policy not in ("download_if_missing", "require_existing"):
72
+ raise ValueError(
73
+ f"Unsupported deepdoc model_policy: {model_policy}. "
74
+ "Expected 'download_if_missing' or 'require_existing'."
75
+ )
76
+
77
+ if model_dir:
78
+ target = Path(model_dir).expanduser()
79
+ validation_error = _model_validation_error(target)
80
+ if validation_error is not None:
81
+ raise RuntimeError(
82
+ f"deepdoc model_dir has missing or invalid required files under {target}: "
83
+ f"{validation_error}"
84
+ )
85
+ return str(target)
86
+
87
+ target = Path(download_dir).expanduser() if download_dir else default_model_dir()
88
+ if _has_required_files(target):
89
+ return str(target)
90
+
91
+ if model_policy == "require_existing":
92
+ raise RuntimeError(
93
+ f"deepdoc model_policy=require_existing but models are missing under {target}"
94
+ )
95
+
96
+ target.mkdir(parents=True, exist_ok=True)
97
+ downloaded = download_models(target)
98
+ validation_error = _model_validation_error(downloaded)
99
+ if validation_error is not None:
100
+ raise RuntimeError(f"Downloaded deepdoc models failed verification: {validation_error}")
101
+ return str(downloaded)