langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Ported from RAGFlow's deepdoc module (Apache-2.0):
|
|
3
|
+
https://github.com/infiniflow/ragflow/tree/main/deepdoc
|
|
4
|
+
Copyright 2025 The InfiniFlow Authors.
|
|
5
|
+
|
|
6
|
+
Ported near-verbatim: geometry, OCR, layout-recognition, and
|
|
7
|
+
table-structure-recognition logic is unchanged. Removed or replaced:
|
|
8
|
+
- All non-PDF format parsers, the resume parser, and the deepdoc_server
|
|
9
|
+
FastAPI service (out of scope -- langparse has its own docx/excel/
|
|
10
|
+
markdown parsers).
|
|
11
|
+
- VisionParser and PlainParser (VisionParser needs an LLM and RAGFlow's
|
|
12
|
+
DB/service stack; PlainParser is a no-OCR pypdf fallback, redundant with
|
|
13
|
+
langparse's own `simple` engine).
|
|
14
|
+
- Ascend NPU code paths and the remote DLA HTTP client branch (this port is
|
|
15
|
+
CPU/ONNX-only).
|
|
16
|
+
- The XGBoost up/down line-merge classifier (updown_cnt_mdl /
|
|
17
|
+
_updown_concat_features) -- confirmed dead code on the live call path in
|
|
18
|
+
the source revision this was ported from: _concat_downward() returns
|
|
19
|
+
immediately after its first two lines.
|
|
20
|
+
- rag_tokenizer (a thin wrapper around a tokenizer bundled in the
|
|
21
|
+
infinity-sdk vector-DB client) -- replaced with tokenizer.py, a small
|
|
22
|
+
jieba-backed shim covering the same call sites (is_chinese/tokenize/tag).
|
|
23
|
+
- common.*/rag.* cross-package imports -- replaced with local equivalents
|
|
24
|
+
(see model_loader.py for model directory resolution).
|
|
25
|
+
|
|
26
|
+
operators.py and postprocess.py are themselves derived from PaddleOCR
|
|
27
|
+
(Apache-2.0) upstream in RAGFlow; that attribution carries through here too.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
__all__ = ["OCR", "LayoutRecognizer", "Recognizer", "TableStructureRecognizer", "RAGFlowPdfParser"]
|
|
31
|
+
|
|
32
|
+
# Lazy (PEP 562) exports: importing this package must stay cheap so that
|
|
33
|
+
# lightweight submodules (model_loader.py, rendering.py, tokenizer.py) can be
|
|
34
|
+
# imported on their own -- e.g. under a `pip install -e ".[dev]"`-only
|
|
35
|
+
# environment -- without transitively pulling in sklearn/cv2/onnxruntime via
|
|
36
|
+
# pdf_parser.py's own heavy dependency chain. Python always runs a package's
|
|
37
|
+
# __init__.py before any of its submodules, so eager `from .x import Y` here
|
|
38
|
+
# would make every submodule import pay that cost.
|
|
39
|
+
_EXPORTS = {
|
|
40
|
+
"OCR": (".ocr", "OCR"),
|
|
41
|
+
"LayoutRecognizer": (".layout_recognizer", "LayoutRecognizer4YOLOv10"),
|
|
42
|
+
"Recognizer": (".recognizer", "Recognizer"),
|
|
43
|
+
"TableStructureRecognizer": (".table_structure_recognizer", "TableStructureRecognizer"),
|
|
44
|
+
"RAGFlowPdfParser": (".pdf_parser", "RAGFlowPdfParser"),
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def __getattr__(name):
|
|
49
|
+
if name in _EXPORTS:
|
|
50
|
+
import importlib
|
|
51
|
+
|
|
52
|
+
module_name, attr_name = _EXPORTS[name]
|
|
53
|
+
module = importlib.import_module(module_name, __name__)
|
|
54
|
+
return getattr(module, attr_name)
|
|
55
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
#
|
|
16
|
+
|
|
17
|
+
import logging
|
|
18
|
+
import re
|
|
19
|
+
from collections import Counter
|
|
20
|
+
from copy import deepcopy
|
|
21
|
+
|
|
22
|
+
import cv2
|
|
23
|
+
import numpy as np
|
|
24
|
+
|
|
25
|
+
from .model_loader import default_model_dir
|
|
26
|
+
from .recognizer import Recognizer
|
|
27
|
+
from .operators import nms
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class LayoutRecognizer(Recognizer):
|
|
31
|
+
labels = [
|
|
32
|
+
"_background_",
|
|
33
|
+
"Text",
|
|
34
|
+
"Title",
|
|
35
|
+
"Figure",
|
|
36
|
+
"Figure caption",
|
|
37
|
+
"Table",
|
|
38
|
+
"Table caption",
|
|
39
|
+
"Header",
|
|
40
|
+
"Footer",
|
|
41
|
+
"Reference",
|
|
42
|
+
"Equation",
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
def __init__(self, domain, model_dir=None):
|
|
46
|
+
self.garbage_layouts = ["footer", "header", "reference"]
|
|
47
|
+
self.client = None
|
|
48
|
+
|
|
49
|
+
model_dir = model_dir or str(default_model_dir())
|
|
50
|
+
super().__init__(self.labels, domain, model_dir)
|
|
51
|
+
|
|
52
|
+
def __call__(self, image_list, ocr_res, scale_factor=3, thr=0.2, batch_size=16, drop=True):
|
|
53
|
+
def __is_garbage(b):
|
|
54
|
+
patt = [r"\(cid\s*:\s*\d+\s*\)"]
|
|
55
|
+
return any([re.search(p, b.get("text", "")) for p in patt])
|
|
56
|
+
|
|
57
|
+
if self.client:
|
|
58
|
+
layouts = self.client.predict(image_list)
|
|
59
|
+
else:
|
|
60
|
+
layouts = super().__call__(image_list, thr, batch_size)
|
|
61
|
+
# save_results(image_list, layouts, self.labels, output_dir='output/', threshold=0.7)
|
|
62
|
+
assert len(image_list) == len(ocr_res)
|
|
63
|
+
# Tag layout type
|
|
64
|
+
boxes = []
|
|
65
|
+
assert len(image_list) == len(layouts)
|
|
66
|
+
garbages = {}
|
|
67
|
+
page_layout = []
|
|
68
|
+
for pn, lts in enumerate(layouts):
|
|
69
|
+
bxs = ocr_res[pn]
|
|
70
|
+
lts = [
|
|
71
|
+
{
|
|
72
|
+
"type": b["type"],
|
|
73
|
+
"score": float(b["score"]),
|
|
74
|
+
"x0": b["bbox"][0] / scale_factor,
|
|
75
|
+
"x1": b["bbox"][2] / scale_factor,
|
|
76
|
+
"top": b["bbox"][1] / scale_factor,
|
|
77
|
+
"bottom": b["bbox"][-1] / scale_factor,
|
|
78
|
+
"page_number": pn,
|
|
79
|
+
}
|
|
80
|
+
for b in lts
|
|
81
|
+
if float(b["score"]) >= 0.4 or b["type"] not in self.garbage_layouts
|
|
82
|
+
]
|
|
83
|
+
lts = self.sort_Y_firstly(lts, np.mean([lt["bottom"] - lt["top"] for lt in lts]) / 2)
|
|
84
|
+
lts = self.layouts_cleanup(bxs, lts)
|
|
85
|
+
page_layout.append(lts)
|
|
86
|
+
|
|
87
|
+
def findLayout(ty):
|
|
88
|
+
nonlocal bxs, lts, self
|
|
89
|
+
lts_ = [lt for lt in lts if lt["type"] == ty]
|
|
90
|
+
i = 0
|
|
91
|
+
while i < len(bxs):
|
|
92
|
+
if bxs[i].get("layout_type"):
|
|
93
|
+
i += 1
|
|
94
|
+
continue
|
|
95
|
+
if __is_garbage(bxs[i]):
|
|
96
|
+
bxs.pop(i)
|
|
97
|
+
continue
|
|
98
|
+
|
|
99
|
+
ii = self.find_overlapped_with_threshold(bxs[i], lts_, thr=0.4)
|
|
100
|
+
if ii is None:
|
|
101
|
+
bxs[i]["layout_type"] = ""
|
|
102
|
+
i += 1
|
|
103
|
+
continue
|
|
104
|
+
lts_[ii]["visited"] = True
|
|
105
|
+
keep_feats = [
|
|
106
|
+
lts_[ii]["type"] == "footer" and bxs[i]["bottom"] < image_list[pn].size[1] * 0.9 / scale_factor,
|
|
107
|
+
lts_[ii]["type"] == "header" and bxs[i]["top"] > image_list[pn].size[1] * 0.1 / scale_factor,
|
|
108
|
+
]
|
|
109
|
+
if drop and lts_[ii]["type"] in self.garbage_layouts and not any(keep_feats):
|
|
110
|
+
if lts_[ii]["type"] not in garbages:
|
|
111
|
+
garbages[lts_[ii]["type"]] = []
|
|
112
|
+
garbages[lts_[ii]["type"]].append(bxs[i]["text"])
|
|
113
|
+
bxs.pop(i)
|
|
114
|
+
continue
|
|
115
|
+
|
|
116
|
+
bxs[i]["layoutno"] = f"{ty}-{ii}"
|
|
117
|
+
bxs[i]["layout_type"] = lts_[ii]["type"] if lts_[ii]["type"] != "equation" else "figure"
|
|
118
|
+
i += 1
|
|
119
|
+
|
|
120
|
+
for lt in ["footer", "header", "reference", "figure caption", "table caption", "title", "table", "text", "figure", "equation"]:
|
|
121
|
+
findLayout(lt)
|
|
122
|
+
|
|
123
|
+
# add box to figure/equation layouts which have no text box.
|
|
124
|
+
# Index within each type's own list and keep the type as the layoutno
|
|
125
|
+
# prefix so these match the namespace findLayout() assigns to
|
|
126
|
+
# text-overlapping boxes (figure-N vs equation-N). Using a combined
|
|
127
|
+
# figure+equation index with a fixed "figure" prefix would collide
|
|
128
|
+
# with figure-N tags from findLayout and merge unrelated regions.
|
|
129
|
+
for ty in ["figure", "equation"]:
|
|
130
|
+
for i, lt in enumerate([lt for lt in lts if lt["type"] == ty]):
|
|
131
|
+
if lt.get("visited"):
|
|
132
|
+
continue
|
|
133
|
+
lt = deepcopy(lt)
|
|
134
|
+
lt.pop("type", None)
|
|
135
|
+
lt["text"] = ""
|
|
136
|
+
lt["layout_type"] = "figure"
|
|
137
|
+
lt["layoutno"] = f"{ty}-{i}"
|
|
138
|
+
logging.debug(f"Created placeholder box {lt['layoutno']} for textless {ty} region")
|
|
139
|
+
bxs.append(lt)
|
|
140
|
+
|
|
141
|
+
boxes.extend(bxs)
|
|
142
|
+
|
|
143
|
+
ocr_res = boxes
|
|
144
|
+
|
|
145
|
+
garbag_set = set()
|
|
146
|
+
for k in garbages.keys():
|
|
147
|
+
garbages[k] = Counter(garbages[k])
|
|
148
|
+
for g, c in garbages[k].items():
|
|
149
|
+
if c > 1:
|
|
150
|
+
garbag_set.add(g)
|
|
151
|
+
|
|
152
|
+
ocr_res = [b for b in ocr_res if b["text"].strip() not in garbag_set]
|
|
153
|
+
return ocr_res, page_layout
|
|
154
|
+
|
|
155
|
+
def forward(self, image_list, thr=0.7, batch_size=16):
|
|
156
|
+
return super().__call__(image_list, thr, batch_size)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class LayoutRecognizer4YOLOv10(LayoutRecognizer):
|
|
160
|
+
labels = [
|
|
161
|
+
"title",
|
|
162
|
+
"Text",
|
|
163
|
+
"Reference",
|
|
164
|
+
"Figure",
|
|
165
|
+
"Figure caption",
|
|
166
|
+
"Table",
|
|
167
|
+
"Table caption",
|
|
168
|
+
"Table caption",
|
|
169
|
+
"Equation",
|
|
170
|
+
"Figure caption",
|
|
171
|
+
]
|
|
172
|
+
|
|
173
|
+
def __init__(self, domain, model_dir=None):
|
|
174
|
+
domain = "layout"
|
|
175
|
+
super().__init__(domain, model_dir)
|
|
176
|
+
self.auto = False
|
|
177
|
+
self.scaleFill = False
|
|
178
|
+
self.scaleup = True
|
|
179
|
+
self.stride = 32
|
|
180
|
+
self.center = True
|
|
181
|
+
|
|
182
|
+
def preprocess(self, image_list):
|
|
183
|
+
inputs = []
|
|
184
|
+
new_shape = self.input_shape # height, width
|
|
185
|
+
for img in image_list:
|
|
186
|
+
shape = img.shape[:2] # current shape [height, width]
|
|
187
|
+
# Scale ratio (new / old)
|
|
188
|
+
r = min(new_shape[0] / shape[0], new_shape[1] / shape[1])
|
|
189
|
+
# Compute padding
|
|
190
|
+
new_unpad = int(round(shape[1] * r)), int(round(shape[0] * r))
|
|
191
|
+
dw, dh = new_shape[1] - new_unpad[0], new_shape[0] - new_unpad[1] # wh padding
|
|
192
|
+
dw /= 2 # divide padding into 2 sides
|
|
193
|
+
dh /= 2
|
|
194
|
+
ww, hh = new_unpad
|
|
195
|
+
img = np.array(cv2.cvtColor(img, cv2.COLOR_BGR2RGB)).astype(np.float32)
|
|
196
|
+
img = cv2.resize(img, new_unpad, interpolation=cv2.INTER_LINEAR)
|
|
197
|
+
top, bottom = int(round(dh - 0.1)) if self.center else 0, int(round(dh + 0.1))
|
|
198
|
+
left, right = int(round(dw - 0.1)) if self.center else 0, int(round(dw + 0.1))
|
|
199
|
+
img = cv2.copyMakeBorder(img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(114, 114, 114)) # add border
|
|
200
|
+
img /= 255.0
|
|
201
|
+
img = img.transpose(2, 0, 1)
|
|
202
|
+
img = img[np.newaxis, :, :, :].astype(np.float32)
|
|
203
|
+
inputs.append({self.input_names[0]: img, "scale_factor": [shape[1] / ww, shape[0] / hh, dw, dh]})
|
|
204
|
+
|
|
205
|
+
return inputs
|
|
206
|
+
|
|
207
|
+
def postprocess(self, boxes, inputs, thr):
|
|
208
|
+
thr = 0.08
|
|
209
|
+
boxes = np.squeeze(boxes)
|
|
210
|
+
scores = boxes[:, 4]
|
|
211
|
+
boxes = boxes[scores > thr, :]
|
|
212
|
+
scores = scores[scores > thr]
|
|
213
|
+
if len(boxes) == 0:
|
|
214
|
+
return []
|
|
215
|
+
class_ids = boxes[:, -1].astype(int)
|
|
216
|
+
boxes = boxes[:, :4]
|
|
217
|
+
boxes[:, 0] -= inputs["scale_factor"][2]
|
|
218
|
+
boxes[:, 2] -= inputs["scale_factor"][2]
|
|
219
|
+
boxes[:, 1] -= inputs["scale_factor"][3]
|
|
220
|
+
boxes[:, 3] -= inputs["scale_factor"][3]
|
|
221
|
+
input_shape = np.array([inputs["scale_factor"][0], inputs["scale_factor"][1], inputs["scale_factor"][0], inputs["scale_factor"][1]])
|
|
222
|
+
boxes = np.multiply(boxes, input_shape, dtype=np.float32)
|
|
223
|
+
|
|
224
|
+
unique_class_ids = np.unique(class_ids)
|
|
225
|
+
indices = []
|
|
226
|
+
for class_id in unique_class_ids:
|
|
227
|
+
class_indices = np.where(class_ids == class_id)[0]
|
|
228
|
+
class_boxes = boxes[class_indices, :]
|
|
229
|
+
class_scores = scores[class_indices]
|
|
230
|
+
class_keep_boxes = nms(class_boxes, class_scores, 0.45)
|
|
231
|
+
indices.extend(class_indices[class_keep_boxes])
|
|
232
|
+
|
|
233
|
+
return [{"type": self.label_list[class_ids[i]].lower(), "bbox": [float(t) for t in boxes[i].tolist()], "score": float(scores[i])} for i in indices]
|
|
234
|
+
|
|
235
|
+
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Model directory resolution for the deepdoc port, mirroring MinerUEngine's
|
|
3
|
+
model_dir/download_dir/model_policy semantics (see
|
|
4
|
+
langparse/engines/pdf/mineru_service.py) instead of upstream deepdoc's
|
|
5
|
+
per-class try/except-then-snapshot_download pattern repeated four times.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
DEEPDOC_REPO_ID = "InfiniFlow/deepdoc"
|
|
14
|
+
# Immutable upstream commit resolved from the Hugging Face model metadata.
|
|
15
|
+
# Updating model weights is a deliberate release change, never an implicit pull.
|
|
16
|
+
DEEPDOC_MODEL_REVISION = "de0e793dc6d744406c96dabd688ccc969f41b443"
|
|
17
|
+
REQUIRED_MODEL_FILES = ("det.onnx", "rec.onnx", "layout.onnx", "tsr.onnx", "ocr.res")
|
|
18
|
+
DEEPDOC_MODEL_SHA256 = {
|
|
19
|
+
"det.onnx": "30a86f5731181461d08021402766601e4302a9b9b9666be8aff402696339cdff",
|
|
20
|
+
"rec.onnx": "1c7cf60de2afd728d512f4190cf37455092b45f06175365c6fc58d8cd7e2a68b",
|
|
21
|
+
"layout.onnx": "de401c03ee30b1c120416dc06f0705237f0c36d3cdb692c9bfefe8a8f98a4b70",
|
|
22
|
+
"tsr.onnx": "1585f88015c60209f16a079a26d944afca790ab7022fe7d0574113ccb9a6f9b4",
|
|
23
|
+
"ocr.res": "28b2362ad4ab2dc38769aa72feb535e3a9ddb3fd2a7585a05920e6393b1dc7f7",
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def default_model_dir() -> Path:
|
|
28
|
+
return Path.home() / ".langparse" / "models" / "deepdoc"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _has_required_files(model_dir: Path) -> bool:
|
|
32
|
+
return _model_validation_error(model_dir) is None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _sha256(file_path: Path) -> str:
|
|
36
|
+
digest = hashlib.sha256()
|
|
37
|
+
with file_path.open("rb") as model_file:
|
|
38
|
+
for chunk in iter(lambda: model_file.read(1024 * 1024), b""):
|
|
39
|
+
digest.update(chunk)
|
|
40
|
+
return digest.hexdigest()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _model_validation_error(model_dir: Path) -> str | None:
|
|
44
|
+
missing = [name for name in REQUIRED_MODEL_FILES if not (model_dir / name).is_file()]
|
|
45
|
+
if missing:
|
|
46
|
+
return f"missing required files: {tuple(missing)}"
|
|
47
|
+
|
|
48
|
+
for name, expected in DEEPDOC_MODEL_SHA256.items():
|
|
49
|
+
actual = _sha256(model_dir / name)
|
|
50
|
+
if actual != expected:
|
|
51
|
+
return f"checksum mismatch for {name}: expected {expected}, got {actual}"
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def download_models(local_dir: Path) -> Path:
|
|
56
|
+
from huggingface_hub import snapshot_download
|
|
57
|
+
|
|
58
|
+
downloaded = snapshot_download(
|
|
59
|
+
repo_id=DEEPDOC_REPO_ID,
|
|
60
|
+
revision=DEEPDOC_MODEL_REVISION,
|
|
61
|
+
local_dir=str(local_dir),
|
|
62
|
+
)
|
|
63
|
+
return Path(downloaded)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def ensure_deepdoc_models(
|
|
67
|
+
model_dir: str | None = None,
|
|
68
|
+
download_dir: str | None = None,
|
|
69
|
+
model_policy: str = "download_if_missing",
|
|
70
|
+
) -> str:
|
|
71
|
+
if model_policy not in ("download_if_missing", "require_existing"):
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"Unsupported deepdoc model_policy: {model_policy}. "
|
|
74
|
+
"Expected 'download_if_missing' or 'require_existing'."
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
if model_dir:
|
|
78
|
+
target = Path(model_dir).expanduser()
|
|
79
|
+
validation_error = _model_validation_error(target)
|
|
80
|
+
if validation_error is not None:
|
|
81
|
+
raise RuntimeError(
|
|
82
|
+
f"deepdoc model_dir has missing or invalid required files under {target}: "
|
|
83
|
+
f"{validation_error}"
|
|
84
|
+
)
|
|
85
|
+
return str(target)
|
|
86
|
+
|
|
87
|
+
target = Path(download_dir).expanduser() if download_dir else default_model_dir()
|
|
88
|
+
if _has_required_files(target):
|
|
89
|
+
return str(target)
|
|
90
|
+
|
|
91
|
+
if model_policy == "require_existing":
|
|
92
|
+
raise RuntimeError(
|
|
93
|
+
f"deepdoc model_policy=require_existing but models are missing under {target}"
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
downloaded = download_models(target)
|
|
98
|
+
validation_error = _model_validation_error(downloaded)
|
|
99
|
+
if validation_error is not None:
|
|
100
|
+
raise RuntimeError(f"Downloaded deepdoc models failed verification: {validation_error}")
|
|
101
|
+
return str(downloaded)
|