langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
langparse/cli.py
ADDED
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import sys
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from langparse import __version__
|
|
9
|
+
from langparse.chunkers.registry import available_chunkers
|
|
10
|
+
from langparse.errors import classify_exception
|
|
11
|
+
from langparse.progress import ProgressEvent
|
|
12
|
+
from langparse.services.batch_service import BatchParseService
|
|
13
|
+
from langparse.services.benchmark_service import BenchmarkService
|
|
14
|
+
from langparse.services.parse_service import ParseService
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def build_parser():
|
|
18
|
+
parser = argparse.ArgumentParser(
|
|
19
|
+
prog="langparse",
|
|
20
|
+
description="Parse documents and evaluate parsing quality.",
|
|
21
|
+
)
|
|
22
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
23
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
24
|
+
|
|
25
|
+
parse_cmd = subparsers.add_parser(
|
|
26
|
+
"parse",
|
|
27
|
+
help="parse one document or a batch of documents",
|
|
28
|
+
)
|
|
29
|
+
parse_cmd.add_argument("inputs", nargs="+")
|
|
30
|
+
parse_cmd.add_argument("--engine", default=None)
|
|
31
|
+
parse_cmd.add_argument("--device", default=None)
|
|
32
|
+
parse_cmd.add_argument("--model-dir", default=None)
|
|
33
|
+
parse_cmd.add_argument("--download-dir", default=None)
|
|
34
|
+
parse_cmd.add_argument("--api-url", default=None)
|
|
35
|
+
parse_cmd.add_argument("--api-host", default=None)
|
|
36
|
+
parse_cmd.add_argument("--api-port", type=int, default=None)
|
|
37
|
+
parse_cmd.add_argument("--api-command", default=None)
|
|
38
|
+
parse_cmd.add_argument("--api-start-timeout", type=float, default=None)
|
|
39
|
+
parse_cmd.add_argument("--mineru-request-timeout", type=float, default=None)
|
|
40
|
+
parse_cmd.add_argument("--mineru-backend", default=None)
|
|
41
|
+
parse_cmd.add_argument("--mineru-server-url", default=None)
|
|
42
|
+
parse_cmd.add_argument(
|
|
43
|
+
"--model-policy", choices=["download_if_missing", "require_existing"], default=None
|
|
44
|
+
)
|
|
45
|
+
parse_cmd.add_argument("--model-source", default=None)
|
|
46
|
+
parse_cmd.add_argument("--auto-install-runtime", action="store_true")
|
|
47
|
+
parse_cmd.add_argument("--runtime-package", default=None)
|
|
48
|
+
parse_cmd.add_argument("--format", default="markdown", help="markdown, json, or workbook-json")
|
|
49
|
+
parse_cmd.add_argument("--batch", action="store_true")
|
|
50
|
+
parse_cmd.add_argument("--output", default=None)
|
|
51
|
+
parse_cmd.add_argument("--output-dir", default=None)
|
|
52
|
+
parse_cmd.add_argument("--max-workers", type=int, default=None)
|
|
53
|
+
parse_cmd.add_argument("--skip-existing", action="store_true")
|
|
54
|
+
parse_cmd.add_argument("--metrics", action="store_true")
|
|
55
|
+
parse_cmd.add_argument("--progress", action="store_true", help="write progress to stderr")
|
|
56
|
+
parse_cmd.add_argument(
|
|
57
|
+
"--chunk",
|
|
58
|
+
action="store_true",
|
|
59
|
+
help="semantically chunk the parsed document and include chunks in the output",
|
|
60
|
+
)
|
|
61
|
+
parse_cmd.add_argument("--chunk-strategy", choices=available_chunkers(), default=None)
|
|
62
|
+
parse_cmd.add_argument(
|
|
63
|
+
"--chunk-size",
|
|
64
|
+
type=int,
|
|
65
|
+
default=None,
|
|
66
|
+
help="strategy size budget (lexical tokens for fixed-token, characters otherwise)",
|
|
67
|
+
)
|
|
68
|
+
parse_cmd.add_argument("--chunk-overlap", type=int, default=None)
|
|
69
|
+
parse_cmd.add_argument(
|
|
70
|
+
"--chunk-profile",
|
|
71
|
+
choices=["retrieval", "analysis"],
|
|
72
|
+
default="retrieval",
|
|
73
|
+
help="choose retrieval-oriented or analysis-oriented chunks",
|
|
74
|
+
)
|
|
75
|
+
parse_cmd.add_argument(
|
|
76
|
+
"--model",
|
|
77
|
+
nargs="?",
|
|
78
|
+
const="",
|
|
79
|
+
default=None,
|
|
80
|
+
help="enable model-assisted parsing (reads OPENAI_MODEL from env if model name omitted)",
|
|
81
|
+
)
|
|
82
|
+
parse_cmd.add_argument("--base-url", default=None, help="override OPENAI_BASE_URL")
|
|
83
|
+
parse_cmd.add_argument(
|
|
84
|
+
"--disambiguation",
|
|
85
|
+
choices=["off", "auto", "required"],
|
|
86
|
+
default=None,
|
|
87
|
+
help="workbook ambiguity resolution mode",
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
benchmark_cmd = subparsers.add_parser(
|
|
91
|
+
"benchmark",
|
|
92
|
+
help="run the general parsing benchmark",
|
|
93
|
+
)
|
|
94
|
+
benchmark_cmd.add_argument("manifest")
|
|
95
|
+
benchmark_cmd.add_argument("--engine", default=None)
|
|
96
|
+
benchmark_cmd.add_argument("--output-dir", default="reports")
|
|
97
|
+
benchmark_cmd.add_argument("--format", default="json")
|
|
98
|
+
benchmark_cmd.add_argument("--max-workers", type=int, default=1)
|
|
99
|
+
benchmark_cmd.add_argument("--api-url", default=None)
|
|
100
|
+
benchmark_cmd.add_argument("--mineru-request-timeout", type=float, default=None)
|
|
101
|
+
benchmark_cmd.add_argument("--mineru-backend", default=None)
|
|
102
|
+
benchmark_cmd.add_argument("--mineru-server-url", default=None)
|
|
103
|
+
benchmark_cmd.add_argument("--device", default=None)
|
|
104
|
+
benchmark_cmd.add_argument("--model-dir", default=None)
|
|
105
|
+
benchmark_cmd.add_argument("--download-dir", default=None)
|
|
106
|
+
benchmark_cmd.add_argument("--auto-install-runtime", action="store_true")
|
|
107
|
+
benchmark_cmd.add_argument("--runtime-package", default=None)
|
|
108
|
+
|
|
109
|
+
eval_cmd = subparsers.add_parser(
|
|
110
|
+
"benchmark-workbook-ambiguity",
|
|
111
|
+
aliases=["eval", "eval-excel"],
|
|
112
|
+
help="evaluate workbook ambiguity handling from a manifest",
|
|
113
|
+
description=(
|
|
114
|
+
"Evaluate deterministic and optional live-model workbook ambiguity handling. "
|
|
115
|
+
"API keys are read from OPENAI_API_KEY, never command-line arguments."
|
|
116
|
+
),
|
|
117
|
+
)
|
|
118
|
+
eval_cmd.add_argument("manifest")
|
|
119
|
+
eval_cmd.add_argument("--output-dir", default="reports/workbook-ambiguity")
|
|
120
|
+
eval_cmd.add_argument(
|
|
121
|
+
"--no-markdown", action="store_true", help="disable markdown summary output"
|
|
122
|
+
)
|
|
123
|
+
eval_cmd.add_argument(
|
|
124
|
+
"--model",
|
|
125
|
+
nargs="?",
|
|
126
|
+
const="",
|
|
127
|
+
default=None,
|
|
128
|
+
help="enable live model evaluation (reads OPENAI_MODEL from env if name omitted)",
|
|
129
|
+
)
|
|
130
|
+
eval_cmd.add_argument("--base-url", default=None, help="override OPENAI_BASE_URL")
|
|
131
|
+
|
|
132
|
+
quality_cmd = subparsers.add_parser(
|
|
133
|
+
"benchmark-workbook-quality",
|
|
134
|
+
aliases=["eval-workbook"],
|
|
135
|
+
help="evaluate complete workbook structure from a versioned manifest",
|
|
136
|
+
)
|
|
137
|
+
quality_cmd.add_argument("manifest")
|
|
138
|
+
quality_cmd.add_argument("--output-dir", default="reports/workbook-quality")
|
|
139
|
+
quality_cmd.add_argument(
|
|
140
|
+
"--no-markdown", action="store_true", help="disable markdown summary output"
|
|
141
|
+
)
|
|
142
|
+
return parser
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
146
|
+
parser = build_parser()
|
|
147
|
+
args = parser.parse_args(argv)
|
|
148
|
+
|
|
149
|
+
try:
|
|
150
|
+
return _run(args, parser)
|
|
151
|
+
except Exception as exc: # noqa: BLE001 - CLI boundary: report, never traceback
|
|
152
|
+
classified = classify_exception(exc)
|
|
153
|
+
print(f"langparse: {classified.error_type.value}: {classified.message}", file=sys.stderr)
|
|
154
|
+
return 2
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _run(args, parser) -> int:
|
|
158
|
+
if args.command in ("benchmark-workbook-quality", "eval-workbook"):
|
|
159
|
+
from langparse.services.workbook_quality_benchmark import (
|
|
160
|
+
WorkbookQualityBenchmarkService,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
report = WorkbookQualityBenchmarkService().run(
|
|
164
|
+
args.manifest,
|
|
165
|
+
output_dir=args.output_dir,
|
|
166
|
+
markdown=not args.no_markdown,
|
|
167
|
+
)
|
|
168
|
+
print(
|
|
169
|
+
"langparse: workbook quality benchmark completed "
|
|
170
|
+
f"with status={report.summary['status']} ({report.run_digest})"
|
|
171
|
+
)
|
|
172
|
+
return 0 if report.summary["status"] == "passed" else 1
|
|
173
|
+
|
|
174
|
+
if args.command in ("benchmark-workbook-ambiguity", "eval", "eval-excel"):
|
|
175
|
+
from langparse.services.workbook_ambiguity_benchmark import (
|
|
176
|
+
WorkbookAmbiguityBenchmarkService,
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
report = WorkbookAmbiguityBenchmarkService().run(
|
|
180
|
+
args.manifest,
|
|
181
|
+
output_dir=args.output_dir,
|
|
182
|
+
markdown=not args.no_markdown,
|
|
183
|
+
model=args.model,
|
|
184
|
+
base_url=args.base_url,
|
|
185
|
+
)
|
|
186
|
+
print(f"langparse: workbook ambiguity benchmark completed ({report.run_digest})")
|
|
187
|
+
return 0
|
|
188
|
+
|
|
189
|
+
if args.command == "benchmark":
|
|
190
|
+
benchmark_kwargs = {
|
|
191
|
+
key: value
|
|
192
|
+
for key, value in {
|
|
193
|
+
"api_url": args.api_url,
|
|
194
|
+
"request_timeout": args.mineru_request_timeout,
|
|
195
|
+
"backend": args.mineru_backend,
|
|
196
|
+
"server_url": args.mineru_server_url,
|
|
197
|
+
"device": args.device,
|
|
198
|
+
"model_dir": args.model_dir,
|
|
199
|
+
"download_dir": args.download_dir,
|
|
200
|
+
"auto_install_runtime": args.auto_install_runtime,
|
|
201
|
+
"runtime_package": args.runtime_package,
|
|
202
|
+
}.items()
|
|
203
|
+
if value is not None and value is not False
|
|
204
|
+
}
|
|
205
|
+
BenchmarkService().run(
|
|
206
|
+
args.manifest,
|
|
207
|
+
output_dir=args.output_dir,
|
|
208
|
+
engine_name=args.engine,
|
|
209
|
+
fmt=args.format,
|
|
210
|
+
max_workers=args.max_workers,
|
|
211
|
+
**benchmark_kwargs,
|
|
212
|
+
)
|
|
213
|
+
return 0
|
|
214
|
+
|
|
215
|
+
if args.command != "parse":
|
|
216
|
+
parser.error(f"Unsupported command: {args.command}")
|
|
217
|
+
|
|
218
|
+
service = ParseService()
|
|
219
|
+
engine_name = args.engine or "simple"
|
|
220
|
+
disambiguation_mode = args.disambiguation or ("auto" if args.model is not None else None)
|
|
221
|
+
model_kwargs = {
|
|
222
|
+
key: value
|
|
223
|
+
for key, value in {
|
|
224
|
+
"workbook_disambiguation": disambiguation_mode,
|
|
225
|
+
"model": args.model,
|
|
226
|
+
"base_url": args.base_url,
|
|
227
|
+
}.items()
|
|
228
|
+
if value is not None
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
parse_kwargs = {
|
|
232
|
+
key: value
|
|
233
|
+
for key, value in {
|
|
234
|
+
"device": args.device,
|
|
235
|
+
"model_dir": args.model_dir,
|
|
236
|
+
"download_dir": args.download_dir,
|
|
237
|
+
"api_url": args.api_url,
|
|
238
|
+
"api_host": args.api_host,
|
|
239
|
+
"api_port": args.api_port,
|
|
240
|
+
"api_command": args.api_command,
|
|
241
|
+
"api_start_timeout": args.api_start_timeout,
|
|
242
|
+
"request_timeout": args.mineru_request_timeout,
|
|
243
|
+
"backend": args.mineru_backend,
|
|
244
|
+
"server_url": args.mineru_server_url,
|
|
245
|
+
"model_policy": args.model_policy,
|
|
246
|
+
"model_source": args.model_source,
|
|
247
|
+
"auto_install_runtime": args.auto_install_runtime,
|
|
248
|
+
"runtime_package": args.runtime_package,
|
|
249
|
+
**model_kwargs,
|
|
250
|
+
}.items()
|
|
251
|
+
if value is not None and value is not False
|
|
252
|
+
}
|
|
253
|
+
chunk_kwargs = {"chunk_profile": args.chunk_profile} if args.chunk else {}
|
|
254
|
+
if any(
|
|
255
|
+
value is not None for value in (args.chunk_strategy, args.chunk_size, args.chunk_overlap)
|
|
256
|
+
):
|
|
257
|
+
if not args.chunk:
|
|
258
|
+
parser.error("chunk strategy/size/overlap options require --chunk")
|
|
259
|
+
chunk_kwargs["chunk_strategy"] = args.chunk_strategy or "semantic"
|
|
260
|
+
chunk_kwargs["chunk_options"] = {
|
|
261
|
+
key: value
|
|
262
|
+
for key, value in {
|
|
263
|
+
"max_chunk_size": args.chunk_size,
|
|
264
|
+
"overlap": args.chunk_overlap,
|
|
265
|
+
}.items()
|
|
266
|
+
if value is not None
|
|
267
|
+
}
|
|
268
|
+
if args.progress:
|
|
269
|
+
parse_kwargs["progress_callback"] = _print_progress
|
|
270
|
+
|
|
271
|
+
if args.batch:
|
|
272
|
+
# One implementation regardless of flags. Without --output-dir the run
|
|
273
|
+
# renders to memory and prints; with it, outputs and reports are written.
|
|
274
|
+
result = BatchParseService().run(
|
|
275
|
+
args.inputs,
|
|
276
|
+
engine_name=engine_name,
|
|
277
|
+
output_dir=args.output_dir,
|
|
278
|
+
fmt=args.format,
|
|
279
|
+
max_workers=args.max_workers,
|
|
280
|
+
skip_existing=args.skip_existing,
|
|
281
|
+
collect_metrics=args.metrics,
|
|
282
|
+
chunk=args.chunk,
|
|
283
|
+
**chunk_kwargs,
|
|
284
|
+
**parse_kwargs,
|
|
285
|
+
)
|
|
286
|
+
for rendered in result.rendered_outputs:
|
|
287
|
+
print(rendered)
|
|
288
|
+
return 0
|
|
289
|
+
|
|
290
|
+
if len(args.inputs) != 1:
|
|
291
|
+
parser.error(
|
|
292
|
+
"Single parse mode accepts exactly one input. Use --batch for multiple inputs."
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
rendered = service.parse_output(
|
|
296
|
+
args.inputs[0],
|
|
297
|
+
engine_name=engine_name,
|
|
298
|
+
fmt=args.format,
|
|
299
|
+
chunk=args.chunk,
|
|
300
|
+
**chunk_kwargs,
|
|
301
|
+
**parse_kwargs,
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
if args.output:
|
|
305
|
+
service.write_output(rendered, Path(args.output))
|
|
306
|
+
else:
|
|
307
|
+
print(rendered)
|
|
308
|
+
|
|
309
|
+
return 0
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _print_progress(event: ProgressEvent) -> None:
|
|
313
|
+
counts = ""
|
|
314
|
+
if event.completed_units is not None:
|
|
315
|
+
total = event.total_units if event.total_units is not None else "?"
|
|
316
|
+
counts = f" {event.completed_units}/{total} {event.unit or ''}"
|
|
317
|
+
percent = f" {event.percent:.0f}%" if event.percent is not None else ""
|
|
318
|
+
# Escape embedded line breaks/control characters from untrusted filenames/messages.
|
|
319
|
+
source = repr(event.source) if event.source else "batch"
|
|
320
|
+
message = f" {event.message!r}" if event.message else ""
|
|
321
|
+
print(
|
|
322
|
+
f"langparse: {source} {event.phase} {event.state}{counts}{percent}{message}",
|
|
323
|
+
file=sys.stderr,
|
|
324
|
+
flush=True,
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
if __name__ == "__main__":
|
|
329
|
+
raise SystemExit(main())
|
langparse/config.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
import copy
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from langparse.logging import get_logger
|
|
8
|
+
|
|
9
|
+
logger = get_logger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Config:
|
|
13
|
+
"""
|
|
14
|
+
Global configuration manager for LangParse.
|
|
15
|
+
Priorities:
|
|
16
|
+
1. Runtime kwargs (passed to functions)
|
|
17
|
+
2. Environment variables (LANGPARSE_*)
|
|
18
|
+
3. Config file (~/.langparse/config.json)
|
|
19
|
+
4. Defaults
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
ENV_MAP = {
|
|
23
|
+
"LANGPARSE_DEFAULT_PDF_ENGINE": "default_pdf_engine",
|
|
24
|
+
"LANGPARSE_MINERU_DEVICE": "engines.mineru.device",
|
|
25
|
+
"LANGPARSE_MINERU_MODEL_DIR": "engines.mineru.model_dir",
|
|
26
|
+
"LANGPARSE_MINERU_DOWNLOAD_DIR": "engines.mineru.download_dir",
|
|
27
|
+
"LANGPARSE_MINERU_ENABLE_OCR": "engines.mineru.enable_ocr",
|
|
28
|
+
"LANGPARSE_MINERU_API_URL": "engines.mineru.api_url",
|
|
29
|
+
"LANGPARSE_MINERU_API_HOST": "engines.mineru.api_host",
|
|
30
|
+
"LANGPARSE_MINERU_API_PORT": "engines.mineru.api_port",
|
|
31
|
+
"LANGPARSE_MINERU_API_COMMAND": "engines.mineru.api_command",
|
|
32
|
+
"LANGPARSE_MINERU_API_START_TIMEOUT": "engines.mineru.api_start_timeout",
|
|
33
|
+
"LANGPARSE_MINERU_REQUEST_TIMEOUT": "engines.mineru.request_timeout",
|
|
34
|
+
"LANGPARSE_MINERU_BACKEND": "engines.mineru.backend",
|
|
35
|
+
"LANGPARSE_MINERU_SERVER_URL": "engines.mineru.server_url",
|
|
36
|
+
"LANGPARSE_MINERU_MODEL_POLICY": "engines.mineru.model_policy",
|
|
37
|
+
"LANGPARSE_MINERU_MODEL_SOURCE": "engines.mineru.model_source",
|
|
38
|
+
"LANGPARSE_MINERU_AUTO_INSTALL_RUNTIME": "engines.mineru.auto_install_runtime",
|
|
39
|
+
"LANGPARSE_MINERU_RUNTIME_PACKAGE": "engines.mineru.runtime_package",
|
|
40
|
+
"LANGPARSE_DEEPDOC_DEVICE": "engines.deepdoc.device",
|
|
41
|
+
"LANGPARSE_DEEPDOC_MODEL_DIR": "engines.deepdoc.model_dir",
|
|
42
|
+
"LANGPARSE_DEEPDOC_DOWNLOAD_DIR": "engines.deepdoc.download_dir",
|
|
43
|
+
"LANGPARSE_DEEPDOC_MODEL_POLICY": "engines.deepdoc.model_policy",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
DEFAULT_CONFIG = {
|
|
47
|
+
"default_pdf_engine": "simple",
|
|
48
|
+
"engines": {
|
|
49
|
+
"mineru": {
|
|
50
|
+
"device": "auto",
|
|
51
|
+
"model_dir": None,
|
|
52
|
+
"download_dir": None,
|
|
53
|
+
"enable_ocr": True,
|
|
54
|
+
"api_url": None,
|
|
55
|
+
"api_host": "127.0.0.1",
|
|
56
|
+
"api_port": 8000,
|
|
57
|
+
"api_command": "mineru-api",
|
|
58
|
+
"api_start_timeout": 30.0,
|
|
59
|
+
"request_timeout": 300.0,
|
|
60
|
+
"backend": None,
|
|
61
|
+
"server_url": None,
|
|
62
|
+
"model_policy": "download_if_missing",
|
|
63
|
+
"model_source": None,
|
|
64
|
+
"auto_install_runtime": False,
|
|
65
|
+
"runtime_package": "mineru>=3.4,<4",
|
|
66
|
+
"extra_options": {},
|
|
67
|
+
},
|
|
68
|
+
"vision_llm": {
|
|
69
|
+
"provider": "openai",
|
|
70
|
+
"model": "gpt-4o",
|
|
71
|
+
"api_key": None,
|
|
72
|
+
},
|
|
73
|
+
"deepdoc": {
|
|
74
|
+
"device": "cpu",
|
|
75
|
+
"model_dir": None,
|
|
76
|
+
"download_dir": None,
|
|
77
|
+
"model_policy": "download_if_missing",
|
|
78
|
+
},
|
|
79
|
+
},
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
def __init__(self):
|
|
83
|
+
self._config = copy.deepcopy(self.DEFAULT_CONFIG)
|
|
84
|
+
self._load_from_file()
|
|
85
|
+
self._load_from_env()
|
|
86
|
+
|
|
87
|
+
def _load_from_file(self):
|
|
88
|
+
config_path = Path.home() / ".langparse" / "config.json"
|
|
89
|
+
if config_path.exists():
|
|
90
|
+
try:
|
|
91
|
+
with open(config_path, encoding="utf-8") as f:
|
|
92
|
+
user_config = json.load(f)
|
|
93
|
+
self._merge_dict(self._config, user_config)
|
|
94
|
+
except Exception as e:
|
|
95
|
+
logger.warning("Failed to load config file %s: %s", config_path, e)
|
|
96
|
+
|
|
97
|
+
def _load_from_env(self):
|
|
98
|
+
for env_key, config_key in self.ENV_MAP.items():
|
|
99
|
+
if env_key not in os.environ:
|
|
100
|
+
continue
|
|
101
|
+
current_value = self.get(config_key)
|
|
102
|
+
parsed_value = self._parse_env_value(os.environ[env_key], current_value)
|
|
103
|
+
self._set_nested_value(config_key, parsed_value)
|
|
104
|
+
|
|
105
|
+
def _parse_env_value(self, value: str, current_value: Any = None) -> Any:
|
|
106
|
+
normalized = value.strip()
|
|
107
|
+
lowered = normalized.lower()
|
|
108
|
+
|
|
109
|
+
if isinstance(current_value, bool):
|
|
110
|
+
if lowered in {"true", "1", "yes", "on"}:
|
|
111
|
+
return True
|
|
112
|
+
if lowered in {"false", "0", "no", "off"}:
|
|
113
|
+
return False
|
|
114
|
+
|
|
115
|
+
if current_value is None and lowered in {"null", "none", ""}:
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
try:
|
|
119
|
+
return json.loads(normalized)
|
|
120
|
+
except json.JSONDecodeError:
|
|
121
|
+
return normalized
|
|
122
|
+
|
|
123
|
+
def _set_nested_value(self, key: str, value: Any) -> None:
|
|
124
|
+
keys = key.split(".")
|
|
125
|
+
target = self._config
|
|
126
|
+
for part in keys[:-1]:
|
|
127
|
+
if part not in target or not isinstance(target[part], dict):
|
|
128
|
+
target[part] = {}
|
|
129
|
+
target = target[part]
|
|
130
|
+
target[keys[-1]] = value
|
|
131
|
+
|
|
132
|
+
def _merge_dict(self, base: dict, update: dict):
|
|
133
|
+
for k, v in update.items():
|
|
134
|
+
if k in base and isinstance(base[k], dict) and isinstance(v, dict):
|
|
135
|
+
self._merge_dict(base[k], v)
|
|
136
|
+
else:
|
|
137
|
+
base[k] = v
|
|
138
|
+
|
|
139
|
+
def resolve_engine_config(
|
|
140
|
+
self, engine_name: str, runtime_kwargs: dict[str, Any]
|
|
141
|
+
) -> dict[str, Any]:
|
|
142
|
+
config_key = f"engines.{engine_name}"
|
|
143
|
+
engine_config = self.get(config_key, {})
|
|
144
|
+
resolved_config = {**engine_config, **runtime_kwargs}
|
|
145
|
+
|
|
146
|
+
config_extra_options = engine_config.get("extra_options")
|
|
147
|
+
runtime_extra_options = runtime_kwargs.get("extra_options")
|
|
148
|
+
if isinstance(config_extra_options, dict) and isinstance(runtime_extra_options, dict):
|
|
149
|
+
resolved_config["extra_options"] = {
|
|
150
|
+
**config_extra_options,
|
|
151
|
+
**runtime_extra_options,
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
return resolved_config
|
|
155
|
+
|
|
156
|
+
def get(self, key: str, default: Any = None) -> Any:
|
|
157
|
+
"""Get a config value using dot notation, e.g. 'engines.mineru.model_dir'"""
|
|
158
|
+
keys = key.split(".")
|
|
159
|
+
val = self._config
|
|
160
|
+
for k in keys:
|
|
161
|
+
if isinstance(val, dict) and k in val:
|
|
162
|
+
val = val[k]
|
|
163
|
+
else:
|
|
164
|
+
return default
|
|
165
|
+
return val
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# Singleton instance
|
|
169
|
+
settings = Config()
|
|
File without changes
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
|
|
3
|
+
from langparse.types import Chunk, Document
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class BaseChunker(ABC):
|
|
7
|
+
"""
|
|
8
|
+
Abstract base class for all text chunkers.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
@abstractmethod
|
|
12
|
+
def chunk(self, document: Document, **kwargs) -> list[Chunk]:
|
|
13
|
+
"""
|
|
14
|
+
Split a Document into a list of Chunks.
|
|
15
|
+
"""
|
|
16
|
+
pass
|
langparse/core/engine.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from collections.abc import Iterator
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from langparse.types import ParsedElement, StructuredData
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class PageResult:
|
|
11
|
+
"""
|
|
12
|
+
Engine-facing iterative page result yielded during parsing.
|
|
13
|
+
Mirrors the normalized parsed page shape so engines can stream page data
|
|
14
|
+
before document assembly without carrying a second incompatible contract.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
page_number: int
|
|
18
|
+
markdown_content: str
|
|
19
|
+
plain_text: str = ""
|
|
20
|
+
elements: list[ParsedElement] = field(default_factory=list)
|
|
21
|
+
tables: list[StructuredData] = field(default_factory=list)
|
|
22
|
+
images: list[StructuredData] = field(default_factory=list)
|
|
23
|
+
metadata: StructuredData = field(default_factory=dict)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class BaseEngine(ABC):
|
|
27
|
+
"""
|
|
28
|
+
Abstract base class for all parsing engines.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
@abstractmethod
|
|
32
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
33
|
+
"""
|
|
34
|
+
Process a file and yield results page by page.
|
|
35
|
+
This allows for streaming processing of large documents.
|
|
36
|
+
"""
|
|
37
|
+
pass
|
langparse/core/parser.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from langparse.core.rendering import document_from_result
|
|
5
|
+
from langparse.types import Document, ParsedDocumentResult
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class BaseParser(ABC):
|
|
9
|
+
"""
|
|
10
|
+
Abstract base class for all document parsers.
|
|
11
|
+
|
|
12
|
+
Parsers implement `parse_result`, which returns the structured
|
|
13
|
+
`ParsedDocumentResult` that metrics, quality checks and batch reporting all
|
|
14
|
+
read. `parse` is the flat Markdown view rendered from that same result, so
|
|
15
|
+
the two can never disagree.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@abstractmethod
|
|
19
|
+
def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
20
|
+
"""
|
|
21
|
+
Parse a file into its structured page/element/table representation.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def parse(self, file_path: str | Path, **kwargs) -> Document:
|
|
25
|
+
"""
|
|
26
|
+
Parse a file and return the rendered Markdown Document.
|
|
27
|
+
"""
|
|
28
|
+
return document_from_result(self.parse_result(file_path, **kwargs))
|
|
29
|
+
|
|
30
|
+
@staticmethod
|
|
31
|
+
def _resolve_existing_path(file_path: str | Path) -> Path:
|
|
32
|
+
path = Path(file_path)
|
|
33
|
+
if not path.exists():
|
|
34
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
35
|
+
return path
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from langparse.types import Document, ParsedDocumentResult
|
|
6
|
+
|
|
7
|
+
PAGE_MARKER_TEMPLATE = "<!-- page_number: {page_number} -->"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def document_from_result(parsed: ParsedDocumentResult) -> Document:
|
|
11
|
+
"""
|
|
12
|
+
Render the structured parse result into the flat Markdown ``Document`` that
|
|
13
|
+
chunkers and end users consume.
|
|
14
|
+
|
|
15
|
+
Page markers are injected only for paginated results. A flow format such as
|
|
16
|
+
plain Markdown has no page boundaries to mark, and injecting a fake one
|
|
17
|
+
would corrupt any markers the source already carries.
|
|
18
|
+
"""
|
|
19
|
+
if parsed.paginated:
|
|
20
|
+
blocks: list[str] = []
|
|
21
|
+
for page in parsed.pages:
|
|
22
|
+
blocks.append(f"\n{PAGE_MARKER_TEMPLATE.format(page_number=page.page_number)}\n")
|
|
23
|
+
blocks.append(page.markdown_content)
|
|
24
|
+
content = "\n".join(blocks)
|
|
25
|
+
else:
|
|
26
|
+
content = "\n".join(page.markdown_content for page in parsed.pages)
|
|
27
|
+
|
|
28
|
+
return Document(content=content, metadata=document_metadata(parsed))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def document_metadata(parsed: ParsedDocumentResult) -> dict:
|
|
32
|
+
"""
|
|
33
|
+
Build the Document metadata.
|
|
34
|
+
|
|
35
|
+
Engine-specific detail stays nested under ``parsed_metadata`` on purpose:
|
|
36
|
+
SemanticChunker copies this dict into every chunk, so flattening MinerU's
|
|
37
|
+
runtime fields here would duplicate them across every chunk written to a
|
|
38
|
+
vector store.
|
|
39
|
+
"""
|
|
40
|
+
return {
|
|
41
|
+
"source": parsed.source,
|
|
42
|
+
"filename": parsed.filename,
|
|
43
|
+
"extension": parsed.metadata.get("extension") or Path(parsed.filename).suffix,
|
|
44
|
+
"engine": parsed.engine,
|
|
45
|
+
# Copied, not aliased: SemanticChunker shallow-copies this dict into
|
|
46
|
+
# every chunk, so sharing one instance would let a later mutation of the
|
|
47
|
+
# parse result reach through into already-emitted chunks.
|
|
48
|
+
"parsed_metadata": dict(parsed.metadata),
|
|
49
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Engines package
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# PDF Engines package
|