langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,790 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: langparse
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A developer-friendly document parsing toolkit with precise, source-grounded Excel understanding.
|
|
5
|
+
Author-email: syw2014 <syw2014@gmail.com>
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/syw2014/langparse
|
|
8
|
+
Project-URL: Repository, https://github.com/syw2014/langparse
|
|
9
|
+
Project-URL: Issues, https://github.com/syw2014/langparse/issues
|
|
10
|
+
Keywords: llm,rag,parsing,chunking,agent,document-parsing,excel,spreadsheet,workbook,xlsx
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
19
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Provides-Extra: pdf
|
|
24
|
+
Requires-Dist: pdfplumber>=0.11.10; extra == "pdf"
|
|
25
|
+
Provides-Extra: docx
|
|
26
|
+
Requires-Dist: python-docx>=1.1.0; extra == "docx"
|
|
27
|
+
Provides-Extra: excel
|
|
28
|
+
Requires-Dist: pandas>=2.0.0; extra == "excel"
|
|
29
|
+
Requires-Dist: openpyxl>=3.1.0; extra == "excel"
|
|
30
|
+
Requires-Dist: tabulate>=0.9.0; extra == "excel"
|
|
31
|
+
Requires-Dist: pillow>=10.0.0; extra == "excel"
|
|
32
|
+
Provides-Extra: model
|
|
33
|
+
Requires-Dist: openai<3.0.0,>=2.0.0; extra == "model"
|
|
34
|
+
Provides-Extra: ocr
|
|
35
|
+
Requires-Dist: rapidocr_onnxruntime>=1.3.0; extra == "ocr"
|
|
36
|
+
Requires-Dist: pdfplumber>=0.11.10; extra == "ocr"
|
|
37
|
+
Provides-Extra: mineru
|
|
38
|
+
Requires-Dist: mineru<4,>=3.4; extra == "mineru"
|
|
39
|
+
Provides-Extra: deepdoc
|
|
40
|
+
Requires-Dist: pdfplumber>=0.11.10; extra == "deepdoc"
|
|
41
|
+
Requires-Dist: opencv-python-headless>=4.9.0; extra == "deepdoc"
|
|
42
|
+
Requires-Dist: onnxruntime>=1.17.0; extra == "deepdoc"
|
|
43
|
+
Requires-Dist: pypdf>=6.19.0; extra == "deepdoc"
|
|
44
|
+
Requires-Dist: huggingface_hub>=0.20.0; extra == "deepdoc"
|
|
45
|
+
Requires-Dist: scikit-learn>=1.3.0; extra == "deepdoc"
|
|
46
|
+
Requires-Dist: shapely>=2.0.0; extra == "deepdoc"
|
|
47
|
+
Requires-Dist: pyclipper>=1.3.0; extra == "deepdoc"
|
|
48
|
+
Requires-Dist: jieba>=0.42.1; extra == "deepdoc"
|
|
49
|
+
Provides-Extra: all
|
|
50
|
+
Requires-Dist: pdfplumber>=0.11.10; extra == "all"
|
|
51
|
+
Requires-Dist: python-docx; extra == "all"
|
|
52
|
+
Requires-Dist: pandas; extra == "all"
|
|
53
|
+
Requires-Dist: openpyxl; extra == "all"
|
|
54
|
+
Requires-Dist: tabulate>=0.9.0; extra == "all"
|
|
55
|
+
Requires-Dist: pillow>=10.0.0; extra == "all"
|
|
56
|
+
Requires-Dist: openai<3.0.0,>=2.0.0; extra == "all"
|
|
57
|
+
Requires-Dist: rapidocr_onnxruntime; extra == "all"
|
|
58
|
+
Requires-Dist: mineru<4,>=3.4; extra == "all"
|
|
59
|
+
Requires-Dist: opencv-python-headless; extra == "all"
|
|
60
|
+
Requires-Dist: onnxruntime; extra == "all"
|
|
61
|
+
Requires-Dist: pypdf>=6.19.0; extra == "all"
|
|
62
|
+
Requires-Dist: huggingface_hub; extra == "all"
|
|
63
|
+
Requires-Dist: scikit-learn; extra == "all"
|
|
64
|
+
Requires-Dist: shapely; extra == "all"
|
|
65
|
+
Requires-Dist: pyclipper; extra == "all"
|
|
66
|
+
Requires-Dist: jieba; extra == "all"
|
|
67
|
+
Provides-Extra: dev
|
|
68
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
69
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
70
|
+
Requires-Dist: ruff>=0.6.0; extra == "dev"
|
|
71
|
+
Requires-Dist: pdfplumber>=0.11.10; extra == "dev"
|
|
72
|
+
Requires-Dist: python-docx>=1.1.0; extra == "dev"
|
|
73
|
+
Requires-Dist: pandas>=2.0.0; extra == "dev"
|
|
74
|
+
Requires-Dist: openpyxl>=3.1.0; extra == "dev"
|
|
75
|
+
Requires-Dist: openai<3.0.0,>=2.0.0; extra == "dev"
|
|
76
|
+
Requires-Dist: tabulate>=0.9.0; extra == "dev"
|
|
77
|
+
Requires-Dist: jieba>=0.42.1; extra == "dev"
|
|
78
|
+
Dynamic: license-file
|
|
79
|
+
|
|
80
|
+
<p align="center">
|
|
81
|
+
<img src="https://raw.githubusercontent.com/syw2014/langparse/main/assets/langparse-mark-512.png" alt="LangParse logo" width="128">
|
|
82
|
+
</p>
|
|
83
|
+
|
|
84
|
+
<h1 align="center">LangParse</h1>
|
|
85
|
+
|
|
86
|
+
<p align="center"><strong>Documents in. Structure out.</strong></p>
|
|
87
|
+
|
|
88
|
+
<p align="center">
|
|
89
|
+
<a href="README_cn.md">简体中文</a>
|
|
90
|
+
</p>
|
|
91
|
+
|
|
92
|
+
<p align="center">
|
|
93
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="Apache 2.0 license"></a>
|
|
94
|
+
<a href="https://pypi.org/project/langparse/"><img src="https://img.shields.io/pypi/v/langparse?include_prereleases" alt="PyPI version"></a>
|
|
95
|
+
<a href="https://github.com/syw2014/langparse/actions"><img src="https://github.com/syw2014/langparse/actions/workflows/tests.yml/badge.svg" alt="Tests"></a>
|
|
96
|
+
</p>
|
|
97
|
+
|
|
98
|
+
LangParse is a developer-friendly Python toolkit for turning documents into
|
|
99
|
+
structured results that programs, agents, and data pipelines can use directly.
|
|
100
|
+
|
|
101
|
+
It has two product pillars:
|
|
102
|
+
|
|
103
|
+
- **Easy document parsing:** one predictable interface for PDF, DOCX, Excel,
|
|
104
|
+
CSV, Markdown, and text, with optional chunking, batch processing, quality
|
|
105
|
+
checks, and pluggable PDF backends.
|
|
106
|
+
- **Precise, rich Excel understanding:** preserve workbook facts and reconstruct
|
|
107
|
+
logical tables, forms, matrices, text regions, and cross-sheet relationships
|
|
108
|
+
instead of flattening a workbook into plain text.
|
|
109
|
+
|
|
110
|
+
PDF and Word support make LangParse useful across a document pipeline. Excel is
|
|
111
|
+
where LangParse goes deliberately deeper.
|
|
112
|
+
|
|
113
|
+
---
|
|
114
|
+
|
|
115
|
+
## Parsing progress (unreleased)
|
|
116
|
+
|
|
117
|
+
Pass `progress_callback` to `AutoParser`, `ParseService`, `PDFParser`, or
|
|
118
|
+
`BatchParseService.run`. The callback receives immutable `ProgressEvent` objects:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
from langparse import AutoParser
|
|
122
|
+
|
|
123
|
+
result = AutoParser.parse_result(
|
|
124
|
+
"budget.xlsx",
|
|
125
|
+
progress_callback=lambda event: print(event.phase, event.state, event.percent),
|
|
126
|
+
)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Events contain `source`, `phase`, `state`, `completed_units`, `total_units`, `unit`,
|
|
130
|
+
`percent`, and `message`. Percent is phase-local (0–100), not an ETA; unknown
|
|
131
|
+
values remain `None`. Only a `file` event with `completed`/`failed` marks the end
|
|
132
|
+
of parsing and requested chunking, before any caller-side export. A completed
|
|
133
|
+
parse can still have partial quality diagnostics.
|
|
134
|
+
|
|
135
|
+
Simple PDF reports page counts; DeepDoc forwards its internal weighted-stage or
|
|
136
|
+
page-batch progress; MinerU reports preparation, remote parsing, and rendering
|
|
137
|
+
without an internal percentage. Excel reports extraction, assembly (including
|
|
138
|
+
optional model disambiguation), and rendering. Direct `ExcelParser` calls also
|
|
139
|
+
emit those stages; the facade/service supplies the file lifecycle.
|
|
140
|
+
|
|
141
|
+
Batch events use `phase="batch"`, `source=""`, and `unit="files"`. Notifications
|
|
142
|
+
follow completion order while returned results retain the batch's sorted file order. Success,
|
|
143
|
+
failure, and skip all count as finished files. Batch completion follows report
|
|
144
|
+
writing and does not mean every file succeeded. Each `batch_item` event names
|
|
145
|
+
the source and reports `completed`/`failed`/`skipped` after that item's output
|
|
146
|
+
write. Empty batches report 0/0 with
|
|
147
|
+
unknown percent. Ordinary callback exceptions are logged by type and isolated
|
|
148
|
+
from parsing. Callbacks are synchronous: keep them brief. One batch serializes
|
|
149
|
+
its callbacks, which can run on different threads; callers coordinate callbacks
|
|
150
|
+
shared across independent runs. There is no task store, background execution,
|
|
151
|
+
cancellation, heartbeat, or HTTP query endpoint.
|
|
152
|
+
|
|
153
|
+
CLI `--progress` writes line-oriented updates to stderr, leaving stdout intact:
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
langparse parse budget.xlsx --format json --progress > result.json
|
|
157
|
+
langparse parse docs/ --batch --output-dir out --progress
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
## Project status
|
|
161
|
+
|
|
162
|
+
The latest GitHub release is `0.1.0`. Core multi-format parsing,
|
|
163
|
+
structured OOXML workbooks, semantic chunking, batch processing, quality checks,
|
|
164
|
+
and CI are available today. LangParse remains pre-1.0; see
|
|
165
|
+
[docs/PROGRESS.md](docs/PROGRESS.md) for the module-by-module source of truth and
|
|
166
|
+
known gaps. See the
|
|
167
|
+
[release scope](docs/RELEASE_0.1.0.md).
|
|
168
|
+
|
|
169
|
+
## Why LangParse?
|
|
170
|
+
|
|
171
|
+
Most document workflows do not need another complicated platform. They need a
|
|
172
|
+
small toolkit that is easy to install, easy to call, and honest about the
|
|
173
|
+
structure it can recover.
|
|
174
|
+
|
|
175
|
+
Excel also needs a different abstraction from PDF and Word. A workbook may
|
|
176
|
+
contain formulas, merged headers, repeated print fragments, forms, matrices,
|
|
177
|
+
hidden rows, comments, links, and tables continued across sheets. Converting it
|
|
178
|
+
straight to Markdown destroys information that later analysis cannot recover.
|
|
179
|
+
|
|
180
|
+
LangParse therefore keeps three layers separate:
|
|
181
|
+
|
|
182
|
+
```mermaid
|
|
183
|
+
flowchart LR
|
|
184
|
+
A["Documents<br/>PDF · DOCX · XLSX · CSV · MD · TXT"] --> B["Simple parsing API"]
|
|
185
|
+
B --> C["Consumable result<br/>Markdown · JSON · chunks"]
|
|
186
|
+
X["Excel / OOXML"] --> F["Workbook facts<br/>cells · formulas · styles · visibility"]
|
|
187
|
+
F --> S["Workbook structure<br/>tables · forms · matrices · relationships"]
|
|
188
|
+
S --> C
|
|
189
|
+
C --> D["Applications<br/>RAG · Agents · data pipelines"]
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
Rich structure is the source of truth. Markdown and retrieval chunks are useful
|
|
193
|
+
views derived from it, not replacements for it.
|
|
194
|
+
|
|
195
|
+
## Core capabilities
|
|
196
|
+
|
|
197
|
+
- **One entry point:** `AutoParser.parse_result(...)` routes supported formats
|
|
198
|
+
through a consistent result contract.
|
|
199
|
+
- **Rich Excel IR:** `.xlsx` and `.xlsm` preserve coordinates, raw and display
|
|
200
|
+
values, formulas, cached values, merges, style fingerprints, visibility,
|
|
201
|
+
dimensions, print areas, comments, hyperlinks, and object anchors.
|
|
202
|
+
- **Semantic workbook reconstruction:** deterministic blocks distinguish
|
|
203
|
+
logical tables, forms, matrices, text, and unclassified regions while keeping
|
|
204
|
+
source ranges and confidence diagnostics.
|
|
205
|
+
- **General document coverage:** Markdown, DOCX, legacy DOC, CSV, text, and PDF;
|
|
206
|
+
PDF currently supports `simple`, MinerU, and DeepDoc backends.
|
|
207
|
+
- **Downstream-ready output:** normalized Markdown/JSON, source-aware chunks,
|
|
208
|
+
batch processing, metrics, and quality checks.
|
|
209
|
+
- **Optional model assistance:** explicit opt-in workbook disambiguation; the
|
|
210
|
+
default path remains offline and deterministic.
|
|
211
|
+
|
|
212
|
+
## Installation
|
|
213
|
+
|
|
214
|
+
Install the current release candidate:
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
pip install "langparse==0.1.0"
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
Install only the optional capabilities you need:
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
pip install "langparse[excel]"
|
|
224
|
+
pip install "langparse[excel,model]" # optional OpenAI workbook disambiguation
|
|
225
|
+
pip install "langparse[deepdoc]"
|
|
226
|
+
pip install "langparse[all]"
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
Calling an existing remote MinerU API needs only the core package. Install
|
|
230
|
+
`langparse[mineru]` only when this Python environment must provide and start a
|
|
231
|
+
local `mineru-api` orchestrator.
|
|
232
|
+
|
|
233
|
+
## Quick start
|
|
234
|
+
|
|
235
|
+
### Parse any supported document
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
from langparse import AutoParser
|
|
239
|
+
|
|
240
|
+
result = AutoParser.parse_result("report.docx")
|
|
241
|
+
|
|
242
|
+
print(result.markdown_content)
|
|
243
|
+
print(result.metadata)
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
The same entry point accepts PDF, DOCX, Excel, CSV, Markdown, and text. For PDF,
|
|
247
|
+
select a backend only when you need one:
|
|
248
|
+
|
|
249
|
+
```python
|
|
250
|
+
result = AutoParser.parse_result("scan.pdf", engine="deepdoc")
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
### Inspect Excel as structure
|
|
254
|
+
|
|
255
|
+
```python
|
|
256
|
+
from langparse import ExcelParser
|
|
257
|
+
from langparse.workbooks import WorkbookIR
|
|
258
|
+
|
|
259
|
+
result = ExcelParser().parse_result("budget.xlsx")
|
|
260
|
+
workbook = result.structure
|
|
261
|
+
assert isinstance(workbook, WorkbookIR)
|
|
262
|
+
assert workbook.snapshot is not None
|
|
263
|
+
|
|
264
|
+
first_sheet = workbook.snapshot.sheets[0]
|
|
265
|
+
print(first_sheet.cells["B2"].formula)
|
|
266
|
+
|
|
267
|
+
for sheet in workbook.sheets:
|
|
268
|
+
for block in sheet.blocks:
|
|
269
|
+
print(block.kind, [ref.key for ref in block.source_refs])
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
The workbook IR remains linked to the original sheet and cell ranges. You can
|
|
273
|
+
derive Markdown or chunks without losing the facts needed for validation and
|
|
274
|
+
analysis.
|
|
275
|
+
|
|
276
|
+
### Use the CLI
|
|
277
|
+
|
|
278
|
+
```bash
|
|
279
|
+
langparse parse report.docx --format markdown
|
|
280
|
+
langparse parse budget.xlsx --format json --chunk
|
|
281
|
+
langparse parse docs/ --batch --chunk --metrics --output-dir out
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
### Chunking
|
|
285
|
+
|
|
286
|
+
Chunks respect a size budget while following Markdown structure. Sections come
|
|
287
|
+
from headings; within a section, blocks are packed up to `max_chunk_size`.
|
|
288
|
+
|
|
289
|
+
```python
|
|
290
|
+
SemanticChunker(max_chunk_size=1000, overlap=0, length_function=len)
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
- **`length_function`** measures chunk size. The default counts characters and
|
|
294
|
+
pulls in no dependencies; pass a tokenizer's encoder to budget in tokens:
|
|
295
|
+
```python
|
|
296
|
+
import tiktoken
|
|
297
|
+
|
|
298
|
+
encoder = tiktoken.get_encoding("cl100k_base")
|
|
299
|
+
SemanticChunker(max_chunk_size=512, length_function=lambda t: len(encoder.encode(t)))
|
|
300
|
+
```
|
|
301
|
+
- **`overlap`** is off by default. It duplicates content into the vector store,
|
|
302
|
+
so it is opt-in.
|
|
303
|
+
- **Tables** that exceed the budget split by row with the header row repeated in
|
|
304
|
+
each part, so every chunk stays readable on its own.
|
|
305
|
+
- **Code blocks** are never split — splitting would leave unterminated fences.
|
|
306
|
+
An oversized one emits whole with `oversized: True` in its metadata.
|
|
307
|
+
- A `#` inside a fenced code block is not treated as a heading.
|
|
308
|
+
|
|
309
|
+
Each chunk carries `header`, `header_level`, `header_path`, `page_numbers` and
|
|
310
|
+
`chunk_index`.
|
|
311
|
+
|
|
312
|
+
From the CLI, `--chunk` adds a `chunks` array to JSON output (and separates
|
|
313
|
+
chunks with `---` in Markdown output), and activates the chunk metrics:
|
|
314
|
+
|
|
315
|
+
```bash
|
|
316
|
+
langparse parse paper.pdf --chunk --format json
|
|
317
|
+
langparse parse docs/ --batch --chunk --metrics --output-dir out
|
|
318
|
+
```
|
|
319
|
+
|
|
320
|
+
### Excel deep dive
|
|
321
|
+
|
|
322
|
+
OOXML workbooks are not treated as paginated pandas tables. Each sheet keeps a
|
|
323
|
+
stable compatibility ordinal, while the result sets `paginated=False` and
|
|
324
|
+
exposes lossless source facts, deterministic logical tables, coverage
|
|
325
|
+
diagnostics, semantic Markdown, and source-aware table-row chunks from one
|
|
326
|
+
parse:
|
|
327
|
+
|
|
328
|
+
```python
|
|
329
|
+
from langparse.services.parse_service import ParseService
|
|
330
|
+
|
|
331
|
+
parsed = ParseService().parse_result(
|
|
332
|
+
"budget.xlsx",
|
|
333
|
+
chunk=True,
|
|
334
|
+
chunk_profile="retrieval",
|
|
335
|
+
)
|
|
336
|
+
analysis_chunks = ParseService().chunk_result(
|
|
337
|
+
parsed,
|
|
338
|
+
chunk_profile="analysis",
|
|
339
|
+
)
|
|
340
|
+
print(parsed.structure.snapshot.sheets[0].cells["B2"].formula)
|
|
341
|
+
print(parsed.diagnostics.coverage_ratio)
|
|
342
|
+
print([block.kind for block in parsed.structure.sheets[0].blocks])
|
|
343
|
+
print(parsed.diagnostics.source_ref_validity_ratio)
|
|
344
|
+
print(parsed.chunks[0].metadata["chunk_type"])
|
|
345
|
+
print(parsed.chunks[0].metadata["source_ranges"])
|
|
346
|
+
print(analysis_chunks[0].structured_payload.get("records"))
|
|
347
|
+
|
|
348
|
+
logical_tables = [
|
|
349
|
+
block.logical_table
|
|
350
|
+
for sheet in parsed.structure.sheets
|
|
351
|
+
for block in sheet.blocks
|
|
352
|
+
if block.logical_table is not None
|
|
353
|
+
]
|
|
354
|
+
cross_sheet_tables = [
|
|
355
|
+
continuation.logical_table for continuation in parsed.structure.table_continuations
|
|
356
|
+
]
|
|
357
|
+
```
|
|
358
|
+
|
|
359
|
+
The parser deterministically separates tables across blank row/column bands,
|
|
360
|
+
merges repeated print fragments, builds multi-level header paths, classifies
|
|
361
|
+
sections/data/totals, and chunks complete logical rows without crossing section
|
|
362
|
+
boundaries. Candidate regions are conservatively classified as logical tables,
|
|
363
|
+
forms, matrices, text, or explicit unclassified raw grids; every kind has a
|
|
364
|
+
source-aware Markdown and chunk path. `structure.snapshot` and compatibility
|
|
365
|
+
tables retain the original cell-level view. High-confidence adjacent-Sheet
|
|
366
|
+
continuations expose one aggregate logical table through
|
|
367
|
+
`structure.table_continuations`; insufficient evidence keeps tables independent
|
|
368
|
+
and records an ambiguous or rejected diagnostic. Markdown and chunks remain
|
|
369
|
+
source-Sheet based rather than duplicating the aggregate, and source-member
|
|
370
|
+
chunks can be regrouped by `continuation_id`. Retrieval/analysis dual chunk
|
|
371
|
+
profiles are built into the library, batch service, and CLI. `retrieval` is the
|
|
372
|
+
default profile and uses a 1000-character budget; `analysis` uses 4000. Both
|
|
373
|
+
preserve complete rows and exact source references. Analysis chunks add
|
|
374
|
+
normalized, source-linked `records` while keeping cell-level facts, including
|
|
375
|
+
formulas and cached values, in `structure.snapshot`. A parsed result can
|
|
376
|
+
generate another profile repeatedly with `chunk_result()` without reparsing or
|
|
377
|
+
mutating its structure. The analysis profile is only available for OOXML
|
|
378
|
+
workbook results: CSV, legacy `.xls`, and non-workbook inputs keep their
|
|
379
|
+
compatibility paths. Use `structure.snapshot` for exact cell or formula
|
|
380
|
+
analysis rather than treating analysis chunks as a replacement for the fact
|
|
381
|
+
layer.
|
|
382
|
+
|
|
383
|
+
```bash
|
|
384
|
+
langparse parse budget.xlsx --chunk --chunk-profile analysis --format json
|
|
385
|
+
```
|
|
386
|
+
|
|
387
|
+
#### Optional workbook model disambiguation
|
|
388
|
+
|
|
389
|
+
Workbook model disambiguation remains explicitly opt-in. The default is `off`:
|
|
390
|
+
constructing `ExcelParser()` or calling `ParseService` without
|
|
391
|
+
`workbook_disambiguation` performs no model Adapter or cache construction, reads
|
|
392
|
+
no provider configuration, and creates no implicit model network work. Install
|
|
393
|
+
the official OpenAI SDK integration separately from the core parser:
|
|
394
|
+
|
|
395
|
+
```bash
|
|
396
|
+
pip install "langparse[excel,model]"
|
|
397
|
+
export OPENAI_API_KEY="..."
|
|
398
|
+
export OPENAI_MODEL="gpt-4o-mini"
|
|
399
|
+
# Optional for an OpenAI-compatible endpoint:
|
|
400
|
+
export OPENAI_BASE_URL="https://example.invalid/v1"
|
|
401
|
+
|
|
402
|
+
langparse parse budget.xlsx --model --disambiguation auto --format json
|
|
403
|
+
```
|
|
404
|
+
|
|
405
|
+
API keys are intentionally not accepted as CLI arguments because process
|
|
406
|
+
arguments and shell history are not secret stores. `--model` or an explicit
|
|
407
|
+
`--disambiguation auto|required` enables network work; environment variables
|
|
408
|
+
alone never enable it. Set `LANGPARSE_DISABLE_MODEL=1` for the runtime kill
|
|
409
|
+
switch.
|
|
410
|
+
|
|
411
|
+
The library interface may either use the built-in `OpenAIWorkbookStructureAdapter`
|
|
412
|
+
or inject another `WorkbookStructureModelAdapter`:
|
|
413
|
+
|
|
414
|
+
```python
|
|
415
|
+
from langparse.parsers.excel_parser import ExcelParser
|
|
416
|
+
from langparse.services.parse_service import ParseService
|
|
417
|
+
from langparse.workbooks.modeling import WorkbookDisambiguation
|
|
418
|
+
|
|
419
|
+
# `adapter` is supplied by the caller and implements
|
|
420
|
+
# WorkbookStructureModelAdapter.
|
|
421
|
+
direct = ExcelParser(disambiguation=WorkbookDisambiguation.auto(adapter)).parse_result(
|
|
422
|
+
"budget.xlsx"
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
strict = ParseService().parse_result(
|
|
426
|
+
"budget.xlsx",
|
|
427
|
+
workbook_disambiguation=WorkbookDisambiguation.required(adapter),
|
|
428
|
+
)
|
|
429
|
+
```
|
|
430
|
+
|
|
431
|
+
Phase 4A is limited to **choice-only region-kind disambiguation**. Only a locally
|
|
432
|
+
ambiguous, unclassified region with at least two compatible registered kinds is
|
|
433
|
+
eligible. A response can only be `selected` with that case's registered
|
|
434
|
+
`case_id + choice_id`, or `abstained`; it cannot express a value, formula,
|
|
435
|
+
coordinate, range, header, row role, continuation, or arbitrary structure.
|
|
436
|
+
Provider-reported confidence is diagnostic only. The selected kind is applied
|
|
437
|
+
from the retained workbook snapshot and must still pass local materialization,
|
|
438
|
+
coverage, reconstruction, row-conservation, continuation, and source-reference
|
|
439
|
+
validation.
|
|
440
|
+
|
|
441
|
+
Model application is workbook-atomic. If any attempted selection cannot be
|
|
442
|
+
materialized, or the tentative workbook fails a continuation or structural
|
|
443
|
+
validator, every attempted selection is restored to its retained deterministic
|
|
444
|
+
block and all validators run again. `required` reports every reverted case as
|
|
445
|
+
unresolved.
|
|
446
|
+
|
|
447
|
+
`auto` keeps the deterministic local fallback and records sanitized diagnostics
|
|
448
|
+
when the provider, cache, limits, response contract, materialization, or final
|
|
449
|
+
validation fails, or when the provider abstains. `required` raises
|
|
450
|
+
`RequiredWorkbookDisambiguationError` for unresolved eligible ambiguity and the
|
|
451
|
+
typed error passes through `ExcelParser` and `ParseService`; a workbook with no
|
|
452
|
+
eligible ambiguity succeeds with zero calls in either mode.
|
|
453
|
+
|
|
454
|
+
An enabled `WorkbookDisambiguation` value owns a private, thread-safe,
|
|
455
|
+
process-local runtime/cache. Reusing that same value across `ExcelParser`,
|
|
456
|
+
`ParseService`, or batch calls allows a validated response to become a
|
|
457
|
+
re-decoded cache hit; `off` constructs no runtime or cache. `max_cases` limits
|
|
458
|
+
cases considered, while `max_calls` is a workbook-wide hard budget of actual
|
|
459
|
+
Adapter invocations, including retries; cache hits consume no calls. Policy
|
|
460
|
+
timeouts must be finite positive non-boolean real values, and count/byte limits
|
|
461
|
+
must be exact positive non-boolean integers.
|
|
462
|
+
|
|
463
|
+
The candidate request is deliberately narrow. It can include the target Sheet
|
|
464
|
+
name and source range, visible cell coordinates and display text, value type,
|
|
465
|
+
style fingerprint, merge geometry, local scalar features, and the registered
|
|
466
|
+
choices for that region. It omits hidden Sheets, formulas and cached formula
|
|
467
|
+
values, comments, hyperlinks, images, other regions, credentials, and provider
|
|
468
|
+
secrets. If any cell in the complete candidate envelope contains a formula—even
|
|
469
|
+
an unlisted cell or merged child—the whole case is locally unavailable and no
|
|
470
|
+
formula or cached result is projected. Cell text is treated as untrusted Prompt Injection data: the Adapter
|
|
471
|
+
port exposes no tool channel, and exact response fields, request checksum,
|
|
472
|
+
case/choice membership, size limits, and local validation prevent cell
|
|
473
|
+
instructions from expanding the operation. Duplicate JSON member names are
|
|
474
|
+
rejected at every response-object depth. Diagnostics do not retain prompts,
|
|
475
|
+
cell text, raw replies, or provider exception messages. The process-local,
|
|
476
|
+
non-persistent cache has a narrower but different contract: it retains only
|
|
477
|
+
response envelope bytes that have already passed strict response decoding, and
|
|
478
|
+
every hit is decoded and validated again. Nothing is written to disk, but
|
|
479
|
+
provider-supplied strings inside that envelope may remain in process memory
|
|
480
|
+
until the owning disambiguation value and its private runtime are released.
|
|
481
|
+
Each model-call audit records local schema, prompt, rule, validator, and privacy
|
|
482
|
+
versions plus the deterministic fallback rule confidence; these values never
|
|
483
|
+
come from the provider.
|
|
484
|
+
|
|
485
|
+
The workbook ambiguity evaluator is available with:
|
|
486
|
+
|
|
487
|
+
```bash
|
|
488
|
+
langparse eval \
|
|
489
|
+
samples/workbook_ambiguity/public-manifest.json \
|
|
490
|
+
--output-dir reports/workbook-ambiguity
|
|
491
|
+
|
|
492
|
+
# Live provider evidence is still explicit:
|
|
493
|
+
langparse eval private-manifest.json --model
|
|
494
|
+
```
|
|
495
|
+
|
|
496
|
+
Reports are immutable by digest and reject incomplete or modified replays.
|
|
497
|
+
`production_ready` additionally requires holdout data, at least 30 ambiguous
|
|
498
|
+
cases, and separate operational staging evidence; the bundled tuning seed can
|
|
499
|
+
never satisfy that release gate by itself.
|
|
500
|
+
|
|
501
|
+
Whole-workbook structural quality has a separate entry point, so model
|
|
502
|
+
disambiguation accuracy is never presented as final Excel accuracy:
|
|
503
|
+
|
|
504
|
+
```bash
|
|
505
|
+
langparse benchmark-workbook-quality \
|
|
506
|
+
samples/workbook_quality/public-manifest.json \
|
|
507
|
+
--output-dir reports/workbook-quality
|
|
508
|
+
```
|
|
509
|
+
|
|
510
|
+
The public tuning seed contains thirteen manually labelled workbooks covering
|
|
511
|
+
logical tables, multiple tables per sheet, adjacent native tables, visually
|
|
512
|
+
separated tables without blank columns, form side notes, matrices, text,
|
|
513
|
+
explicit fallback, cross-sheet continuation, repeated print fragments, chart
|
|
514
|
+
facts, formulas, named ranges, and hidden sheets. Reports score block precision/recall, header paths,
|
|
515
|
+
row roles, forms/matrices, continuations, source references, fallback, and
|
|
516
|
+
object fact/semantic coverage. A failed gate returns exit code `1`; reports omit
|
|
517
|
+
cell values and annotation content and include the complete truth digest in the
|
|
518
|
+
immutable run digest. The public seed prevents regression; production evidence
|
|
519
|
+
still requires a separate private holdout.
|
|
520
|
+
|
|
521
|
+
Cost circuit breakers never infer prices from a model name. Library callers
|
|
522
|
+
that set `max_cost_usd_per_workbook` must also supply
|
|
523
|
+
`input_cost_usd_per_million`, `output_cost_usd_per_million`, and a stable
|
|
524
|
+
`cost_pricing_version` in `WorkbookModelPolicy`. These rates should come from
|
|
525
|
+
the deployment's provider contract, including for OpenAI-compatible endpoints.
|
|
526
|
+
|
|
527
|
+
Phase 4B includes the optional OpenAI SDK Adapter, environment-based provider
|
|
528
|
+
configuration, a strict structured-response contract, observed usage/cost
|
|
529
|
+
circuit breakers, and an immutable evaluation report pipeline. This is a
|
|
530
|
+
usable provider path, but not production-effectiveness evidence by itself:
|
|
531
|
+
release still requires a representative private holdout, staging latency/cost
|
|
532
|
+
and failure-mode evidence, and a provider privacy review. The observed token
|
|
533
|
+
and cost budgets stop retries or later calls after reported usage reaches the
|
|
534
|
+
limit; they cannot prevent the first provider call from exceeding a budget and
|
|
535
|
+
therefore are circuit breakers rather than billing guarantees.
|
|
536
|
+
|
|
537
|
+
Summary/index chunks, rich `.xls`/`.xlsb` adapters, image/chart semantic blocks,
|
|
538
|
+
standard bundle output, and further production hardening remain follow-up work;
|
|
539
|
+
screenshots and VLM are Phase 4C, a second domain contract is Phase 4D, and
|
|
540
|
+
delimited and legacy inputs keep the compatibility adapter for now.
|
|
541
|
+
|
|
542
|
+
### Scanned PDFs
|
|
543
|
+
|
|
544
|
+
The `simple` engine falls back to OCR when a page turns out to be an image.
|
|
545
|
+
Detection needs both a page-covering image and a thin text layer — a scanned
|
|
546
|
+
page often carries a watermark, and that watermark text is enough to clear any
|
|
547
|
+
threshold low enough to avoid firing on genuinely sparse text pages.
|
|
548
|
+
|
|
549
|
+
```bash
|
|
550
|
+
pip install "langparse[ocr]"
|
|
551
|
+
```
|
|
552
|
+
|
|
553
|
+
```python
|
|
554
|
+
PDFParser(engine="simple", enable_ocr=True, ocr_min_chars=500)
|
|
555
|
+
```
|
|
556
|
+
|
|
557
|
+
Pages that took the fallback report `ocr_applied` and `ocr_text_chars` in their
|
|
558
|
+
metadata, which surface in `ParseMetrics`. Without `rapidocr_onnxruntime`
|
|
559
|
+
installed the parse still succeeds — the page simply keeps whatever text layer
|
|
560
|
+
it had, rather than failing.
|
|
561
|
+
|
|
562
|
+
### Measuring parse fidelity
|
|
563
|
+
|
|
564
|
+
Quality checks measure structure: page counts, whether any tables were found.
|
|
565
|
+
They say nothing about whether the content is *correct*. To measure that, give a
|
|
566
|
+
benchmark sample a reference:
|
|
567
|
+
|
|
568
|
+
```json
|
|
569
|
+
{
|
|
570
|
+
"id": "report-01",
|
|
571
|
+
"path": "samples/report.pdf",
|
|
572
|
+
"expected_markdown": "samples/report.expected.md",
|
|
573
|
+
"expected_tables": [[["Header A", "Header B"], ["1", "2"]]]
|
|
574
|
+
}
|
|
575
|
+
```
|
|
576
|
+
|
|
577
|
+
- **Text** is scored by word-level normalised edit distance. Words rather than
|
|
578
|
+
characters, because a reflowed line break is not an error but a dropped word
|
|
579
|
+
is.
|
|
580
|
+
- **Tables** are scored by TEDS. Cell substitution costs the normalised
|
|
581
|
+
character distance between the two cells, so a typo scores better than a
|
|
582
|
+
wrong value, and a dropped row costs more than a changed cell.
|
|
583
|
+
|
|
584
|
+
Samples without a reference are reported as unscored, never as perfect.
|
|
585
|
+
|
|
586
|
+
### MinerU Runtime
|
|
587
|
+
|
|
588
|
+
LangParse can run MinerU through `mineru-api`.
|
|
589
|
+
|
|
590
|
+
Runtime selection works like this:
|
|
591
|
+
- If you pass or configure `api_url`, LangParse calls that MinerU service directly.
|
|
592
|
+
- A remote `mineru-api` backed by a separate vLLM server also receives `backend` and `server_url` as `/file_parse` form fields. This path does **not** require the local `[mineru]` extra.
|
|
593
|
+
- If `api_url` is not set, LangParse will try to start a local `mineru-api` service and manage its lifecycle for the current parse.
|
|
594
|
+
- If `mineru-api` is not installed, pass `--auto-install-runtime` or `auto_install_runtime=True` to let LangParse install the configured runtime package in the current Python environment before starting the local service.
|
|
595
|
+
|
|
596
|
+
You can still control CPU/GPU selection and model/download directories through runtime parameters or configuration.
|
|
597
|
+
|
|
598
|
+
For local managed services:
|
|
599
|
+
- `model_dir` means "use this already-downloaded MinerU model directory"
|
|
600
|
+
- `download_dir` becomes the MinerU home root used by the local service, so MinerU will keep its default cache/config layout under that directory
|
|
601
|
+
- `model_policy="require_existing"` disables first-run download fallback and requires an existing local model setup
|
|
602
|
+
|
|
603
|
+
```python
|
|
604
|
+
from langparse import AutoParser
|
|
605
|
+
|
|
606
|
+
doc = AutoParser.parse(
|
|
607
|
+
"paper.pdf",
|
|
608
|
+
engine="mineru",
|
|
609
|
+
api_url="http://mineru.example:25820",
|
|
610
|
+
backend="vlm-http-client",
|
|
611
|
+
server_url="http://vlm.example:21670",
|
|
612
|
+
request_timeout=900,
|
|
613
|
+
)
|
|
614
|
+
```
|
|
615
|
+
|
|
616
|
+
```python
|
|
617
|
+
from langparse import AutoParser
|
|
618
|
+
|
|
619
|
+
cpu_doc = AutoParser.parse(
|
|
620
|
+
"paper.pdf",
|
|
621
|
+
engine="mineru",
|
|
622
|
+
device="cpu",
|
|
623
|
+
download_dir="./downloads",
|
|
624
|
+
)
|
|
625
|
+
```
|
|
626
|
+
|
|
627
|
+
```python
|
|
628
|
+
from langparse import AutoParser
|
|
629
|
+
|
|
630
|
+
local_doc = AutoParser.parse(
|
|
631
|
+
"paper.pdf",
|
|
632
|
+
engine="mineru",
|
|
633
|
+
model_dir="./preloaded-models",
|
|
634
|
+
model_policy="require_existing",
|
|
635
|
+
)
|
|
636
|
+
```
|
|
637
|
+
|
|
638
|
+
Environment variables:
|
|
639
|
+
|
|
640
|
+
```bash
|
|
641
|
+
export LANGPARSE_MINERU_API_URL=http://127.0.0.1:8000
|
|
642
|
+
export LANGPARSE_MINERU_BACKEND=vlm-http-client
|
|
643
|
+
export LANGPARSE_MINERU_SERVER_URL=http://vlm.example:21670
|
|
644
|
+
export LANGPARSE_MINERU_REQUEST_TIMEOUT=900
|
|
645
|
+
export LANGPARSE_MINERU_DEVICE=cuda
|
|
646
|
+
export LANGPARSE_MINERU_MODEL_DIR=./models
|
|
647
|
+
export LANGPARSE_MINERU_DOWNLOAD_DIR=./downloads
|
|
648
|
+
export LANGPARSE_MINERU_MODEL_POLICY=require_existing
|
|
649
|
+
export LANGPARSE_MINERU_AUTO_INSTALL_RUNTIME=true
|
|
650
|
+
```
|
|
651
|
+
|
|
652
|
+
### CLI Examples
|
|
653
|
+
|
|
654
|
+
The CLI handles every supported format, not just PDF. `--engine` applies to
|
|
655
|
+
PDFs only; other formats route to their own parser automatically:
|
|
656
|
+
|
|
657
|
+
```bash
|
|
658
|
+
langparse parse report.docx --format json
|
|
659
|
+
langparse parse notes.md --output notes.out.md
|
|
660
|
+
langparse parse mixed_folder/ --batch --output-dir out --metrics
|
|
661
|
+
```
|
|
662
|
+
|
|
663
|
+
Supported extensions: `.pdf`, `.docx`, `.doc`, `.xlsx`, `.xlsm`, `.xls`, `.csv`, `.md`, `.txt`.
|
|
664
|
+
Batch directory expansion picks up all of them; unsupported files exit with
|
|
665
|
+
code 2 and a one-line message.
|
|
666
|
+
|
|
667
|
+
Single-file parsing:
|
|
668
|
+
|
|
669
|
+
```bash
|
|
670
|
+
langparse parse paper.pdf --engine mineru \
|
|
671
|
+
--api-url http://mineru.example:25820 \
|
|
672
|
+
--mineru-backend vlm-http-client \
|
|
673
|
+
--mineru-server-url http://vlm.example:21670 \
|
|
674
|
+
--mineru-request-timeout 900 \
|
|
675
|
+
--format json
|
|
676
|
+
```
|
|
677
|
+
|
|
678
|
+
Batch parsing:
|
|
679
|
+
|
|
680
|
+
```bash
|
|
681
|
+
langparse parse docs/ --engine mineru --batch --output-dir out --format json
|
|
682
|
+
```
|
|
683
|
+
|
|
684
|
+
Batch parsing with lightweight metrics and skip-existing behavior:
|
|
685
|
+
|
|
686
|
+
```bash
|
|
687
|
+
langparse parse docs/ --engine mineru --batch --output-dir out --format json --max-workers 4 --skip-existing --metrics
|
|
688
|
+
```
|
|
689
|
+
|
|
690
|
+
Run a product-readiness benchmark:
|
|
691
|
+
|
|
692
|
+
```bash
|
|
693
|
+
langparse benchmark samples/public.example.json --engine mineru --output-dir reports --max-workers 2
|
|
694
|
+
```
|
|
695
|
+
|
|
696
|
+
Benchmark reports include success rate, elapsed time, pages per second, table counts, OCR indicators, reading-order warnings, header/footer filtering counts, and image/caption metadata coverage.
|
|
697
|
+
|
|
698
|
+
If you want LangParse to manage a local MinerU service, omit `--api-url`. You can also override the local launch command and bind address:
|
|
699
|
+
|
|
700
|
+
```bash
|
|
701
|
+
langparse parse paper.pdf --engine mineru --api-command "mineru-api" --api-host 127.0.0.1 --api-port 8000
|
|
702
|
+
```
|
|
703
|
+
|
|
704
|
+
Install MinerU automatically in the current Python environment if `mineru-api` is missing:
|
|
705
|
+
|
|
706
|
+
```bash
|
|
707
|
+
langparse parse paper.pdf --engine mineru --auto-install-runtime --device cpu --format json
|
|
708
|
+
```
|
|
709
|
+
|
|
710
|
+
Use an existing local model directory without allowing implicit downloads:
|
|
711
|
+
|
|
712
|
+
```bash
|
|
713
|
+
langparse parse paper.pdf --engine mineru --model-dir ./preloaded-models --model-policy require_existing
|
|
714
|
+
```
|
|
715
|
+
|
|
716
|
+
## 🛠️ Development & Local Testing
|
|
717
|
+
|
|
718
|
+
LangParse uses [`uv`](https://github.com/astral-sh/uv) for environment and dependency management. The checked-in `.venv` is uv-managed and intentionally has **no `pip`**, so run everything through `uv run` (a bare `pip`/`python` on your shell may resolve to a different interpreter, e.g. Anaconda).
|
|
719
|
+
|
|
720
|
+
### Set up the environment
|
|
721
|
+
|
|
722
|
+
```bash
|
|
723
|
+
# Install all dependencies (including dev/test) from uv.lock
|
|
724
|
+
uv sync --all-extras
|
|
725
|
+
|
|
726
|
+
# Or install just what you need
|
|
727
|
+
uv sync # core only (no third-party dependencies)
|
|
728
|
+
uv pip install -e ".[pdf]" # PDF parsing (pdfplumber)
|
|
729
|
+
uv pip install -e ".[docx]" # Word parsing (python-docx)
|
|
730
|
+
uv pip install -e ".[excel]" # Excel parsing (pandas + openpyxl)
|
|
731
|
+
uv pip install -e ".[model]" # Optional OpenAI workbook disambiguation
|
|
732
|
+
uv pip install -e ".[ocr]" # OCR (rapidocr_onnxruntime)
|
|
733
|
+
uv pip install -e ".[mineru]"# MinerU API/orchestrator (local backend is explicit)
|
|
734
|
+
uv pip install -e ".[deepdoc]"# DeepDoc runtime (OCR/layout/table ONNX weights, ~100MB download on first run)
|
|
735
|
+
uv pip install -e ".[all]" # everything above
|
|
736
|
+
```
|
|
737
|
+
|
|
738
|
+
> Note: the core install has **no third-party dependencies**. The PDF/DOCX/Excel parsers require the optional extras above; without them a parse fails with an `ImportError` naming the missing package rather than crashing. `pip install -e ".[dev]"` is enough to run the test suite.
|
|
739
|
+
|
|
740
|
+
### Run the tests
|
|
741
|
+
|
|
742
|
+
```bash
|
|
743
|
+
uv run pytest -q
|
|
744
|
+
```
|
|
745
|
+
|
|
746
|
+
### Smoke-test locally
|
|
747
|
+
|
|
748
|
+
```bash
|
|
749
|
+
# Markdown parse + semantic chunk (no extra deps needed)
|
|
750
|
+
uv run python examples/basic_usage.py
|
|
751
|
+
|
|
752
|
+
# Parse a PDF of your own (requires the [pdf] extra)
|
|
753
|
+
uv run langparse parse your.pdf --engine simple --format json
|
|
754
|
+
|
|
755
|
+
# Run the benchmark on the bundled manifest template
|
|
756
|
+
uv run langparse benchmark samples/public.example.json --engine simple --output-dir reports
|
|
757
|
+
```
|
|
758
|
+
|
|
759
|
+
The repository ships `samples/public.example.json` as a benchmark manifest template. `data/` is where local test documents go; it is git-ignored, so bring your own.
|
|
760
|
+
|
|
761
|
+
## 💬 Contact
|
|
762
|
+
|
|
763
|
+
For questions, feature requests, or bug reports, the preferred method is to **open an issue** on this GitHub repository. This allows for transparent discussion and helps other users who might have the same question.
|
|
764
|
+
|
|
765
|
+
## Citing LangParse
|
|
766
|
+
|
|
767
|
+
If you use LangParse in your research, product, or publication, we would appreciate a citation! You can use the following BibTeX entry:
|
|
768
|
+
|
|
769
|
+
```bibtex
|
|
770
|
+
@software{LangParse_2026,
|
|
771
|
+
author = {syw2014},
|
|
772
|
+
title = {LangParse: A developer-friendly document parsing toolkit with source-grounded Excel understanding},
|
|
773
|
+
month = {September},
|
|
774
|
+
year = {2026},
|
|
775
|
+
publisher = {GitHub},
|
|
776
|
+
url = {https://github.com/syw2014/langparse}
|
|
777
|
+
}
|
|
778
|
+
```
|
|
779
|
+
|
|
780
|
+
## Changelog
|
|
781
|
+
See [CHANGELOG.md](CHANGELOG.md) ([中文](CHANGELOG_cn.md)) for release notes and the dated development history.
|
|
782
|
+
|
|
783
|
+
## License
|
|
784
|
+
This project is licensed under the [Apache 2.0 License](https://www.apache.org/licenses/LICENSE-2.0).
|
|
785
|
+
|
|
786
|
+
### v0.1.0 capabilities
|
|
787
|
+
|
|
788
|
+
- [Workbook Bundle 与查询](docs/WORKBOOK_BUNDLE.md)
|
|
789
|
+
- [分块策略与 CLI](docs/CHUNKING.md)
|
|
790
|
+
- [Agent Skill 与接入示例](SKILLS.md)
|