langparse 0.1.0__tar.gz → 0.1.0rc2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {langparse-0.1.0/langparse.egg-info → langparse-0.1.0rc2}/PKG-INFO +6 -58
- {langparse-0.1.0 → langparse-0.1.0rc2}/README.md +3 -55
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/__init__.py +1 -16
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/workbook.py +1 -43
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/cli.py +1 -44
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc_engine.py +3 -27
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru.py +4 -28
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/simple.py +1 -8
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/errors.py +1 -19
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/excel_parser.py +3 -15
- langparse-0.1.0rc2/langparse/py.typed +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/batch_service.py +6 -92
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/parse_service.py +31 -86
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/types.py +0 -2
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/__init__.py +0 -4
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/adapters.py +32 -70
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/assembly.py +2 -22
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/evaluator.py +2 -18
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/schema.py +2 -14
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/rendering.py +4 -10
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/types.py +1 -8
- {langparse-0.1.0 → langparse-0.1.0rc2/langparse.egg-info}/PKG-INFO +6 -58
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/SOURCES.txt +1 -28
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/requires.txt +2 -2
- {langparse-0.1.0 → langparse-0.1.0rc2}/pyproject.toml +3 -4
- langparse-0.1.0rc2/tests/test_errors.py +29 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_benchmark.py +2 -2
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_evaluator.py +0 -17
- langparse-0.1.0/MANIFEST.in +0 -6
- langparse-0.1.0/SKILLS.md +0 -22
- langparse-0.1.0/docs/CHUNKING.md +0 -51
- langparse-0.1.0/docs/RELEASE_0.1.0.md +0 -38
- langparse-0.1.0/docs/WORKBOOK_BUNDLE.md +0 -42
- langparse-0.1.0/examples/agent_ingestion.py +0 -40
- langparse-0.1.0/langparse/chunkers/__init__.py +0 -12
- langparse-0.1.0/langparse/chunkers/registry.py +0 -38
- langparse-0.1.0/langparse/chunkers/text.py +0 -96
- langparse-0.1.0/langparse/progress.py +0 -77
- langparse-0.1.0/langparse/workbooks/bundle-v1.schema.json +0 -71
- langparse-0.1.0/langparse/workbooks/bundle.py +0 -341
- langparse-0.1.0/langparse/workbooks/lineage.py +0 -117
- langparse-0.1.0/langparse/workbooks/objects.py +0 -229
- langparse-0.1.0/langparse/workbooks/quality/bundle.py +0 -53
- langparse-0.1.0/langparse/workbooks/quality/facts.py +0 -142
- langparse-0.1.0/langparse/workbooks/reference_types.py +0 -73
- langparse-0.1.0/langparse/workbooks/references.py +0 -178
- langparse-0.1.0/skills/langparse/SKILL.md +0 -38
- langparse-0.1.0/tests/fixtures/workbook-bundle-v1.json +0 -11
- langparse-0.1.0/tests/test_batch_progress.py +0 -224
- langparse-0.1.0/tests/test_chunk_strategies.py +0 -131
- langparse-0.1.0/tests/test_errors.py +0 -60
- langparse-0.1.0/tests/test_pdf_progress.py +0 -105
- langparse-0.1.0/tests/test_progress.py +0 -126
- langparse-0.1.0/tests/test_workbook_bundle.py +0 -138
- langparse-0.1.0/tests/test_workbook_fact_quality.py +0 -124
- langparse-0.1.0/tests/test_workbook_lineage.py +0 -141
- langparse-0.1.0/tests/test_workbook_objects.py +0 -196
- {langparse-0.1.0 → langparse-0.1.0rc2}/LICENSE +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/autoparser.py +0 -0
- {langparse-0.1.0/langparse/core → langparse-0.1.0rc2/langparse/chunkers}/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/blocks.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/profiles.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/semantic.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/config.py +0 -0
- {langparse-0.1.0/langparse/parsers → langparse-0.1.0rc2/langparse/core}/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/chunker.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/engine.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/rendering.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/layout_recognizer.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/model_loader.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/ocr.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/operators.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/pdf_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/postprocess.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/recognizer.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/rendering.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/table_structure_recognizer.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/tokenizer.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/utils.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru_client.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru_service.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/ocr.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/other.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/vision_llm.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/logging.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/metrics.py +0 -0
- /langparse-0.1.0/langparse/py.typed → /langparse-0.1.0rc2/langparse/parsers/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/docx_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/markdown_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/pdf_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/registry.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/sniff.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/benchmark_service.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/fidelity.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/output_paths.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/quality.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/workbook_ambiguity_benchmark.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/workbook_quality_benchmark.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/blocks.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/classification.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/continuation.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/evaluator.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/schema.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/labels.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/cache.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/config.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/contract.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/disambiguation.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/openai_adapter.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/policy.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/ports.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/pricing.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/types.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/__init__.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/regions.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/tables.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/dependency_links.txt +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/entry_points.txt +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/top_level.txt +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/setup.cfg +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_autoparser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_batch_service.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_benchmark_service.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_blocks.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunk_pipeline.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunker.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunker_sizing.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_cli.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_config.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_content_sniffing.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_credential_isolation.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_engine.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_model_loader.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_rendering.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_tokenizer.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_engine_registry.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_evaluation_strictness.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_logical_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_model_cli.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_model_modes.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_structural_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_fidelity.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_logging.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_metrics.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_mineru_engine.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_mineru_live_integration.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_ocr.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_ooxml_adapter.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_optional_dependency_boundaries.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parse_service.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parser_results.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parsers.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_pdf_parser.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_quality.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_quota_fail_closed.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_real_sample_regression.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_result_envelope.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_ambiguity_benchmark.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_ambiguity_evaluator.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly_blocks.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly_modeling.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_blocks.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_chunk_profiles.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_chunker.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_classification.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_continuation.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_disambiguation.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_evaluation_schema.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_config.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_contract.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_policy.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_pricing.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_safety_drills.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_openai_adapter.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_schema.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_real_layouts.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_region_assessment.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_regions.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_rendering.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_tables.py +0 -0
- {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: langparse
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.0rc2
|
|
4
4
|
Summary: A developer-friendly document parsing toolkit with precise, source-grounded Excel understanding.
|
|
5
5
|
Author-email: syw2014 <syw2014@gmail.com>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -40,7 +40,7 @@ Provides-Extra: deepdoc
|
|
|
40
40
|
Requires-Dist: pdfplumber>=0.11.10; extra == "deepdoc"
|
|
41
41
|
Requires-Dist: opencv-python-headless>=4.9.0; extra == "deepdoc"
|
|
42
42
|
Requires-Dist: onnxruntime>=1.17.0; extra == "deepdoc"
|
|
43
|
-
Requires-Dist: pypdf>=6.
|
|
43
|
+
Requires-Dist: pypdf>=6.15.0; extra == "deepdoc"
|
|
44
44
|
Requires-Dist: huggingface_hub>=0.20.0; extra == "deepdoc"
|
|
45
45
|
Requires-Dist: scikit-learn>=1.3.0; extra == "deepdoc"
|
|
46
46
|
Requires-Dist: shapely>=2.0.0; extra == "deepdoc"
|
|
@@ -58,7 +58,7 @@ Requires-Dist: rapidocr_onnxruntime; extra == "all"
|
|
|
58
58
|
Requires-Dist: mineru<4,>=3.4; extra == "all"
|
|
59
59
|
Requires-Dist: opencv-python-headless; extra == "all"
|
|
60
60
|
Requires-Dist: onnxruntime; extra == "all"
|
|
61
|
-
Requires-Dist: pypdf>=6.
|
|
61
|
+
Requires-Dist: pypdf>=6.15.0; extra == "all"
|
|
62
62
|
Requires-Dist: huggingface_hub; extra == "all"
|
|
63
63
|
Requires-Dist: scikit-learn; extra == "all"
|
|
64
64
|
Requires-Dist: shapely; extra == "all"
|
|
@@ -112,59 +112,13 @@ where LangParse goes deliberately deeper.
|
|
|
112
112
|
|
|
113
113
|
---
|
|
114
114
|
|
|
115
|
-
## Parsing progress (unreleased)
|
|
116
|
-
|
|
117
|
-
Pass `progress_callback` to `AutoParser`, `ParseService`, `PDFParser`, or
|
|
118
|
-
`BatchParseService.run`. The callback receives immutable `ProgressEvent` objects:
|
|
119
|
-
|
|
120
|
-
```python
|
|
121
|
-
from langparse import AutoParser
|
|
122
|
-
|
|
123
|
-
result = AutoParser.parse_result(
|
|
124
|
-
"budget.xlsx",
|
|
125
|
-
progress_callback=lambda event: print(event.phase, event.state, event.percent),
|
|
126
|
-
)
|
|
127
|
-
```
|
|
128
|
-
|
|
129
|
-
Events contain `source`, `phase`, `state`, `completed_units`, `total_units`, `unit`,
|
|
130
|
-
`percent`, and `message`. Percent is phase-local (0–100), not an ETA; unknown
|
|
131
|
-
values remain `None`. Only a `file` event with `completed`/`failed` marks the end
|
|
132
|
-
of parsing and requested chunking, before any caller-side export. A completed
|
|
133
|
-
parse can still have partial quality diagnostics.
|
|
134
|
-
|
|
135
|
-
Simple PDF reports page counts; DeepDoc forwards its internal weighted-stage or
|
|
136
|
-
page-batch progress; MinerU reports preparation, remote parsing, and rendering
|
|
137
|
-
without an internal percentage. Excel reports extraction, assembly (including
|
|
138
|
-
optional model disambiguation), and rendering. Direct `ExcelParser` calls also
|
|
139
|
-
emit those stages; the facade/service supplies the file lifecycle.
|
|
140
|
-
|
|
141
|
-
Batch events use `phase="batch"`, `source=""`, and `unit="files"`. Notifications
|
|
142
|
-
follow completion order while returned results retain the batch's sorted file order. Success,
|
|
143
|
-
failure, and skip all count as finished files. Batch completion follows report
|
|
144
|
-
writing and does not mean every file succeeded. Each `batch_item` event names
|
|
145
|
-
the source and reports `completed`/`failed`/`skipped` after that item's output
|
|
146
|
-
write. Empty batches report 0/0 with
|
|
147
|
-
unknown percent. Ordinary callback exceptions are logged by type and isolated
|
|
148
|
-
from parsing. Callbacks are synchronous: keep them brief. One batch serializes
|
|
149
|
-
its callbacks, which can run on different threads; callers coordinate callbacks
|
|
150
|
-
shared across independent runs. There is no task store, background execution,
|
|
151
|
-
cancellation, heartbeat, or HTTP query endpoint.
|
|
152
|
-
|
|
153
|
-
CLI `--progress` writes line-oriented updates to stderr, leaving stdout intact:
|
|
154
|
-
|
|
155
|
-
```bash
|
|
156
|
-
langparse parse budget.xlsx --format json --progress > result.json
|
|
157
|
-
langparse parse docs/ --batch --output-dir out --progress
|
|
158
|
-
```
|
|
159
|
-
|
|
160
115
|
## Project status
|
|
161
116
|
|
|
162
|
-
The
|
|
117
|
+
The current release candidate is `0.1.0rc2`. Core multi-format parsing,
|
|
163
118
|
structured OOXML workbooks, semantic chunking, batch processing, quality checks,
|
|
164
119
|
and CI are available today. LangParse remains pre-1.0; see
|
|
165
120
|
[docs/PROGRESS.md](docs/PROGRESS.md) for the module-by-module source of truth and
|
|
166
|
-
known gaps.
|
|
167
|
-
[release scope](docs/RELEASE_0.1.0.md).
|
|
121
|
+
known gaps.
|
|
168
122
|
|
|
169
123
|
## Why LangParse?
|
|
170
124
|
|
|
@@ -214,7 +168,7 @@ views derived from it, not replacements for it.
|
|
|
214
168
|
Install the current release candidate:
|
|
215
169
|
|
|
216
170
|
```bash
|
|
217
|
-
pip install "langparse==0.1.
|
|
171
|
+
pip install --pre "langparse==0.1.0rc2"
|
|
218
172
|
```
|
|
219
173
|
|
|
220
174
|
Install only the optional capabilities you need:
|
|
@@ -782,9 +736,3 @@ See [CHANGELOG.md](CHANGELOG.md) ([中文](CHANGELOG_cn.md)) for release notes a
|
|
|
782
736
|
|
|
783
737
|
## License
|
|
784
738
|
This project is licensed under the [Apache 2.0 License](https://www.apache.org/licenses/LICENSE-2.0).
|
|
785
|
-
|
|
786
|
-
### v0.1.0 capabilities
|
|
787
|
-
|
|
788
|
-
- [Workbook Bundle 与查询](docs/WORKBOOK_BUNDLE.md)
|
|
789
|
-
- [分块策略与 CLI](docs/CHUNKING.md)
|
|
790
|
-
- [Agent Skill 与接入示例](SKILLS.md)
|
|
@@ -33,59 +33,13 @@ where LangParse goes deliberately deeper.
|
|
|
33
33
|
|
|
34
34
|
---
|
|
35
35
|
|
|
36
|
-
## Parsing progress (unreleased)
|
|
37
|
-
|
|
38
|
-
Pass `progress_callback` to `AutoParser`, `ParseService`, `PDFParser`, or
|
|
39
|
-
`BatchParseService.run`. The callback receives immutable `ProgressEvent` objects:
|
|
40
|
-
|
|
41
|
-
```python
|
|
42
|
-
from langparse import AutoParser
|
|
43
|
-
|
|
44
|
-
result = AutoParser.parse_result(
|
|
45
|
-
"budget.xlsx",
|
|
46
|
-
progress_callback=lambda event: print(event.phase, event.state, event.percent),
|
|
47
|
-
)
|
|
48
|
-
```
|
|
49
|
-
|
|
50
|
-
Events contain `source`, `phase`, `state`, `completed_units`, `total_units`, `unit`,
|
|
51
|
-
`percent`, and `message`. Percent is phase-local (0–100), not an ETA; unknown
|
|
52
|
-
values remain `None`. Only a `file` event with `completed`/`failed` marks the end
|
|
53
|
-
of parsing and requested chunking, before any caller-side export. A completed
|
|
54
|
-
parse can still have partial quality diagnostics.
|
|
55
|
-
|
|
56
|
-
Simple PDF reports page counts; DeepDoc forwards its internal weighted-stage or
|
|
57
|
-
page-batch progress; MinerU reports preparation, remote parsing, and rendering
|
|
58
|
-
without an internal percentage. Excel reports extraction, assembly (including
|
|
59
|
-
optional model disambiguation), and rendering. Direct `ExcelParser` calls also
|
|
60
|
-
emit those stages; the facade/service supplies the file lifecycle.
|
|
61
|
-
|
|
62
|
-
Batch events use `phase="batch"`, `source=""`, and `unit="files"`. Notifications
|
|
63
|
-
follow completion order while returned results retain the batch's sorted file order. Success,
|
|
64
|
-
failure, and skip all count as finished files. Batch completion follows report
|
|
65
|
-
writing and does not mean every file succeeded. Each `batch_item` event names
|
|
66
|
-
the source and reports `completed`/`failed`/`skipped` after that item's output
|
|
67
|
-
write. Empty batches report 0/0 with
|
|
68
|
-
unknown percent. Ordinary callback exceptions are logged by type and isolated
|
|
69
|
-
from parsing. Callbacks are synchronous: keep them brief. One batch serializes
|
|
70
|
-
its callbacks, which can run on different threads; callers coordinate callbacks
|
|
71
|
-
shared across independent runs. There is no task store, background execution,
|
|
72
|
-
cancellation, heartbeat, or HTTP query endpoint.
|
|
73
|
-
|
|
74
|
-
CLI `--progress` writes line-oriented updates to stderr, leaving stdout intact:
|
|
75
|
-
|
|
76
|
-
```bash
|
|
77
|
-
langparse parse budget.xlsx --format json --progress > result.json
|
|
78
|
-
langparse parse docs/ --batch --output-dir out --progress
|
|
79
|
-
```
|
|
80
|
-
|
|
81
36
|
## Project status
|
|
82
37
|
|
|
83
|
-
The
|
|
38
|
+
The current release candidate is `0.1.0rc2`. Core multi-format parsing,
|
|
84
39
|
structured OOXML workbooks, semantic chunking, batch processing, quality checks,
|
|
85
40
|
and CI are available today. LangParse remains pre-1.0; see
|
|
86
41
|
[docs/PROGRESS.md](docs/PROGRESS.md) for the module-by-module source of truth and
|
|
87
|
-
known gaps.
|
|
88
|
-
[release scope](docs/RELEASE_0.1.0.md).
|
|
42
|
+
known gaps.
|
|
89
43
|
|
|
90
44
|
## Why LangParse?
|
|
91
45
|
|
|
@@ -135,7 +89,7 @@ views derived from it, not replacements for it.
|
|
|
135
89
|
Install the current release candidate:
|
|
136
90
|
|
|
137
91
|
```bash
|
|
138
|
-
pip install "langparse==0.1.
|
|
92
|
+
pip install --pre "langparse==0.1.0rc2"
|
|
139
93
|
```
|
|
140
94
|
|
|
141
95
|
Install only the optional capabilities you need:
|
|
@@ -703,9 +657,3 @@ See [CHANGELOG.md](CHANGELOG.md) ([中文](CHANGELOG_cn.md)) for release notes a
|
|
|
703
657
|
|
|
704
658
|
## License
|
|
705
659
|
This project is licensed under the [Apache 2.0 License](https://www.apache.org/licenses/LICENSE-2.0).
|
|
706
|
-
|
|
707
|
-
### v0.1.0 capabilities
|
|
708
|
-
|
|
709
|
-
- [Workbook Bundle 与查询](docs/WORKBOOK_BUNDLE.md)
|
|
710
|
-
- [分块策略与 CLI](docs/CHUNKING.md)
|
|
711
|
-
- [Agent Skill 与接入示例](SKILLS.md)
|
|
@@ -3,14 +3,7 @@ from importlib.metadata import version as _distribution_version
|
|
|
3
3
|
__version__ = _distribution_version("langparse")
|
|
4
4
|
|
|
5
5
|
from langparse.autoparser import AutoParser
|
|
6
|
-
from langparse.chunkers import
|
|
7
|
-
FixedTokenChunker,
|
|
8
|
-
SemanticChunker,
|
|
9
|
-
SlidingWindowChunker,
|
|
10
|
-
available_chunkers,
|
|
11
|
-
create_chunker,
|
|
12
|
-
register_chunker,
|
|
13
|
-
)
|
|
6
|
+
from langparse.chunkers.semantic import SemanticChunker
|
|
14
7
|
from langparse.core.chunker import BaseChunker
|
|
15
8
|
from langparse.core.parser import BaseParser
|
|
16
9
|
from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
|
|
@@ -18,7 +11,6 @@ from langparse.parsers.docx_parser import DocxParser
|
|
|
18
11
|
from langparse.parsers.excel_parser import ExcelParser
|
|
19
12
|
from langparse.parsers.markdown_parser import MarkdownParser
|
|
20
13
|
from langparse.parsers.pdf_parser import PDFParser
|
|
21
|
-
from langparse.progress import ProgressCallback, ProgressEvent
|
|
22
14
|
from langparse.types import (
|
|
23
15
|
Chunk,
|
|
24
16
|
Document,
|
|
@@ -42,14 +34,7 @@ __all__ = [
|
|
|
42
34
|
"DocxParser",
|
|
43
35
|
"ExcelParser",
|
|
44
36
|
"SemanticChunker",
|
|
45
|
-
"FixedTokenChunker",
|
|
46
|
-
"SlidingWindowChunker",
|
|
47
|
-
"available_chunkers",
|
|
48
|
-
"create_chunker",
|
|
49
|
-
"register_chunker",
|
|
50
37
|
"ParseMetrics",
|
|
51
|
-
"ProgressCallback",
|
|
52
|
-
"ProgressEvent",
|
|
53
38
|
"BatchItemResult",
|
|
54
39
|
"BatchRunResult",
|
|
55
40
|
]
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
from collections.abc import Callable
|
|
4
|
-
from dataclasses import asdict
|
|
5
4
|
|
|
6
5
|
from openpyxl.utils import get_column_letter, range_boundaries
|
|
7
6
|
|
|
@@ -19,7 +18,6 @@ from langparse.workbooks.types import (
|
|
|
19
18
|
LogicalTable,
|
|
20
19
|
MatrixBlock,
|
|
21
20
|
MatrixHeader,
|
|
22
|
-
SourceRef,
|
|
23
21
|
TableContinuation,
|
|
24
22
|
TextBlock,
|
|
25
23
|
TextLine,
|
|
@@ -115,18 +113,6 @@ class WorkbookStructuralChunker:
|
|
|
115
113
|
sheet_snapshot.visibility if sheet_snapshot is not None else sheet_ir.visibility
|
|
116
114
|
)
|
|
117
115
|
chunk.metadata["hidden_row_numbers"] = sorted(referenced_rows & hidden_rows)
|
|
118
|
-
if self.policy.analysis_records:
|
|
119
|
-
lineage = workbook_ir.lineage
|
|
120
|
-
dependencies = []
|
|
121
|
-
for source in chunk.metadata["source_ranges"]:
|
|
122
|
-
name, cell_range = source.rsplit("!", 1)
|
|
123
|
-
for edge in (
|
|
124
|
-
lineage.dependencies_of(SourceRef(name, cell_range)) if lineage else []
|
|
125
|
-
):
|
|
126
|
-
payload = asdict(edge)
|
|
127
|
-
if payload not in dependencies:
|
|
128
|
-
dependencies.append(payload)
|
|
129
|
-
chunk.structured_payload["dependencies"] = dependencies
|
|
130
116
|
|
|
131
117
|
def _validate_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
|
|
132
118
|
workbook_ir = parsed.structure
|
|
@@ -155,12 +141,7 @@ class WorkbookStructuralChunker:
|
|
|
155
141
|
for chunk in chunks:
|
|
156
142
|
source_ranges = chunk.metadata["source_ranges"]
|
|
157
143
|
for source_range in source_ranges:
|
|
158
|
-
|
|
159
|
-
from langparse.workbooks.objects import validate_object_source
|
|
160
|
-
|
|
161
|
-
validate_object_source(workbook_ir.snapshot, source_range)
|
|
162
|
-
else:
|
|
163
|
-
_source_range_is_valid(workbook_ir.snapshot, source_range)
|
|
144
|
+
_source_range_is_valid(workbook_ir.snapshot, source_range)
|
|
164
145
|
if chunk.metadata["chunk_type"] != "table_rows":
|
|
165
146
|
continue
|
|
166
147
|
payload = chunk.structured_payload
|
|
@@ -186,29 +167,6 @@ class WorkbookStructuralChunker:
|
|
|
186
167
|
block: WorkbookBlock,
|
|
187
168
|
chunk_index_offset: int,
|
|
188
169
|
) -> list[Chunk]:
|
|
189
|
-
if block.kind in {"chart", "image"}:
|
|
190
|
-
from langparse.workbooks.objects import render_object
|
|
191
|
-
|
|
192
|
-
content = render_object(block)
|
|
193
|
-
metadata = document_metadata(parsed)
|
|
194
|
-
metadata.update(
|
|
195
|
-
{
|
|
196
|
-
"chunk_type": block.kind,
|
|
197
|
-
"chunk_index": chunk_index_offset,
|
|
198
|
-
"sheet_name": sheet_name,
|
|
199
|
-
"sheet_ordinal": sheet_ordinal,
|
|
200
|
-
"source_ranges": [ref.key for ref in block.source_refs],
|
|
201
|
-
"object_id": block.block_id,
|
|
202
|
-
"oversized": self.length_function(content) > self.max_chunk_size,
|
|
203
|
-
}
|
|
204
|
-
)
|
|
205
|
-
return [
|
|
206
|
-
Chunk(
|
|
207
|
-
content=content,
|
|
208
|
-
metadata=metadata,
|
|
209
|
-
structured_payload={"object": block.metadata["object"]},
|
|
210
|
-
)
|
|
211
|
-
]
|
|
212
170
|
if block.logical_table is not None:
|
|
213
171
|
return self._chunk_logical_table(
|
|
214
172
|
parsed,
|
|
@@ -6,9 +6,7 @@ from collections.abc import Sequence
|
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
|
|
8
8
|
from langparse import __version__
|
|
9
|
-
from langparse.chunkers.registry import available_chunkers
|
|
10
9
|
from langparse.errors import classify_exception
|
|
11
|
-
from langparse.progress import ProgressEvent
|
|
12
10
|
from langparse.services.batch_service import BatchParseService
|
|
13
11
|
from langparse.services.benchmark_service import BenchmarkService
|
|
14
12
|
from langparse.services.parse_service import ParseService
|
|
@@ -45,27 +43,18 @@ def build_parser():
|
|
|
45
43
|
parse_cmd.add_argument("--model-source", default=None)
|
|
46
44
|
parse_cmd.add_argument("--auto-install-runtime", action="store_true")
|
|
47
45
|
parse_cmd.add_argument("--runtime-package", default=None)
|
|
48
|
-
parse_cmd.add_argument("--format", default="markdown"
|
|
46
|
+
parse_cmd.add_argument("--format", default="markdown")
|
|
49
47
|
parse_cmd.add_argument("--batch", action="store_true")
|
|
50
48
|
parse_cmd.add_argument("--output", default=None)
|
|
51
49
|
parse_cmd.add_argument("--output-dir", default=None)
|
|
52
50
|
parse_cmd.add_argument("--max-workers", type=int, default=None)
|
|
53
51
|
parse_cmd.add_argument("--skip-existing", action="store_true")
|
|
54
52
|
parse_cmd.add_argument("--metrics", action="store_true")
|
|
55
|
-
parse_cmd.add_argument("--progress", action="store_true", help="write progress to stderr")
|
|
56
53
|
parse_cmd.add_argument(
|
|
57
54
|
"--chunk",
|
|
58
55
|
action="store_true",
|
|
59
56
|
help="semantically chunk the parsed document and include chunks in the output",
|
|
60
57
|
)
|
|
61
|
-
parse_cmd.add_argument("--chunk-strategy", choices=available_chunkers(), default=None)
|
|
62
|
-
parse_cmd.add_argument(
|
|
63
|
-
"--chunk-size",
|
|
64
|
-
type=int,
|
|
65
|
-
default=None,
|
|
66
|
-
help="strategy size budget (lexical tokens for fixed-token, characters otherwise)",
|
|
67
|
-
)
|
|
68
|
-
parse_cmd.add_argument("--chunk-overlap", type=int, default=None)
|
|
69
58
|
parse_cmd.add_argument(
|
|
70
59
|
"--chunk-profile",
|
|
71
60
|
choices=["retrieval", "analysis"],
|
|
@@ -251,22 +240,6 @@ def _run(args, parser) -> int:
|
|
|
251
240
|
if value is not None and value is not False
|
|
252
241
|
}
|
|
253
242
|
chunk_kwargs = {"chunk_profile": args.chunk_profile} if args.chunk else {}
|
|
254
|
-
if any(
|
|
255
|
-
value is not None for value in (args.chunk_strategy, args.chunk_size, args.chunk_overlap)
|
|
256
|
-
):
|
|
257
|
-
if not args.chunk:
|
|
258
|
-
parser.error("chunk strategy/size/overlap options require --chunk")
|
|
259
|
-
chunk_kwargs["chunk_strategy"] = args.chunk_strategy or "semantic"
|
|
260
|
-
chunk_kwargs["chunk_options"] = {
|
|
261
|
-
key: value
|
|
262
|
-
for key, value in {
|
|
263
|
-
"max_chunk_size": args.chunk_size,
|
|
264
|
-
"overlap": args.chunk_overlap,
|
|
265
|
-
}.items()
|
|
266
|
-
if value is not None
|
|
267
|
-
}
|
|
268
|
-
if args.progress:
|
|
269
|
-
parse_kwargs["progress_callback"] = _print_progress
|
|
270
243
|
|
|
271
244
|
if args.batch:
|
|
272
245
|
# One implementation regardless of flags. Without --output-dir the run
|
|
@@ -309,21 +282,5 @@ def _run(args, parser) -> int:
|
|
|
309
282
|
return 0
|
|
310
283
|
|
|
311
284
|
|
|
312
|
-
def _print_progress(event: ProgressEvent) -> None:
|
|
313
|
-
counts = ""
|
|
314
|
-
if event.completed_units is not None:
|
|
315
|
-
total = event.total_units if event.total_units is not None else "?"
|
|
316
|
-
counts = f" {event.completed_units}/{total} {event.unit or ''}"
|
|
317
|
-
percent = f" {event.percent:.0f}%" if event.percent is not None else ""
|
|
318
|
-
# Escape embedded line breaks/control characters from untrusted filenames/messages.
|
|
319
|
-
source = repr(event.source) if event.source else "batch"
|
|
320
|
-
message = f" {event.message!r}" if event.message else ""
|
|
321
|
-
print(
|
|
322
|
-
f"langparse: {source} {event.phase} {event.state}{counts}{percent}{message}",
|
|
323
|
-
file=sys.stderr,
|
|
324
|
-
flush=True,
|
|
325
|
-
)
|
|
326
|
-
|
|
327
|
-
|
|
328
285
|
if __name__ == "__main__":
|
|
329
286
|
raise SystemExit(main())
|
|
@@ -6,7 +6,6 @@ from typing import Any
|
|
|
6
6
|
from langparse.core.engine import PageResult
|
|
7
7
|
from langparse.engines.pdf.simple import BasePDFEngine
|
|
8
8
|
from langparse.logging import get_logger
|
|
9
|
-
from langparse.progress import ProgressCallback, ProgressReporter
|
|
10
9
|
from langparse.types import ParsedDocumentResult
|
|
11
10
|
|
|
12
11
|
logger = get_logger(__name__)
|
|
@@ -98,15 +97,7 @@ class DeepDocEngine(BasePDFEngine):
|
|
|
98
97
|
logger.warning("Skipping OCR-applied classification for %s: %s", file_path, exc)
|
|
99
98
|
return {}
|
|
100
99
|
|
|
101
|
-
def process_document(
|
|
102
|
-
self,
|
|
103
|
-
file_path: Path,
|
|
104
|
-
*,
|
|
105
|
-
progress_callback: ProgressCallback | None = None,
|
|
106
|
-
**kwargs: Any,
|
|
107
|
-
) -> ParsedDocumentResult:
|
|
108
|
-
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
109
|
-
reporter.emit("preparing", message="Preparing DeepDoc parser")
|
|
100
|
+
def process_document(self, file_path: Path, **kwargs: Any) -> ParsedDocumentResult:
|
|
110
101
|
try:
|
|
111
102
|
from langparse.engines.pdf.deepdoc.rendering import render_pages
|
|
112
103
|
except ImportError as exc:
|
|
@@ -117,19 +108,8 @@ class DeepDocEngine(BasePDFEngine):
|
|
|
117
108
|
with self._parser_lock:
|
|
118
109
|
if self._parser is None:
|
|
119
110
|
self._parser = self._build_parser()
|
|
120
|
-
|
|
121
|
-
boxes = self._parser.parse_into_bboxes(str(file_path))
|
|
122
|
-
else:
|
|
123
|
-
last_percent = 0.0
|
|
124
|
-
|
|
125
|
-
def on_progress(progress, message=""):
|
|
126
|
-
nonlocal last_percent
|
|
127
|
-
last_percent = max(last_percent, min(100.0, max(0.0, progress * 100)))
|
|
128
|
-
reporter.emit("parsing", percent=last_percent, message=message)
|
|
129
|
-
|
|
130
|
-
boxes = self._parser.parse_into_bboxes(str(file_path), callback=on_progress)
|
|
111
|
+
boxes = self._parser.parse_into_bboxes(str(file_path))
|
|
131
112
|
|
|
132
|
-
reporter.emit("rendering")
|
|
133
113
|
ocr_pages = self._classify_ocr_pages(file_path)
|
|
134
114
|
pages = render_pages(boxes, ocr_pages=ocr_pages)
|
|
135
115
|
return ParsedDocumentResult(
|
|
@@ -146,11 +126,7 @@ class DeepDocEngine(BasePDFEngine):
|
|
|
146
126
|
},
|
|
147
127
|
)
|
|
148
128
|
|
|
149
|
-
def process(
|
|
150
|
-
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
151
|
-
) -> Iterator[PageResult]:
|
|
152
|
-
if progress_callback is not None:
|
|
153
|
-
kwargs["progress_callback"] = progress_callback
|
|
129
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
154
130
|
parsed = self.process_document(file_path, **kwargs)
|
|
155
131
|
for page in parsed.pages:
|
|
156
132
|
yield PageResult(
|
|
@@ -6,7 +6,6 @@ from langparse.core.engine import PageResult
|
|
|
6
6
|
from langparse.engines.pdf.mineru_client import MinerUClient
|
|
7
7
|
from langparse.engines.pdf.mineru_service import MinerUServiceManager
|
|
8
8
|
from langparse.engines.pdf.simple import BasePDFEngine
|
|
9
|
-
from langparse.progress import ProgressCallback, ProgressReporter
|
|
10
9
|
from langparse.types import ParsedDocumentResult, ParsedElement, ParsedPageResult
|
|
11
10
|
|
|
12
11
|
_SENSITIVE_OPTION_KEYS = frozenset(
|
|
@@ -152,35 +151,16 @@ class MinerUEngine(BasePDFEngine):
|
|
|
152
151
|
def _create_service_manager(self) -> MinerUServiceManager:
|
|
153
152
|
return MinerUServiceManager(**self._build_service_config())
|
|
154
153
|
|
|
155
|
-
def _run_mineru(
|
|
156
|
-
self,
|
|
157
|
-
file_path: Path,
|
|
158
|
-
runtime_config: dict[str, Any],
|
|
159
|
-
progress_callback: ProgressCallback | None = None,
|
|
160
|
-
) -> list[dict[str, Any]]:
|
|
161
|
-
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
154
|
+
def _run_mineru(self, file_path: Path, runtime_config: dict[str, Any]) -> list[dict[str, Any]]:
|
|
162
155
|
manager = self._create_service_manager()
|
|
163
156
|
with manager.running_service() as base_url:
|
|
164
157
|
client = self._create_client(base_url)
|
|
165
|
-
reporter.emit("parsing", message="Waiting for MinerU response")
|
|
166
158
|
return client.parse_file(file_path, runtime_config)
|
|
167
159
|
|
|
168
|
-
def process_document(
|
|
169
|
-
self,
|
|
170
|
-
file_path: Path,
|
|
171
|
-
*,
|
|
172
|
-
progress_callback: ProgressCallback | None = None,
|
|
173
|
-
**kwargs: Any,
|
|
174
|
-
) -> ParsedDocumentResult:
|
|
175
|
-
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
176
|
-
reporter.emit("preparing", message="Preparing MinerU service")
|
|
160
|
+
def process_document(self, file_path: Path, **kwargs: Any) -> ParsedDocumentResult:
|
|
177
161
|
self._ensure_runtime()
|
|
178
162
|
runtime_config = self._build_runtime_config(**kwargs)
|
|
179
|
-
|
|
180
|
-
raw_pages = self._run_mineru(file_path, runtime_config)
|
|
181
|
-
else:
|
|
182
|
-
raw_pages = self._run_mineru(file_path, runtime_config, progress_callback)
|
|
183
|
-
reporter.emit("rendering")
|
|
163
|
+
raw_pages = self._run_mineru(file_path, runtime_config)
|
|
184
164
|
pages = [
|
|
185
165
|
ParsedPageResult(
|
|
186
166
|
page_number=item["page_number"],
|
|
@@ -241,11 +221,7 @@ class MinerUEngine(BasePDFEngine):
|
|
|
241
221
|
},
|
|
242
222
|
)
|
|
243
223
|
|
|
244
|
-
def process(
|
|
245
|
-
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
246
|
-
) -> Iterator[PageResult]:
|
|
247
|
-
if progress_callback is not None:
|
|
248
|
-
kwargs["progress_callback"] = progress_callback
|
|
224
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
249
225
|
parsed = self.process_document(file_path, **kwargs)
|
|
250
226
|
for page in parsed.pages:
|
|
251
227
|
yield PageResult(
|
|
@@ -10,7 +10,6 @@ from langparse.engines.pdf.ocr import (
|
|
|
10
10
|
needs_ocr,
|
|
11
11
|
ocr_page_text,
|
|
12
12
|
)
|
|
13
|
-
from langparse.progress import ProgressCallback, ProgressReporter
|
|
14
13
|
|
|
15
14
|
|
|
16
15
|
class BasePDFEngine(BaseEngine):
|
|
@@ -50,10 +49,7 @@ class SimplePDFEngine(BasePDFEngine):
|
|
|
50
49
|
# guarantee -- correctness over throughput on an already slow path.
|
|
51
50
|
self._ocr_lock = threading.Lock()
|
|
52
51
|
|
|
53
|
-
def process(
|
|
54
|
-
self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
55
|
-
) -> Iterator[PageResult]:
|
|
56
|
-
reporter = ProgressReporter(str(file_path), progress_callback)
|
|
52
|
+
def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
|
|
57
53
|
try:
|
|
58
54
|
import pdfplumber
|
|
59
55
|
except ImportError:
|
|
@@ -76,9 +72,6 @@ class SimplePDFEngine(BasePDFEngine):
|
|
|
76
72
|
if table_markdown:
|
|
77
73
|
markdown_content = "\n\n".join([text, "\n".join(table_markdown)]).strip()
|
|
78
74
|
|
|
79
|
-
reporter.emit(
|
|
80
|
-
"parsing", completed_units=i + 1, total_units=len(pdf.pages), unit="pages"
|
|
81
|
-
)
|
|
82
75
|
yield PageResult(
|
|
83
76
|
page_number=i + 1,
|
|
84
77
|
markdown_content=markdown_content,
|
|
@@ -1,10 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
import re
|
|
4
3
|
from dataclasses import dataclass
|
|
5
4
|
from enum import Enum
|
|
6
|
-
from subprocess import TimeoutExpired
|
|
7
|
-
from urllib.error import URLError
|
|
8
5
|
|
|
9
6
|
|
|
10
7
|
class ErrorType(str, Enum):
|
|
@@ -45,7 +42,7 @@ def classify_exception(exc: BaseException) -> ClassifiedError:
|
|
|
45
42
|
return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
|
|
46
43
|
if "unable to start local mineru-api" in lowered:
|
|
47
44
|
return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
|
|
48
|
-
if
|
|
45
|
+
if "timed out" in lowered or "timeout" in lowered:
|
|
49
46
|
return ClassifiedError(ErrorType.ENGINE_TIMEOUT, message)
|
|
50
47
|
if "ocr" in lowered and "unavailable" in lowered:
|
|
51
48
|
return ClassifiedError(ErrorType.OCR_UNAVAILABLE, message)
|
|
@@ -53,18 +50,3 @@ def classify_exception(exc: BaseException) -> ClassifiedError:
|
|
|
53
50
|
return ClassifiedError(ErrorType.TABLE_EXTRACTION_FAILED, message)
|
|
54
51
|
|
|
55
52
|
return ClassifiedError(ErrorType.PARSE_FAILED, message)
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def _is_timeout(exc: BaseException) -> bool:
|
|
59
|
-
"""Recognize standard timeout types through explicit backend wrappers."""
|
|
60
|
-
seen: set[int] = set()
|
|
61
|
-
current: BaseException | None = exc
|
|
62
|
-
while current is not None and id(current) not in seen:
|
|
63
|
-
seen.add(id(current))
|
|
64
|
-
if isinstance(current, (TimeoutError, TimeoutExpired)):
|
|
65
|
-
return True
|
|
66
|
-
if isinstance(current, URLError) and isinstance(current.reason, BaseException):
|
|
67
|
-
current = current.reason
|
|
68
|
-
else:
|
|
69
|
-
current = current.__cause__
|
|
70
|
-
return False
|
|
@@ -3,7 +3,6 @@ from pathlib import Path
|
|
|
3
3
|
|
|
4
4
|
from langparse.core.parser import BaseParser
|
|
5
5
|
from langparse.parsers.sniff import looks_like_ole_binary, looks_like_zip_ooxml
|
|
6
|
-
from langparse.progress import ProgressCallback, ProgressReporter
|
|
7
6
|
from langparse.types import ParsedDocumentResult, ParsedElement, ParseDiagnostics, ParsedPageResult
|
|
8
7
|
from langparse.workbooks.modeling import (
|
|
9
8
|
RequiredWorkbookDisambiguationError,
|
|
@@ -65,14 +64,11 @@ class ExcelParser(BaseParser):
|
|
|
65
64
|
WorkbookDisambiguation.off() if disambiguation is None else disambiguation
|
|
66
65
|
)
|
|
67
66
|
|
|
68
|
-
def parse_result(
|
|
69
|
-
self, file_path: str | Path, *, progress_callback: ProgressCallback | None = None, **kwargs
|
|
70
|
-
) -> ParsedDocumentResult:
|
|
67
|
+
def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
|
|
71
68
|
path = self._resolve_existing_path(file_path)
|
|
72
|
-
reporter = ProgressReporter(path, progress_callback)
|
|
73
69
|
|
|
74
70
|
if looks_like_zip_ooxml(path):
|
|
75
|
-
return self._parse_ooxml(path
|
|
71
|
+
return self._parse_ooxml(path)
|
|
76
72
|
|
|
77
73
|
try:
|
|
78
74
|
import pandas as pd
|
|
@@ -86,7 +82,6 @@ class ExcelParser(BaseParser):
|
|
|
86
82
|
# otherwise be handed to the wrong pandas reader. Content decides:
|
|
87
83
|
# a real workbook is either a ZIP-OOXML or legacy-OLE container;
|
|
88
84
|
# anything else is read as delimited text regardless of its label.
|
|
89
|
-
reporter.emit("extracting")
|
|
90
85
|
is_legacy_workbook = looks_like_ole_binary(path)
|
|
91
86
|
if is_legacy_workbook:
|
|
92
87
|
sheets = pd.read_excel(path, sheet_name=None)
|
|
@@ -110,7 +105,6 @@ class ExcelParser(BaseParser):
|
|
|
110
105
|
)
|
|
111
106
|
}
|
|
112
107
|
|
|
113
|
-
reporter.emit("rendering")
|
|
114
108
|
pages = [
|
|
115
109
|
self._page_for_sheet(index + 1, sheet_name, frame)
|
|
116
110
|
for index, (sheet_name, frame) in enumerate(sheets.items())
|
|
@@ -138,7 +132,7 @@ class ExcelParser(BaseParser):
|
|
|
138
132
|
),
|
|
139
133
|
)
|
|
140
134
|
|
|
141
|
-
def _parse_ooxml(self, path: Path
|
|
135
|
+
def _parse_ooxml(self, path: Path) -> ParsedDocumentResult:
|
|
142
136
|
try:
|
|
143
137
|
from langparse.workbooks.adapters import OOXMLWorkbookAdapter
|
|
144
138
|
from langparse.workbooks.assembly import assemble_baseline, assemble_workbook
|
|
@@ -152,9 +146,7 @@ class ExcelParser(BaseParser):
|
|
|
152
146
|
"Install with `pip install langparse[excel]`."
|
|
153
147
|
) from None
|
|
154
148
|
|
|
155
|
-
reporter.emit("extracting")
|
|
156
149
|
snapshot = OOXMLWorkbookAdapter().snapshot(path)
|
|
157
|
-
reporter.emit("assembling")
|
|
158
150
|
try:
|
|
159
151
|
structure, diagnostics = assemble_workbook(
|
|
160
152
|
snapshot,
|
|
@@ -168,10 +160,6 @@ class ExcelParser(BaseParser):
|
|
|
168
160
|
diagnostics.warnings.append(
|
|
169
161
|
f"Semantic workbook assembly failed; retained raw-grid fallback: {type(exc).__name__}"
|
|
170
162
|
)
|
|
171
|
-
from langparse.workbooks.objects import attach_objects
|
|
172
|
-
|
|
173
|
-
attach_objects(snapshot, structure, diagnostics)
|
|
174
|
-
reporter.emit("rendering")
|
|
175
163
|
pages = compatibility_pages(snapshot, structure)
|
|
176
164
|
markdown = render_workbook_markdown(snapshot, structure)
|
|
177
165
|
return ParsedDocumentResult(
|
|
File without changes
|