docslicer 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {docslicer-0.2.0/src/docslicer.egg-info → docslicer-0.2.1}/PKG-INFO +159 -10
- {docslicer-0.2.0 → docslicer-0.2.1}/README.md +155 -9
- {docslicer-0.2.0 → docslicer-0.2.1}/pyproject.toml +8 -1
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_config.py +2 -1
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_orchestrator.py +5 -1
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_result.py +81 -14
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/html_orchestrator.py +46 -17
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_01_box_extractor.py +25 -6
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_01_static_box_extractor.py +219 -110
- docslicer-0.2.1/src/docslicer/mcp/__init__.py +16 -0
- docslicer-0.2.1/src/docslicer/mcp/_store.py +336 -0
- docslicer-0.2.1/src/docslicer/mcp/server.py +1062 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/schema.py +4 -0
- docslicer-0.2.1/src/docslicer/ocr/__init__.py +9 -0
- docslicer-0.2.1/src/docslicer/ocr/_availability.py +61 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/pdf_orchestrator.py +34 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_05_heading_detector.py +21 -10
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_07_block_merger.py +36 -5
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_08_chunk_builder.py +188 -35
- {docslicer-0.2.0 → docslicer-0.2.1/src/docslicer.egg-info}/PKG-INFO +159 -10
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/SOURCES.txt +6 -1
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/entry_points.txt +1 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/requires.txt +4 -0
- docslicer-0.2.1/tests/test_table_cell_newlines.py +92 -0
- docslicer-0.2.0/src/docslicer/ocr/__init__.py +0 -3
- {docslicer-0.2.0 → docslicer-0.2.1}/LICENSE +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/LICENSE-COMMERCIAL.md +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/setup.cfg +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/color_utils.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/cpu.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/registry_aggregator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/text_merge.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/export_debug.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/reorder_columns.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/exhibit_patterns.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/hierarchy_type_patterns.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/page_label_patterns.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_loader.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/gutter_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/layouts.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/line_merger.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/line_number_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/reading_order.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/shape_processor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/oxm_package.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/parallel.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/password.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/safe_call.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_header.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_normalize.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_schema.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/text_utils.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/timing.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/cli.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/common_author_names.csv +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/exhibit_patterns.yaml +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/hierarchy_type_patterns.yaml +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/page_label_patterns.yaml +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/docx_orchestrator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/native_metadata.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_01_package_reader.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_02_run_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_03_chart_point_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_04_table_cell_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_05_paragraph_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_06_line_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_07_style_prefiller.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/extract_boxes.js +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/native_metadata.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_02_box_cleaner.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_03_page_label_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_04_line_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_05_table_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_06_style_prefiller.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/consolidate.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/generator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/ocr_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/page_analysis.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/text_fallback.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/ocr_orchestrator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_01_word_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_02_word_colorizer.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_03_shape_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_04_text_cleaner.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_05_font_size_estimator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/coordinates.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/form_fields.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/form_label_link.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/line_classification.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/page_rotation.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/script_thresholds.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/struct_context.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/struct_tree.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/native_metadata.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_01_word_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_02_image_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_03_shape_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_04_link_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_05_struct_group.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_06_style_prefiller.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_07_stream_group.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_08_reading_order.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_09_word_relationships.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_10_cell_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_11_page_label_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_12_cell_grouper.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_13_line_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_14_table_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/native_metadata.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/pptx_orchestrator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_01_package_reader.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_02_run_extractor.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_03_chart_point_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_04_table_cell_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_05_paragraph_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_06_reading_order.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_07_line_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_08_style_prefiller.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/config.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/cookie_consent.js +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/dispatcher.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/http_fetcher.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/sec_fetcher.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/models.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/stealth_init.js +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/__init__.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/shared_orchestrator.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_01_navigation_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_02_toc_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_03_exhibit_detector.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_04_section_classifier.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_06_hierarchy_builder.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/dependency_links.txt +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/top_level.txt +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_api.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_charts.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_document_parser.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_errors.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_exports.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_loose_box_reconstruction.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_packaging.py +0 -0
- {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_smoke.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: docslicer
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Deterministic hierarchical document parser and chunker
|
|
5
5
|
Author-email: "Market Framer Inc." <jelle@docslicer.ai>
|
|
6
6
|
License-Expression: AGPL-3.0-only
|
|
@@ -50,6 +50,9 @@ Provides-Extra: crypto
|
|
|
50
50
|
Requires-Dist: msoffcrypto-tool>=5.0; extra == "crypto"
|
|
51
51
|
Provides-Extra: parquet
|
|
52
52
|
Requires-Dist: pyarrow>=18.0; extra == "parquet"
|
|
53
|
+
Provides-Extra: mcp
|
|
54
|
+
Requires-Dist: mcp>=2.0; extra == "mcp"
|
|
55
|
+
Requires-Dist: tiktoken>=0.10; extra == "mcp"
|
|
53
56
|
Provides-Extra: dev
|
|
54
57
|
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
55
58
|
Requires-Dist: pytest-asyncio>=0.24; extra == "dev"
|
|
@@ -57,11 +60,17 @@ Dynamic: license-file
|
|
|
57
60
|
|
|
58
61
|
# DocSlicer
|
|
59
62
|
|
|
60
|
-
[](LICENSE) [](LICENSE-COMMERCIAL.md)
|
|
63
|
+
[](https://pypi.org/project/docslicer/) [](LICENSE) [](LICENSE-COMMERCIAL.md)
|
|
61
64
|
|
|
62
|
-
Lightning-fast, deterministic
|
|
65
|
+
Lightning-fast (31 pages/sec), deterministic document parser and chunker for business documents. No LLM calls or heavy ML models.
|
|
63
66
|
|
|
64
|
-
DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, and a navigable heading hierarchy
|
|
67
|
+
DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, markdown and a navigable heading hierarchy.
|
|
68
|
+
|
|
69
|
+
Top score on [BizDocBench](https://github.com/DocSlicer/BizDocBench) (0.88 overall vs 0.70 for the next-best tool). 0.80 table accuracy, 0.98 content faithfulness, 0.85 heading recognition and hierarchy preservation, and 0.76 RAG retrieval performance.
|
|
70
|
+
|
|
71
|
+
**Add DocSlicer to your AI pipeline:**
|
|
72
|
+
- Classic RAG: the layout-aware chunker gives you clean non-overlapping chunks, each carrying its full heading breadcrumb, ready to embed.
|
|
73
|
+
- Vectorless RAG: for when you want an answer out of a document right now. The agent pulls the outline, picks the section it needs, and navigates to the correct section of text without embedding the whole document
|
|
65
74
|
|
|
66
75
|
```python
|
|
67
76
|
import docslicer
|
|
@@ -117,6 +126,23 @@ if __name__ == "__main__":
|
|
|
117
126
|
|
|
118
127
|
---
|
|
119
128
|
|
|
129
|
+
## Benchmarks
|
|
130
|
+
|
|
131
|
+
Measured with [BizDocBench](https://github.com/DocSlicer/BizDocBench) — an open benchmark for multi-format business document parsing. All scores are 0–1 (higher is better); `pages_per_sec_aggregate` is throughput across the full corpus.
|
|
132
|
+
|
|
133
|
+
| Tool | Score | Coverage | Speed | Hierarchy | Faithfulness | Tables | Retrieval | Pages/sec |
|
|
134
|
+
|---|---|---|---|---|---|---|---|---|
|
|
135
|
+
| **docslicer** | **0.8796** | 1.0000 | 0.8836 | 0.8466 | 0.9824 | 0.8047 | 0.7601 | 31.27 |
|
|
136
|
+
| docling | 0.7036 | 1.0000 | 0.3805 | 0.4905 | 0.8927 | 0.7467 | 0.7111 | 3.46 |
|
|
137
|
+
| markitdown | 0.5838 | 1.0000 | 0.8513 | 0.0604 | 0.7972 | 0.2584 | 0.5357 | 27.42 |
|
|
138
|
+
| unstructured | 0.5798 | 0.9091 | 0.1073 | 0.4327 | 0.9057 | 0.4812 | 0.6430 | 0.52 |
|
|
139
|
+
| opendataloader | 0.5359 | 0.5844 | 1.0000 | 0.3853 | 0.6484 | 0.2655 | 0.3317 | 117.26 |
|
|
140
|
+
| pymupdf4llm | 0.4519 | 0.5974 | 0.6492 | 0.1089 | 0.6456 | 0.3551 | 0.3552 | 11.84 |
|
|
141
|
+
| mineru | 0.4107 | 0.5974 | 0.1353 | 0.4220 | 0.6176 | 0.3012 | 0.3910 | 0.70 |
|
|
142
|
+
| marker | 0.3735 | 0.5974 | 0.1598 | 0.1926 | 0.6121 | 0.3012 | 0.3778 | 0.87 |
|
|
143
|
+
|
|
144
|
+
---
|
|
145
|
+
|
|
120
146
|
## Install
|
|
121
147
|
|
|
122
148
|
```bash
|
|
@@ -135,6 +161,7 @@ pip install 'docslicer[ocr]' # scanned PDF support via Tesseract + OpenCV
|
|
|
135
161
|
# Linux: apt install tesseract-ocr
|
|
136
162
|
# macOS: brew install tesseract
|
|
137
163
|
|
|
164
|
+
pip install 'docslicer[mcp]' # MCP server for LLM clients (Claude, Cursor, …)
|
|
138
165
|
pip install 'docslicer[llm]' # exact token counts via tiktoken (exact_tokens=True)
|
|
139
166
|
pip install 'docslicer[crypto]' # password-protected Office files (msoffcrypto-tool)
|
|
140
167
|
pip install 'docslicer[parquet]' # Parquet export support
|
|
@@ -437,22 +464,51 @@ result.tables_by_page(14)
|
|
|
437
464
|
result.charts_by_page(14)
|
|
438
465
|
```
|
|
439
466
|
|
|
467
|
+
### Parse once, navigate many times
|
|
468
|
+
|
|
469
|
+
A parsed result is plain data, so you can persist it and reload it later. When an
|
|
470
|
+
agent asks many questions about the same document, there's no need to parse it
|
|
471
|
+
again on every question:
|
|
472
|
+
|
|
473
|
+
```python
|
|
474
|
+
from pathlib import Path
|
|
475
|
+
import docslicer
|
|
476
|
+
|
|
477
|
+
cache = Path("annual_report.json")
|
|
478
|
+
|
|
479
|
+
if cache.exists():
|
|
480
|
+
result = docslicer.ParseResult.load(cache)
|
|
481
|
+
else:
|
|
482
|
+
result = docslicer.parse_document("annual_report.pdf")
|
|
483
|
+
result.save(cache)
|
|
484
|
+
```
|
|
485
|
+
|
|
486
|
+
A reloaded result supports the full API — `hierarchy`, `find_heading`,
|
|
487
|
+
`chunks_under`, `tables` — so a long-running agent session or document server can
|
|
488
|
+
keep documents open across requests without re-parsing.
|
|
489
|
+
|
|
440
490
|
---
|
|
441
491
|
|
|
442
492
|
## Export
|
|
443
493
|
|
|
494
|
+
`save()` decides what to write from the path you give it.
|
|
495
|
+
|
|
444
496
|
```python
|
|
445
|
-
# Save
|
|
446
|
-
result.save("
|
|
447
|
-
|
|
448
|
-
# (+ charts.parquet when the document has charts)
|
|
497
|
+
# Save the whole result and reload it later — keeps the heading hierarchy
|
|
498
|
+
result.save("result.json") # same output as result.to_json()
|
|
499
|
+
result = docslicer.ParseResult.load("result.json")
|
|
449
500
|
|
|
450
|
-
#
|
|
501
|
+
# A single collection, in the format you name
|
|
451
502
|
result.save("chunks.csv")
|
|
452
503
|
result.save("charts.jsonl") # stems: chunks | blocks | tables | charts | metadata
|
|
453
|
-
result.save("result.json") # full parse result as JSON
|
|
454
504
|
result.export_chunks_jsonl("chunks.jsonl")
|
|
455
505
|
|
|
506
|
+
# One file per collection
|
|
507
|
+
result.save("output/")
|
|
508
|
+
# → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
|
|
509
|
+
# (+ charts.parquet when the document has charts)
|
|
510
|
+
# Falls back to .csv unless the [parquet] extra is installed.
|
|
511
|
+
|
|
456
512
|
# Render as Markdown or plain text
|
|
457
513
|
md = result.export_to_markdown(include_tables=True)
|
|
458
514
|
txt = result.export_to_text()
|
|
@@ -461,6 +517,9 @@ txt = result.export_to_text()
|
|
|
461
517
|
df = result.chunks_df()
|
|
462
518
|
```
|
|
463
519
|
|
|
520
|
+
Only `result.json` round-trips — the collection and directory forms write flat rows
|
|
521
|
+
without the heading hierarchy, so `ParseResult.load()` can't read them back.
|
|
522
|
+
|
|
464
523
|
### Debug mode
|
|
465
524
|
|
|
466
525
|
```python
|
|
@@ -492,6 +551,96 @@ pip install 'docslicer[ocr]'
|
|
|
492
551
|
|
|
493
552
|
---
|
|
494
553
|
|
|
554
|
+
## MCP server
|
|
555
|
+
|
|
556
|
+
DocSlicer ships an [MCP](https://modelcontextprotocol.io) server, so LLM clients
|
|
557
|
+
(Claude Desktop, Claude Code, Cursor, …) can parse and read documents directly.
|
|
558
|
+
|
|
559
|
+
```bash
|
|
560
|
+
pip install 'docslicer[mcp]'
|
|
561
|
+
docslicer-mcp # stdio — what desktop clients launch
|
|
562
|
+
docslicer-mcp --transport http --port 8000
|
|
563
|
+
```
|
|
564
|
+
|
|
565
|
+
Register it with a client by adding to its MCP config:
|
|
566
|
+
|
|
567
|
+
```jsonc
|
|
568
|
+
{
|
|
569
|
+
"mcpServers": {
|
|
570
|
+
"docslicer": {
|
|
571
|
+
"command": "docslicer-mcp",
|
|
572
|
+
"env": { "DOCSLICER_MCP_ROOT": "/Users/you/Documents" }
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
```
|
|
577
|
+
|
|
578
|
+
### How it works
|
|
579
|
+
|
|
580
|
+
A parsed document is far larger than a model's context window, so the server
|
|
581
|
+
never returns one in a single call. `parse` registers the document and hands
|
|
582
|
+
back a `doc_id` handle plus a heading outline. Every other tool takes that
|
|
583
|
+
handle and returns a bounded slice — the model pulls in only what it needs.
|
|
584
|
+
|
|
585
|
+
| Tool | Returns |
|
|
586
|
+
| --- | --- |
|
|
587
|
+
| `parse` | `doc_id` handle, title, page count, heading outline |
|
|
588
|
+
| `get_outline` | The outline again, for when it scrolls out of context |
|
|
589
|
+
| `read` | The text under one or more headings, named from the outline |
|
|
590
|
+
| `search` | Headings to `read`, ranked, each with a snippet |
|
|
591
|
+
| `to_markdown` | Writes the whole document to disk; returns the path |
|
|
592
|
+
|
|
593
|
+
Every outline line carries what reading it would cost:
|
|
594
|
+
|
|
595
|
+
```
|
|
596
|
+
- Financial statements ~48k
|
|
597
|
+
- Note 14 — Segment reporting ~900
|
|
598
|
+
- Note 15 — Income taxes ~2.1k
|
|
599
|
+
```
|
|
600
|
+
|
|
601
|
+
That figure is the same estimate `read` reports back, so a budget made from the
|
|
602
|
+
outline holds when it is spent. Sizes are cumulative — a parent never costs less
|
|
603
|
+
than the children beneath it — which is what makes "descend or just read it" a
|
|
604
|
+
decision the model can make before spending the context rather than after.
|
|
605
|
+
|
|
606
|
+
`read` takes heading text exactly as the outline prints it. Where a heading
|
|
607
|
+
appears twice, prefixing any ancestor disambiguates it (`"Notes > Revenue"`);
|
|
608
|
+
the full chain is never required. Returned text is interleaved with `[Page X]`
|
|
609
|
+
markers using the document's own page labels (`S-23`, `iv`), so a quotation can
|
|
610
|
+
be cited to the page it actually came from rather than to wherever its section
|
|
611
|
+
began.
|
|
612
|
+
|
|
613
|
+
`search` is the fallback for when the outline does not settle the question —
|
|
614
|
+
headings that name nothing useful (`Note 14`, `Item 7A`), or a figure buried in
|
|
615
|
+
a table no heading mentions. It combines a whole-word literal match with BM25
|
|
616
|
+
over the chunks, and returns *places*, not answers: each hit is a heading to
|
|
617
|
+
pass to `read`. Query terms that appear nowhere in the document are reported
|
|
618
|
+
back, so a query that scored well on one rare word can be recognised as the bad
|
|
619
|
+
query it was.
|
|
620
|
+
|
|
621
|
+
`to_markdown` is the escape hatch for when the user wants the document itself
|
|
622
|
+
rather than an answer drawn from it. It writes to disk and returns a path, so
|
|
623
|
+
nothing enters the model's context and document size stops mattering.
|
|
624
|
+
|
|
625
|
+
Parsed results are cached on disk, so re-parsing the same file with the same
|
|
626
|
+
options is free. The cache key includes the file's size and mtime — edit the
|
|
627
|
+
document and the next `parse` re-parses it automatically.
|
|
628
|
+
|
|
629
|
+
### Configuration
|
|
630
|
+
|
|
631
|
+
| Variable | Effect |
|
|
632
|
+
| --- | --- |
|
|
633
|
+
| `DOCSLICER_MCP_ROOT` | Restrict file sources **and** written output to this directory tree |
|
|
634
|
+
| `DOCSLICER_MCP_ALLOW_URLS` | Set to `0` to reject `http(s)` sources |
|
|
635
|
+
| `DOCSLICER_MCP_CACHE` | Where parsed results are persisted (default `~/.cache/docslicer-mcp`) |
|
|
636
|
+
| `DOCSLICER_MCP_CACHE_MAX_MB` | Cache size ceiling, oldest pruned first (default `2048`; `0` disables) |
|
|
637
|
+
|
|
638
|
+
Set `DOCSLICER_MCP_ROOT` when exposing the server to anything but yourself —
|
|
639
|
+
without it, any readable path on the machine is parseable, and `to_markdown`
|
|
640
|
+
can write anywhere the server process can.
|
|
641
|
+
|
|
642
|
+
---
|
|
643
|
+
|
|
495
644
|
## Format-specific functions
|
|
496
645
|
|
|
497
646
|
If you know the format upfront and want explicit failure on unexpected input, use the
|
|
@@ -1,10 +1,16 @@
|
|
|
1
1
|
# DocSlicer
|
|
2
2
|
|
|
3
|
-
[](LICENSE) [](LICENSE-COMMERCIAL.md)
|
|
3
|
+
[](https://pypi.org/project/docslicer/) [](LICENSE) [](LICENSE-COMMERCIAL.md)
|
|
4
4
|
|
|
5
|
-
Lightning-fast, deterministic
|
|
5
|
+
Lightning-fast (31 pages/sec), deterministic document parser and chunker for business documents. No LLM calls or heavy ML models.
|
|
6
6
|
|
|
7
|
-
DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, and a navigable heading hierarchy
|
|
7
|
+
DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, markdown and a navigable heading hierarchy.
|
|
8
|
+
|
|
9
|
+
Top score on [BizDocBench](https://github.com/DocSlicer/BizDocBench) (0.88 overall vs 0.70 for the next-best tool). 0.80 table accuracy, 0.98 content faithfulness, 0.85 heading recognition and hierarchy preservation, and 0.76 RAG retrieval performance.
|
|
10
|
+
|
|
11
|
+
**Add DocSlicer to your AI pipeline:**
|
|
12
|
+
- Classic RAG: the layout-aware chunker gives you clean non-overlapping chunks, each carrying its full heading breadcrumb, ready to embed.
|
|
13
|
+
- Vectorless RAG: for when you want an answer out of a document right now. The agent pulls the outline, picks the section it needs, and navigates to the correct section of text without embedding the whole document
|
|
8
14
|
|
|
9
15
|
```python
|
|
10
16
|
import docslicer
|
|
@@ -60,6 +66,23 @@ if __name__ == "__main__":
|
|
|
60
66
|
|
|
61
67
|
---
|
|
62
68
|
|
|
69
|
+
## Benchmarks
|
|
70
|
+
|
|
71
|
+
Measured with [BizDocBench](https://github.com/DocSlicer/BizDocBench) — an open benchmark for multi-format business document parsing. All scores are 0–1 (higher is better); `pages_per_sec_aggregate` is throughput across the full corpus.
|
|
72
|
+
|
|
73
|
+
| Tool | Score | Coverage | Speed | Hierarchy | Faithfulness | Tables | Retrieval | Pages/sec |
|
|
74
|
+
|---|---|---|---|---|---|---|---|---|
|
|
75
|
+
| **docslicer** | **0.8796** | 1.0000 | 0.8836 | 0.8466 | 0.9824 | 0.8047 | 0.7601 | 31.27 |
|
|
76
|
+
| docling | 0.7036 | 1.0000 | 0.3805 | 0.4905 | 0.8927 | 0.7467 | 0.7111 | 3.46 |
|
|
77
|
+
| markitdown | 0.5838 | 1.0000 | 0.8513 | 0.0604 | 0.7972 | 0.2584 | 0.5357 | 27.42 |
|
|
78
|
+
| unstructured | 0.5798 | 0.9091 | 0.1073 | 0.4327 | 0.9057 | 0.4812 | 0.6430 | 0.52 |
|
|
79
|
+
| opendataloader | 0.5359 | 0.5844 | 1.0000 | 0.3853 | 0.6484 | 0.2655 | 0.3317 | 117.26 |
|
|
80
|
+
| pymupdf4llm | 0.4519 | 0.5974 | 0.6492 | 0.1089 | 0.6456 | 0.3551 | 0.3552 | 11.84 |
|
|
81
|
+
| mineru | 0.4107 | 0.5974 | 0.1353 | 0.4220 | 0.6176 | 0.3012 | 0.3910 | 0.70 |
|
|
82
|
+
| marker | 0.3735 | 0.5974 | 0.1598 | 0.1926 | 0.6121 | 0.3012 | 0.3778 | 0.87 |
|
|
83
|
+
|
|
84
|
+
---
|
|
85
|
+
|
|
63
86
|
## Install
|
|
64
87
|
|
|
65
88
|
```bash
|
|
@@ -78,6 +101,7 @@ pip install 'docslicer[ocr]' # scanned PDF support via Tesseract + OpenCV
|
|
|
78
101
|
# Linux: apt install tesseract-ocr
|
|
79
102
|
# macOS: brew install tesseract
|
|
80
103
|
|
|
104
|
+
pip install 'docslicer[mcp]' # MCP server for LLM clients (Claude, Cursor, …)
|
|
81
105
|
pip install 'docslicer[llm]' # exact token counts via tiktoken (exact_tokens=True)
|
|
82
106
|
pip install 'docslicer[crypto]' # password-protected Office files (msoffcrypto-tool)
|
|
83
107
|
pip install 'docslicer[parquet]' # Parquet export support
|
|
@@ -380,22 +404,51 @@ result.tables_by_page(14)
|
|
|
380
404
|
result.charts_by_page(14)
|
|
381
405
|
```
|
|
382
406
|
|
|
407
|
+
### Parse once, navigate many times
|
|
408
|
+
|
|
409
|
+
A parsed result is plain data, so you can persist it and reload it later. When an
|
|
410
|
+
agent asks many questions about the same document, there's no need to parse it
|
|
411
|
+
again on every question:
|
|
412
|
+
|
|
413
|
+
```python
|
|
414
|
+
from pathlib import Path
|
|
415
|
+
import docslicer
|
|
416
|
+
|
|
417
|
+
cache = Path("annual_report.json")
|
|
418
|
+
|
|
419
|
+
if cache.exists():
|
|
420
|
+
result = docslicer.ParseResult.load(cache)
|
|
421
|
+
else:
|
|
422
|
+
result = docslicer.parse_document("annual_report.pdf")
|
|
423
|
+
result.save(cache)
|
|
424
|
+
```
|
|
425
|
+
|
|
426
|
+
A reloaded result supports the full API — `hierarchy`, `find_heading`,
|
|
427
|
+
`chunks_under`, `tables` — so a long-running agent session or document server can
|
|
428
|
+
keep documents open across requests without re-parsing.
|
|
429
|
+
|
|
383
430
|
---
|
|
384
431
|
|
|
385
432
|
## Export
|
|
386
433
|
|
|
434
|
+
`save()` decides what to write from the path you give it.
|
|
435
|
+
|
|
387
436
|
```python
|
|
388
|
-
# Save
|
|
389
|
-
result.save("
|
|
390
|
-
|
|
391
|
-
# (+ charts.parquet when the document has charts)
|
|
437
|
+
# Save the whole result and reload it later — keeps the heading hierarchy
|
|
438
|
+
result.save("result.json") # same output as result.to_json()
|
|
439
|
+
result = docslicer.ParseResult.load("result.json")
|
|
392
440
|
|
|
393
|
-
#
|
|
441
|
+
# A single collection, in the format you name
|
|
394
442
|
result.save("chunks.csv")
|
|
395
443
|
result.save("charts.jsonl") # stems: chunks | blocks | tables | charts | metadata
|
|
396
|
-
result.save("result.json") # full parse result as JSON
|
|
397
444
|
result.export_chunks_jsonl("chunks.jsonl")
|
|
398
445
|
|
|
446
|
+
# One file per collection
|
|
447
|
+
result.save("output/")
|
|
448
|
+
# → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
|
|
449
|
+
# (+ charts.parquet when the document has charts)
|
|
450
|
+
# Falls back to .csv unless the [parquet] extra is installed.
|
|
451
|
+
|
|
399
452
|
# Render as Markdown or plain text
|
|
400
453
|
md = result.export_to_markdown(include_tables=True)
|
|
401
454
|
txt = result.export_to_text()
|
|
@@ -404,6 +457,9 @@ txt = result.export_to_text()
|
|
|
404
457
|
df = result.chunks_df()
|
|
405
458
|
```
|
|
406
459
|
|
|
460
|
+
Only `result.json` round-trips — the collection and directory forms write flat rows
|
|
461
|
+
without the heading hierarchy, so `ParseResult.load()` can't read them back.
|
|
462
|
+
|
|
407
463
|
### Debug mode
|
|
408
464
|
|
|
409
465
|
```python
|
|
@@ -435,6 +491,96 @@ pip install 'docslicer[ocr]'
|
|
|
435
491
|
|
|
436
492
|
---
|
|
437
493
|
|
|
494
|
+
## MCP server
|
|
495
|
+
|
|
496
|
+
DocSlicer ships an [MCP](https://modelcontextprotocol.io) server, so LLM clients
|
|
497
|
+
(Claude Desktop, Claude Code, Cursor, …) can parse and read documents directly.
|
|
498
|
+
|
|
499
|
+
```bash
|
|
500
|
+
pip install 'docslicer[mcp]'
|
|
501
|
+
docslicer-mcp # stdio — what desktop clients launch
|
|
502
|
+
docslicer-mcp --transport http --port 8000
|
|
503
|
+
```
|
|
504
|
+
|
|
505
|
+
Register it with a client by adding to its MCP config:
|
|
506
|
+
|
|
507
|
+
```jsonc
|
|
508
|
+
{
|
|
509
|
+
"mcpServers": {
|
|
510
|
+
"docslicer": {
|
|
511
|
+
"command": "docslicer-mcp",
|
|
512
|
+
"env": { "DOCSLICER_MCP_ROOT": "/Users/you/Documents" }
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
```
|
|
517
|
+
|
|
518
|
+
### How it works
|
|
519
|
+
|
|
520
|
+
A parsed document is far larger than a model's context window, so the server
|
|
521
|
+
never returns one in a single call. `parse` registers the document and hands
|
|
522
|
+
back a `doc_id` handle plus a heading outline. Every other tool takes that
|
|
523
|
+
handle and returns a bounded slice — the model pulls in only what it needs.
|
|
524
|
+
|
|
525
|
+
| Tool | Returns |
|
|
526
|
+
| --- | --- |
|
|
527
|
+
| `parse` | `doc_id` handle, title, page count, heading outline |
|
|
528
|
+
| `get_outline` | The outline again, for when it scrolls out of context |
|
|
529
|
+
| `read` | The text under one or more headings, named from the outline |
|
|
530
|
+
| `search` | Headings to `read`, ranked, each with a snippet |
|
|
531
|
+
| `to_markdown` | Writes the whole document to disk; returns the path |
|
|
532
|
+
|
|
533
|
+
Every outline line carries what reading it would cost:
|
|
534
|
+
|
|
535
|
+
```
|
|
536
|
+
- Financial statements ~48k
|
|
537
|
+
- Note 14 — Segment reporting ~900
|
|
538
|
+
- Note 15 — Income taxes ~2.1k
|
|
539
|
+
```
|
|
540
|
+
|
|
541
|
+
That figure is the same estimate `read` reports back, so a budget made from the
|
|
542
|
+
outline holds when it is spent. Sizes are cumulative — a parent never costs less
|
|
543
|
+
than the children beneath it — which is what makes "descend or just read it" a
|
|
544
|
+
decision the model can make before spending the context rather than after.
|
|
545
|
+
|
|
546
|
+
`read` takes heading text exactly as the outline prints it. Where a heading
|
|
547
|
+
appears twice, prefixing any ancestor disambiguates it (`"Notes > Revenue"`);
|
|
548
|
+
the full chain is never required. Returned text is interleaved with `[Page X]`
|
|
549
|
+
markers using the document's own page labels (`S-23`, `iv`), so a quotation can
|
|
550
|
+
be cited to the page it actually came from rather than to wherever its section
|
|
551
|
+
began.
|
|
552
|
+
|
|
553
|
+
`search` is the fallback for when the outline does not settle the question —
|
|
554
|
+
headings that name nothing useful (`Note 14`, `Item 7A`), or a figure buried in
|
|
555
|
+
a table no heading mentions. It combines a whole-word literal match with BM25
|
|
556
|
+
over the chunks, and returns *places*, not answers: each hit is a heading to
|
|
557
|
+
pass to `read`. Query terms that appear nowhere in the document are reported
|
|
558
|
+
back, so a query that scored well on one rare word can be recognised as the bad
|
|
559
|
+
query it was.
|
|
560
|
+
|
|
561
|
+
`to_markdown` is the escape hatch for when the user wants the document itself
|
|
562
|
+
rather than an answer drawn from it. It writes to disk and returns a path, so
|
|
563
|
+
nothing enters the model's context and document size stops mattering.
|
|
564
|
+
|
|
565
|
+
Parsed results are cached on disk, so re-parsing the same file with the same
|
|
566
|
+
options is free. The cache key includes the file's size and mtime — edit the
|
|
567
|
+
document and the next `parse` re-parses it automatically.
|
|
568
|
+
|
|
569
|
+
### Configuration
|
|
570
|
+
|
|
571
|
+
| Variable | Effect |
|
|
572
|
+
| --- | --- |
|
|
573
|
+
| `DOCSLICER_MCP_ROOT` | Restrict file sources **and** written output to this directory tree |
|
|
574
|
+
| `DOCSLICER_MCP_ALLOW_URLS` | Set to `0` to reject `http(s)` sources |
|
|
575
|
+
| `DOCSLICER_MCP_CACHE` | Where parsed results are persisted (default `~/.cache/docslicer-mcp`) |
|
|
576
|
+
| `DOCSLICER_MCP_CACHE_MAX_MB` | Cache size ceiling, oldest pruned first (default `2048`; `0` disables) |
|
|
577
|
+
|
|
578
|
+
Set `DOCSLICER_MCP_ROOT` when exposing the server to anything but yourself —
|
|
579
|
+
without it, any readable path on the machine is parseable, and `to_markdown`
|
|
580
|
+
can write anywhere the server process can.
|
|
581
|
+
|
|
582
|
+
---
|
|
583
|
+
|
|
438
584
|
## Format-specific functions
|
|
439
585
|
|
|
440
586
|
If you know the format upfront and want explicit failure on unexpected input, use the
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "docslicer"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "Deterministic hierarchical document parser and chunker"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "AGPL-3.0-only"
|
|
@@ -65,6 +65,12 @@ crypto = [
|
|
|
65
65
|
parquet = [
|
|
66
66
|
"pyarrow>=18.0",
|
|
67
67
|
]
|
|
68
|
+
mcp = [
|
|
69
|
+
"mcp>=2.0",
|
|
70
|
+
# The server reports token costs the model budgets against, so it counts
|
|
71
|
+
# exactly rather than estimating. Falls back to chars/4 if unusable.
|
|
72
|
+
"tiktoken>=0.10",
|
|
73
|
+
]
|
|
68
74
|
dev = [
|
|
69
75
|
"pytest>=8.0",
|
|
70
76
|
"pytest-asyncio>=0.24",
|
|
@@ -72,6 +78,7 @@ dev = [
|
|
|
72
78
|
|
|
73
79
|
[project.scripts]
|
|
74
80
|
docslicer = "docslicer.cli:main"
|
|
81
|
+
docslicer-mcp = "docslicer.mcp.server:main"
|
|
75
82
|
|
|
76
83
|
[tool.pytest.ini_options]
|
|
77
84
|
testpaths = ["tests"]
|
|
@@ -13,7 +13,8 @@ class ParseConfig:
|
|
|
13
13
|
"""User-facing configuration for a parse.
|
|
14
14
|
|
|
15
15
|
Chunking: ``max_chunk_size`` / ``optimal_chunk_size`` / ``min_chunk_size``
|
|
16
|
-
bound chunk length
|
|
16
|
+
bound chunk length in characters (always — ``exact_tokens`` changes how the
|
|
17
|
+
resulting chunks are *counted*, not where they are cut);
|
|
17
18
|
``chunking`` toggles chunking entirely and ``merge_small_chunks`` folds
|
|
18
19
|
undersized chunks into neighbours. ``table_representation`` selects how
|
|
19
20
|
tables are serialized ("markdown", "jsonl", or "melted"). ``extra_fields``
|
|
@@ -64,6 +64,7 @@ def _resolve_metadata(
|
|
|
64
64
|
source_url=source_url,
|
|
65
65
|
file_size_bytes=file_size_bytes,
|
|
66
66
|
is_password_protected=bool(discovered.get("is_password_protected", False)),
|
|
67
|
+
renderer=discovered.get("renderer"),
|
|
67
68
|
page_count=int(discovered.get("page_count") or 0),
|
|
68
69
|
page_width=discovered.get("page_width"),
|
|
69
70
|
page_height=discovered.get("page_height"),
|
|
@@ -435,8 +436,11 @@ def _build_result(
|
|
|
435
436
|
doc_token_count: int | None = None
|
|
436
437
|
doc_token_count_exact = False
|
|
437
438
|
if not df_chunks.empty and "token_count" in df_chunks.columns:
|
|
439
|
+
from .shared.step_08_chunk_builder import token_encoder
|
|
438
440
|
doc_token_count = int(df_chunks["token_count"].fillna(0).sum())
|
|
439
|
-
|
|
441
|
+
# Requesting exact counts is not getting them: tiktoken may be missing, or
|
|
442
|
+
# unable to fetch its vocabulary. Report what the counter actually did.
|
|
443
|
+
doc_token_count_exact = bool(config.exact_tokens) and token_encoder() is not None
|
|
440
444
|
|
|
441
445
|
metadata = _resolve_metadata(
|
|
442
446
|
discovered_metadata, source_url, source_filename, file_size_bytes, run_id, df_blocks,
|
|
@@ -502,24 +502,81 @@ def _collect_block_ids(node: HierarchyNode, recursive: bool) -> set[str]:
|
|
|
502
502
|
return ids
|
|
503
503
|
|
|
504
504
|
|
|
505
|
+
def _split_table_row(line: str) -> list[str]:
|
|
506
|
+
"""Split one pipe-table line into stripped cells, dropping the outer pipes."""
|
|
507
|
+
cells = line.strip().split("|")
|
|
508
|
+
if cells and cells[0].strip() == "":
|
|
509
|
+
cells = cells[1:]
|
|
510
|
+
if cells and cells[-1].strip() == "":
|
|
511
|
+
cells = cells[:-1]
|
|
512
|
+
return [c.strip() for c in cells]
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _is_sep_cell(cell: str) -> bool:
|
|
516
|
+
return bool(cell) and all(c in "-:" for c in cell)
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _is_sep_row(row: list[str]) -> bool:
|
|
520
|
+
return any(row) and all(_is_sep_cell(c) for c in row if c)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _gfm_normalize_table(markdown: str) -> str:
|
|
524
|
+
"""Rewrite a table so it renders under GitHub-flavored Markdown.
|
|
525
|
+
|
|
526
|
+
``_format_table_markdown`` is faithful to the source grid: it draws the
|
|
527
|
+
separator under the *last* header row, and omits it entirely for tables with
|
|
528
|
+
no header cells. GFM instead requires exactly one header row followed by one
|
|
529
|
+
separator, so:
|
|
530
|
+
|
|
531
|
+
* multiple header rows collapse into one, joined per column with " > "
|
|
532
|
+
(the convention ``_format_table_melted`` already uses for header paths), and
|
|
533
|
+
* headerless tables gain a blank header row, which keeps every source row
|
|
534
|
+
in the body rather than promoting one to a header the detector rejected.
|
|
535
|
+
|
|
536
|
+
Text that is not a pipe table passes through untouched.
|
|
537
|
+
"""
|
|
538
|
+
lines = markdown.strip().splitlines()
|
|
539
|
+
if not lines or any(l.strip() and not l.strip().startswith("|") for l in lines):
|
|
540
|
+
return markdown
|
|
541
|
+
|
|
542
|
+
rows = [_split_table_row(l) for l in lines if l.strip().startswith("|")]
|
|
543
|
+
if not rows:
|
|
544
|
+
return markdown
|
|
545
|
+
|
|
546
|
+
n_cols = max(len(r) for r in rows)
|
|
547
|
+
for row in rows:
|
|
548
|
+
while len(row) < n_cols:
|
|
549
|
+
row.append("")
|
|
550
|
+
|
|
551
|
+
sep = ["---"] * n_cols
|
|
552
|
+
first_sep = next((i for i, row in enumerate(rows) if _is_sep_row(row)), None)
|
|
553
|
+
|
|
554
|
+
if first_sep is None:
|
|
555
|
+
rows = [[""] * n_cols, sep] + rows
|
|
556
|
+
elif first_sep == 0:
|
|
557
|
+
rows = [[""] * n_cols] + rows
|
|
558
|
+
elif first_sep > 1:
|
|
559
|
+
header = []
|
|
560
|
+
for col in range(n_cols):
|
|
561
|
+
parts: list[str] = []
|
|
562
|
+
for row in rows[:first_sep]:
|
|
563
|
+
# A rowspan header cell repeats down the header rows — "Region >
|
|
564
|
+
# Region" carries no more than "Region" does.
|
|
565
|
+
if row[col] and (not parts or parts[-1] != row[col]):
|
|
566
|
+
parts.append(row[col])
|
|
567
|
+
header.append(" > ".join(parts))
|
|
568
|
+
rows = [header, sep] + rows[first_sep + 1:]
|
|
569
|
+
|
|
570
|
+
return "\n".join("| " + " | ".join(row) + " |" for row in rows)
|
|
571
|
+
|
|
572
|
+
|
|
505
573
|
def _prettify_table(markdown: str) -> str:
|
|
506
574
|
"""Reformat a markdown table so pipe characters are vertically aligned."""
|
|
507
575
|
lines = markdown.strip().splitlines()
|
|
508
576
|
if not lines:
|
|
509
577
|
return markdown
|
|
510
578
|
|
|
511
|
-
|
|
512
|
-
cells = line.strip().split("|")
|
|
513
|
-
if cells and cells[0].strip() == "":
|
|
514
|
-
cells = cells[1:]
|
|
515
|
-
if cells and cells[-1].strip() == "":
|
|
516
|
-
cells = cells[:-1]
|
|
517
|
-
return [c.strip() for c in cells]
|
|
518
|
-
|
|
519
|
-
def _is_sep(cell: str) -> bool:
|
|
520
|
-
return bool(cell) and all(c in "-:" for c in cell)
|
|
521
|
-
|
|
522
|
-
rows = [_split_row(l) for l in lines if l.strip().startswith("|")]
|
|
579
|
+
rows = [_split_table_row(l) for l in lines if l.strip().startswith("|")]
|
|
523
580
|
if not rows:
|
|
524
581
|
return markdown
|
|
525
582
|
|
|
@@ -528,7 +585,7 @@ def _prettify_table(markdown: str) -> str:
|
|
|
528
585
|
while len(row) < n_cols:
|
|
529
586
|
row.append("")
|
|
530
587
|
|
|
531
|
-
sep_indices = {i for i, row in enumerate(rows) if
|
|
588
|
+
sep_indices = {i for i, row in enumerate(rows) if _is_sep_row(row)}
|
|
532
589
|
col_widths = [3] * n_cols
|
|
533
590
|
for i, row in enumerate(rows):
|
|
534
591
|
if i not in sep_indices:
|
|
@@ -664,8 +721,16 @@ class ParseResult:
|
|
|
664
721
|
include_toc: bool = True,
|
|
665
722
|
include_furniture: bool = True,
|
|
666
723
|
prettify: bool = True,
|
|
724
|
+
gfm_tables: bool = True,
|
|
667
725
|
) -> str:
|
|
668
|
-
"""Render the document as Markdown using blocks as the source of truth.
|
|
726
|
+
"""Render the document as Markdown using blocks as the source of truth.
|
|
727
|
+
|
|
728
|
+
``gfm_tables`` rewrites tables to the one-header-row-plus-separator shape
|
|
729
|
+
GitHub-flavored Markdown renderers require. Pass ``False`` for the faithful
|
|
730
|
+
grid — multiple header rows, and no separator for tables that genuinely
|
|
731
|
+
have no header — which is what ``Table.markdown`` and the block text always
|
|
732
|
+
carry regardless of this flag.
|
|
733
|
+
"""
|
|
669
734
|
_HEADING_ROLES = {
|
|
670
735
|
"heading", "toc_heading", "exhibit_heading", "hybrid_heading_paragraph",
|
|
671
736
|
}
|
|
@@ -704,6 +769,8 @@ class ParseResult:
|
|
|
704
769
|
if include_tables:
|
|
705
770
|
table = tables_by_id.get(block.table_ids[0]) if block.table_ids else None
|
|
706
771
|
raw = table.markdown if table else text
|
|
772
|
+
if gfm_tables:
|
|
773
|
+
raw = _gfm_normalize_table(raw)
|
|
707
774
|
parts.append(_prettify_table(raw) if prettify else raw)
|
|
708
775
|
elif block.type == "chart":
|
|
709
776
|
if text:
|