docslicer 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. {docslicer-0.2.0/src/docslicer.egg-info → docslicer-0.2.1}/PKG-INFO +159 -10
  2. {docslicer-0.2.0 → docslicer-0.2.1}/README.md +155 -9
  3. {docslicer-0.2.0 → docslicer-0.2.1}/pyproject.toml +8 -1
  4. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_config.py +2 -1
  5. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_orchestrator.py +5 -1
  6. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_result.py +81 -14
  7. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/html_orchestrator.py +46 -17
  8. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_01_box_extractor.py +25 -6
  9. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_01_static_box_extractor.py +219 -110
  10. docslicer-0.2.1/src/docslicer/mcp/__init__.py +16 -0
  11. docslicer-0.2.1/src/docslicer/mcp/_store.py +336 -0
  12. docslicer-0.2.1/src/docslicer/mcp/server.py +1062 -0
  13. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/schema.py +4 -0
  14. docslicer-0.2.1/src/docslicer/ocr/__init__.py +9 -0
  15. docslicer-0.2.1/src/docslicer/ocr/_availability.py +61 -0
  16. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/pdf_orchestrator.py +34 -0
  17. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_05_heading_detector.py +21 -10
  18. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_07_block_merger.py +36 -5
  19. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_08_chunk_builder.py +188 -35
  20. {docslicer-0.2.0 → docslicer-0.2.1/src/docslicer.egg-info}/PKG-INFO +159 -10
  21. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/SOURCES.txt +6 -1
  22. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/entry_points.txt +1 -0
  23. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/requires.txt +4 -0
  24. docslicer-0.2.1/tests/test_table_cell_newlines.py +92 -0
  25. docslicer-0.2.0/src/docslicer/ocr/__init__.py +0 -3
  26. {docslicer-0.2.0 → docslicer-0.2.1}/LICENSE +0 -0
  27. {docslicer-0.2.0 → docslicer-0.2.1}/LICENSE-COMMERCIAL.md +0 -0
  28. {docslicer-0.2.0 → docslicer-0.2.1}/setup.cfg +0 -0
  29. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/__init__.py +0 -0
  30. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/__init__.py +0 -0
  31. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/color_utils.py +0 -0
  32. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/cpu.py +0 -0
  33. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/__init__.py +0 -0
  34. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/registry_aggregator.py +0 -0
  35. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_aggregation/text_merge.py +0 -0
  36. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/__init__.py +0 -0
  37. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/export_debug.py +0 -0
  38. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/df_export/reorder_columns.py +0 -0
  39. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/__init__.py +0 -0
  40. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/__init__.py +0 -0
  41. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/exhibit_patterns.py +0 -0
  42. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/hierarchy_type_patterns.py +0 -0
  43. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_compilers/page_label_patterns.py +0 -0
  44. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/io/yaml_loader.py +0 -0
  45. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/__init__.py +0 -0
  46. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/gutter_detector.py +0 -0
  47. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/layouts.py +0 -0
  48. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/line_merger.py +0 -0
  49. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/line_number_detector.py +0 -0
  50. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/reading_order.py +0 -0
  51. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/layout/shape_processor.py +0 -0
  52. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/oxm_package.py +0 -0
  53. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/parallel.py +0 -0
  54. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/password.py +0 -0
  55. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/safe_call.py +0 -0
  56. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/__init__.py +0 -0
  57. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_header.py +0 -0
  58. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_normalize.py +0 -0
  59. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/table/table_schema.py +0 -0
  60. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/text_utils.py +0 -0
  61. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/_utils/timing.py +0 -0
  62. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/cli.py +0 -0
  63. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/common_author_names.csv +0 -0
  64. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/exhibit_patterns.yaml +0 -0
  65. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/hierarchy_type_patterns.yaml +0 -0
  66. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/config/page_label_patterns.yaml +0 -0
  67. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/__init__.py +0 -0
  68. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/docx_orchestrator.py +0 -0
  69. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/native_metadata.py +0 -0
  70. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_01_package_reader.py +0 -0
  71. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_02_run_extractor.py +0 -0
  72. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_03_chart_point_builder.py +0 -0
  73. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_04_table_cell_builder.py +0 -0
  74. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_05_paragraph_builder.py +0 -0
  75. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_06_line_builder.py +0 -0
  76. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/docx/step_07_style_prefiller.py +0 -0
  77. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/__init__.py +0 -0
  78. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/extract_boxes.js +0 -0
  79. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/native_metadata.py +0 -0
  80. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_02_box_cleaner.py +0 -0
  81. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_03_page_label_detector.py +0 -0
  82. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_04_line_builder.py +0 -0
  83. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_05_table_extractor.py +0 -0
  84. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/html/step_06_style_prefiller.py +0 -0
  85. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/__init__.py +0 -0
  86. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/consolidate.py +0 -0
  87. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/generator.py +0 -0
  88. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/ocr_detector.py +0 -0
  89. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/page_analysis.py +0 -0
  90. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/metadata/text_fallback.py +0 -0
  91. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/ocr_orchestrator.py +0 -0
  92. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_01_word_extractor.py +0 -0
  93. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_02_word_colorizer.py +0 -0
  94. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_03_shape_extractor.py +0 -0
  95. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_04_text_cleaner.py +0 -0
  96. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/ocr/step_05_font_size_estimator.py +0 -0
  97. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/__init__.py +0 -0
  98. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/__init__.py +0 -0
  99. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/coordinates.py +0 -0
  100. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/form_fields.py +0 -0
  101. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/form_label_link.py +0 -0
  102. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/line_classification.py +0 -0
  103. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/page_rotation.py +0 -0
  104. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/script_thresholds.py +0 -0
  105. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/struct_context.py +0 -0
  106. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/_utils/struct_tree.py +0 -0
  107. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/native_metadata.py +0 -0
  108. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_01_word_extractor.py +0 -0
  109. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_02_image_extractor.py +0 -0
  110. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_03_shape_extractor.py +0 -0
  111. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_04_link_extractor.py +0 -0
  112. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_05_struct_group.py +0 -0
  113. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_06_style_prefiller.py +0 -0
  114. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_07_stream_group.py +0 -0
  115. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_08_reading_order.py +0 -0
  116. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_09_word_relationships.py +0 -0
  117. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_10_cell_builder.py +0 -0
  118. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_11_page_label_detector.py +0 -0
  119. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_12_cell_grouper.py +0 -0
  120. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_13_line_builder.py +0 -0
  121. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pdf/step_14_table_builder.py +0 -0
  122. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/__init__.py +0 -0
  123. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/native_metadata.py +0 -0
  124. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/pptx_orchestrator.py +0 -0
  125. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_01_package_reader.py +0 -0
  126. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_02_run_extractor.py +0 -0
  127. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_03_chart_point_builder.py +0 -0
  128. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_04_table_cell_builder.py +0 -0
  129. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_05_paragraph_builder.py +0 -0
  130. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_06_reading_order.py +0 -0
  131. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_07_line_builder.py +0 -0
  132. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/pptx/step_08_style_prefiller.py +0 -0
  133. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/__init__.py +0 -0
  134. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/config.py +0 -0
  135. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/cookie_consent.js +0 -0
  136. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/dispatcher.py +0 -0
  137. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/__init__.py +0 -0
  138. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/http_fetcher.py +0 -0
  139. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/fetchers/sec_fetcher.py +0 -0
  140. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/models.py +0 -0
  141. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/scraping/stealth_init.js +0 -0
  142. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/__init__.py +0 -0
  143. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/shared_orchestrator.py +0 -0
  144. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_01_navigation_detector.py +0 -0
  145. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_02_toc_detector.py +0 -0
  146. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_03_exhibit_detector.py +0 -0
  147. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_04_section_classifier.py +0 -0
  148. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer/shared/step_06_hierarchy_builder.py +0 -0
  149. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/dependency_links.txt +0 -0
  150. {docslicer-0.2.0 → docslicer-0.2.1}/src/docslicer.egg-info/top_level.txt +0 -0
  151. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_api.py +0 -0
  152. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_charts.py +0 -0
  153. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_document_parser.py +0 -0
  154. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_errors.py +0 -0
  155. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_exports.py +0 -0
  156. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_loose_box_reconstruction.py +0 -0
  157. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_packaging.py +0 -0
  158. {docslicer-0.2.0 → docslicer-0.2.1}/tests/test_smoke.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: docslicer
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Deterministic hierarchical document parser and chunker
5
5
  Author-email: "Market Framer Inc." <jelle@docslicer.ai>
6
6
  License-Expression: AGPL-3.0-only
@@ -50,6 +50,9 @@ Provides-Extra: crypto
50
50
  Requires-Dist: msoffcrypto-tool>=5.0; extra == "crypto"
51
51
  Provides-Extra: parquet
52
52
  Requires-Dist: pyarrow>=18.0; extra == "parquet"
53
+ Provides-Extra: mcp
54
+ Requires-Dist: mcp>=2.0; extra == "mcp"
55
+ Requires-Dist: tiktoken>=0.10; extra == "mcp"
53
56
  Provides-Extra: dev
54
57
  Requires-Dist: pytest>=8.0; extra == "dev"
55
58
  Requires-Dist: pytest-asyncio>=0.24; extra == "dev"
@@ -57,11 +60,17 @@ Dynamic: license-file
57
60
 
58
61
  # DocSlicer
59
62
 
60
- [![License: AGPL v3](https://img.shields.io/badge/License-AGPL_v3-blue.svg)](LICENSE) [![Commercial license available](https://img.shields.io/badge/License-Commercial-green.svg)](LICENSE-COMMERCIAL.md)
63
+ [![PyPI](https://img.shields.io/pypi/v/docslicer.svg)](https://pypi.org/project/docslicer/) [![License: AGPL v3](https://img.shields.io/badge/License-AGPL_v3-blue.svg)](LICENSE) [![Commercial license available](https://img.shields.io/badge/License-Commercial-green.svg)](LICENSE-COMMERCIAL.md)
61
64
 
62
- Lightning-fast, deterministic hierarchical document parser and chunker for business documents. No LLM calls or heavy ML models.
65
+ Lightning-fast (31 pages/sec), deterministic document parser and chunker for business documents. No LLM calls or heavy ML models.
63
66
 
64
- DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, and a navigable heading hierarchy — preserving the document's own structure instead of guessing at it.
67
+ DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, markdown and a navigable heading hierarchy.
68
+
69
+ Top score on [BizDocBench](https://github.com/DocSlicer/BizDocBench) (0.88 overall vs 0.70 for the next-best tool). 0.80 table accuracy, 0.98 content faithfulness, 0.85 heading recognition and hierarchy preservation, and 0.76 RAG retrieval performance.
70
+
71
+ **Add DocSlicer to your AI pipeline:**
72
+ - Classic RAG: the layout-aware chunker gives you clean non-overlapping chunks, each carrying its full heading breadcrumb, ready to embed.
73
+ - Vectorless RAG: for when you want an answer out of a document right now. The agent pulls the outline, picks the section it needs, and navigates to the correct section of text without embedding the whole document
65
74
 
66
75
  ```python
67
76
  import docslicer
@@ -117,6 +126,23 @@ if __name__ == "__main__":
117
126
 
118
127
  ---
119
128
 
129
+ ## Benchmarks
130
+
131
+ Measured with [BizDocBench](https://github.com/DocSlicer/BizDocBench) — an open benchmark for multi-format business document parsing. All scores are 0–1 (higher is better); `pages_per_sec_aggregate` is throughput across the full corpus.
132
+
133
+ | Tool | Score | Coverage | Speed | Hierarchy | Faithfulness | Tables | Retrieval | Pages/sec |
134
+ |---|---|---|---|---|---|---|---|---|
135
+ | **docslicer** | **0.8796** | 1.0000 | 0.8836 | 0.8466 | 0.9824 | 0.8047 | 0.7601 | 31.27 |
136
+ | docling | 0.7036 | 1.0000 | 0.3805 | 0.4905 | 0.8927 | 0.7467 | 0.7111 | 3.46 |
137
+ | markitdown | 0.5838 | 1.0000 | 0.8513 | 0.0604 | 0.7972 | 0.2584 | 0.5357 | 27.42 |
138
+ | unstructured | 0.5798 | 0.9091 | 0.1073 | 0.4327 | 0.9057 | 0.4812 | 0.6430 | 0.52 |
139
+ | opendataloader | 0.5359 | 0.5844 | 1.0000 | 0.3853 | 0.6484 | 0.2655 | 0.3317 | 117.26 |
140
+ | pymupdf4llm | 0.4519 | 0.5974 | 0.6492 | 0.1089 | 0.6456 | 0.3551 | 0.3552 | 11.84 |
141
+ | mineru | 0.4107 | 0.5974 | 0.1353 | 0.4220 | 0.6176 | 0.3012 | 0.3910 | 0.70 |
142
+ | marker | 0.3735 | 0.5974 | 0.1598 | 0.1926 | 0.6121 | 0.3012 | 0.3778 | 0.87 |
143
+
144
+ ---
145
+
120
146
  ## Install
121
147
 
122
148
  ```bash
@@ -135,6 +161,7 @@ pip install 'docslicer[ocr]' # scanned PDF support via Tesseract + OpenCV
135
161
  # Linux: apt install tesseract-ocr
136
162
  # macOS: brew install tesseract
137
163
 
164
+ pip install 'docslicer[mcp]' # MCP server for LLM clients (Claude, Cursor, …)
138
165
  pip install 'docslicer[llm]' # exact token counts via tiktoken (exact_tokens=True)
139
166
  pip install 'docslicer[crypto]' # password-protected Office files (msoffcrypto-tool)
140
167
  pip install 'docslicer[parquet]' # Parquet export support
@@ -437,22 +464,51 @@ result.tables_by_page(14)
437
464
  result.charts_by_page(14)
438
465
  ```
439
466
 
467
+ ### Parse once, navigate many times
468
+
469
+ A parsed result is plain data, so you can persist it and reload it later. When an
470
+ agent asks many questions about the same document, there's no need to parse it
471
+ again on every question:
472
+
473
+ ```python
474
+ from pathlib import Path
475
+ import docslicer
476
+
477
+ cache = Path("annual_report.json")
478
+
479
+ if cache.exists():
480
+ result = docslicer.ParseResult.load(cache)
481
+ else:
482
+ result = docslicer.parse_document("annual_report.pdf")
483
+ result.save(cache)
484
+ ```
485
+
486
+ A reloaded result supports the full API — `hierarchy`, `find_heading`,
487
+ `chunks_under`, `tables` — so a long-running agent session or document server can
488
+ keep documents open across requests without re-parsing.
489
+
440
490
  ---
441
491
 
442
492
  ## Export
443
493
 
494
+ `save()` decides what to write from the path you give it.
495
+
444
496
  ```python
445
- # Save everything
446
- result.save("output/")
447
- # → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
448
- # (+ charts.parquet when the document has charts)
497
+ # Save the whole result and reload it later — keeps the heading hierarchy
498
+ result.save("result.json") # same output as result.to_json()
499
+ result = docslicer.ParseResult.load("result.json")
449
500
 
450
- # Specific formats
501
+ # A single collection, in the format you name
451
502
  result.save("chunks.csv")
452
503
  result.save("charts.jsonl") # stems: chunks | blocks | tables | charts | metadata
453
- result.save("result.json") # full parse result as JSON
454
504
  result.export_chunks_jsonl("chunks.jsonl")
455
505
 
506
+ # One file per collection
507
+ result.save("output/")
508
+ # → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
509
+ # (+ charts.parquet when the document has charts)
510
+ # Falls back to .csv unless the [parquet] extra is installed.
511
+
456
512
  # Render as Markdown or plain text
457
513
  md = result.export_to_markdown(include_tables=True)
458
514
  txt = result.export_to_text()
@@ -461,6 +517,9 @@ txt = result.export_to_text()
461
517
  df = result.chunks_df()
462
518
  ```
463
519
 
520
+ Only `result.json` round-trips — the collection and directory forms write flat rows
521
+ without the heading hierarchy, so `ParseResult.load()` can't read them back.
522
+
464
523
  ### Debug mode
465
524
 
466
525
  ```python
@@ -492,6 +551,96 @@ pip install 'docslicer[ocr]'
492
551
 
493
552
  ---
494
553
 
554
+ ## MCP server
555
+
556
+ DocSlicer ships an [MCP](https://modelcontextprotocol.io) server, so LLM clients
557
+ (Claude Desktop, Claude Code, Cursor, …) can parse and read documents directly.
558
+
559
+ ```bash
560
+ pip install 'docslicer[mcp]'
561
+ docslicer-mcp # stdio — what desktop clients launch
562
+ docslicer-mcp --transport http --port 8000
563
+ ```
564
+
565
+ Register it with a client by adding to its MCP config:
566
+
567
+ ```jsonc
568
+ {
569
+ "mcpServers": {
570
+ "docslicer": {
571
+ "command": "docslicer-mcp",
572
+ "env": { "DOCSLICER_MCP_ROOT": "/Users/you/Documents" }
573
+ }
574
+ }
575
+ }
576
+ ```
577
+
578
+ ### How it works
579
+
580
+ A parsed document is far larger than a model's context window, so the server
581
+ never returns one in a single call. `parse` registers the document and hands
582
+ back a `doc_id` handle plus a heading outline. Every other tool takes that
583
+ handle and returns a bounded slice — the model pulls in only what it needs.
584
+
585
+ | Tool | Returns |
586
+ | --- | --- |
587
+ | `parse` | `doc_id` handle, title, page count, heading outline |
588
+ | `get_outline` | The outline again, for when it scrolls out of context |
589
+ | `read` | The text under one or more headings, named from the outline |
590
+ | `search` | Headings to `read`, ranked, each with a snippet |
591
+ | `to_markdown` | Writes the whole document to disk; returns the path |
592
+
593
+ Every outline line carries what reading it would cost:
594
+
595
+ ```
596
+ - Financial statements ~48k
597
+ - Note 14 — Segment reporting ~900
598
+ - Note 15 — Income taxes ~2.1k
599
+ ```
600
+
601
+ That figure is the same estimate `read` reports back, so a budget made from the
602
+ outline holds when it is spent. Sizes are cumulative — a parent never costs less
603
+ than the children beneath it — which is what makes "descend or just read it" a
604
+ decision the model can make before spending the context rather than after.
605
+
606
+ `read` takes heading text exactly as the outline prints it. Where a heading
607
+ appears twice, prefixing any ancestor disambiguates it (`"Notes > Revenue"`);
608
+ the full chain is never required. Returned text is interleaved with `[Page X]`
609
+ markers using the document's own page labels (`S-23`, `iv`), so a quotation can
610
+ be cited to the page it actually came from rather than to wherever its section
611
+ began.
612
+
613
+ `search` is the fallback for when the outline does not settle the question —
614
+ headings that name nothing useful (`Note 14`, `Item 7A`), or a figure buried in
615
+ a table no heading mentions. It combines a whole-word literal match with BM25
616
+ over the chunks, and returns *places*, not answers: each hit is a heading to
617
+ pass to `read`. Query terms that appear nowhere in the document are reported
618
+ back, so a query that scored well on one rare word can be recognised as the bad
619
+ query it was.
620
+
621
+ `to_markdown` is the escape hatch for when the user wants the document itself
622
+ rather than an answer drawn from it. It writes to disk and returns a path, so
623
+ nothing enters the model's context and document size stops mattering.
624
+
625
+ Parsed results are cached on disk, so re-parsing the same file with the same
626
+ options is free. The cache key includes the file's size and mtime — edit the
627
+ document and the next `parse` re-parses it automatically.
628
+
629
+ ### Configuration
630
+
631
+ | Variable | Effect |
632
+ | --- | --- |
633
+ | `DOCSLICER_MCP_ROOT` | Restrict file sources **and** written output to this directory tree |
634
+ | `DOCSLICER_MCP_ALLOW_URLS` | Set to `0` to reject `http(s)` sources |
635
+ | `DOCSLICER_MCP_CACHE` | Where parsed results are persisted (default `~/.cache/docslicer-mcp`) |
636
+ | `DOCSLICER_MCP_CACHE_MAX_MB` | Cache size ceiling, oldest pruned first (default `2048`; `0` disables) |
637
+
638
+ Set `DOCSLICER_MCP_ROOT` when exposing the server to anything but yourself —
639
+ without it, any readable path on the machine is parseable, and `to_markdown`
640
+ can write anywhere the server process can.
641
+
642
+ ---
643
+
495
644
  ## Format-specific functions
496
645
 
497
646
  If you know the format upfront and want explicit failure on unexpected input, use the
@@ -1,10 +1,16 @@
1
1
  # DocSlicer
2
2
 
3
- [![License: AGPL v3](https://img.shields.io/badge/License-AGPL_v3-blue.svg)](LICENSE) [![Commercial license available](https://img.shields.io/badge/License-Commercial-green.svg)](LICENSE-COMMERCIAL.md)
3
+ [![PyPI](https://img.shields.io/pypi/v/docslicer.svg)](https://pypi.org/project/docslicer/) [![License: AGPL v3](https://img.shields.io/badge/License-AGPL_v3-blue.svg)](LICENSE) [![Commercial license available](https://img.shields.io/badge/License-Commercial-green.svg)](LICENSE-COMMERCIAL.md)
4
4
 
5
- Lightning-fast, deterministic hierarchical document parser and chunker for business documents. No LLM calls or heavy ML models.
5
+ Lightning-fast (31 pages/sec), deterministic document parser and chunker for business documents. No LLM calls or heavy ML models.
6
6
 
7
- DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, and a navigable heading hierarchy — preserving the document's own structure instead of guessing at it.
7
+ DocSlicer turns PDFs, Word documents, HTML pages, and PowerPoint files into clean chunks, structured blocks, tables, charts, markdown and a navigable heading hierarchy.
8
+
9
+ Top score on [BizDocBench](https://github.com/DocSlicer/BizDocBench) (0.88 overall vs 0.70 for the next-best tool). 0.80 table accuracy, 0.98 content faithfulness, 0.85 heading recognition and hierarchy preservation, and 0.76 RAG retrieval performance.
10
+
11
+ **Add DocSlicer to your AI pipeline:**
12
+ - Classic RAG: the layout-aware chunker gives you clean non-overlapping chunks, each carrying its full heading breadcrumb, ready to embed.
13
+ - Vectorless RAG: for when you want an answer out of a document right now. The agent pulls the outline, picks the section it needs, and navigates to the correct section of text without embedding the whole document
8
14
 
9
15
  ```python
10
16
  import docslicer
@@ -60,6 +66,23 @@ if __name__ == "__main__":
60
66
 
61
67
  ---
62
68
 
69
+ ## Benchmarks
70
+
71
+ Measured with [BizDocBench](https://github.com/DocSlicer/BizDocBench) — an open benchmark for multi-format business document parsing. All scores are 0–1 (higher is better); `pages_per_sec_aggregate` is throughput across the full corpus.
72
+
73
+ | Tool | Score | Coverage | Speed | Hierarchy | Faithfulness | Tables | Retrieval | Pages/sec |
74
+ |---|---|---|---|---|---|---|---|---|
75
+ | **docslicer** | **0.8796** | 1.0000 | 0.8836 | 0.8466 | 0.9824 | 0.8047 | 0.7601 | 31.27 |
76
+ | docling | 0.7036 | 1.0000 | 0.3805 | 0.4905 | 0.8927 | 0.7467 | 0.7111 | 3.46 |
77
+ | markitdown | 0.5838 | 1.0000 | 0.8513 | 0.0604 | 0.7972 | 0.2584 | 0.5357 | 27.42 |
78
+ | unstructured | 0.5798 | 0.9091 | 0.1073 | 0.4327 | 0.9057 | 0.4812 | 0.6430 | 0.52 |
79
+ | opendataloader | 0.5359 | 0.5844 | 1.0000 | 0.3853 | 0.6484 | 0.2655 | 0.3317 | 117.26 |
80
+ | pymupdf4llm | 0.4519 | 0.5974 | 0.6492 | 0.1089 | 0.6456 | 0.3551 | 0.3552 | 11.84 |
81
+ | mineru | 0.4107 | 0.5974 | 0.1353 | 0.4220 | 0.6176 | 0.3012 | 0.3910 | 0.70 |
82
+ | marker | 0.3735 | 0.5974 | 0.1598 | 0.1926 | 0.6121 | 0.3012 | 0.3778 | 0.87 |
83
+
84
+ ---
85
+
63
86
  ## Install
64
87
 
65
88
  ```bash
@@ -78,6 +101,7 @@ pip install 'docslicer[ocr]' # scanned PDF support via Tesseract + OpenCV
78
101
  # Linux: apt install tesseract-ocr
79
102
  # macOS: brew install tesseract
80
103
 
104
+ pip install 'docslicer[mcp]' # MCP server for LLM clients (Claude, Cursor, …)
81
105
  pip install 'docslicer[llm]' # exact token counts via tiktoken (exact_tokens=True)
82
106
  pip install 'docslicer[crypto]' # password-protected Office files (msoffcrypto-tool)
83
107
  pip install 'docslicer[parquet]' # Parquet export support
@@ -380,22 +404,51 @@ result.tables_by_page(14)
380
404
  result.charts_by_page(14)
381
405
  ```
382
406
 
407
+ ### Parse once, navigate many times
408
+
409
+ A parsed result is plain data, so you can persist it and reload it later. When an
410
+ agent asks many questions about the same document, there's no need to parse it
411
+ again on every question:
412
+
413
+ ```python
414
+ from pathlib import Path
415
+ import docslicer
416
+
417
+ cache = Path("annual_report.json")
418
+
419
+ if cache.exists():
420
+ result = docslicer.ParseResult.load(cache)
421
+ else:
422
+ result = docslicer.parse_document("annual_report.pdf")
423
+ result.save(cache)
424
+ ```
425
+
426
+ A reloaded result supports the full API — `hierarchy`, `find_heading`,
427
+ `chunks_under`, `tables` — so a long-running agent session or document server can
428
+ keep documents open across requests without re-parsing.
429
+
383
430
  ---
384
431
 
385
432
  ## Export
386
433
 
434
+ `save()` decides what to write from the path you give it.
435
+
387
436
  ```python
388
- # Save everything
389
- result.save("output/")
390
- # → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
391
- # (+ charts.parquet when the document has charts)
437
+ # Save the whole result and reload it later — keeps the heading hierarchy
438
+ result.save("result.json") # same output as result.to_json()
439
+ result = docslicer.ParseResult.load("result.json")
392
440
 
393
- # Specific formats
441
+ # A single collection, in the format you name
394
442
  result.save("chunks.csv")
395
443
  result.save("charts.jsonl") # stems: chunks | blocks | tables | charts | metadata
396
- result.save("result.json") # full parse result as JSON
397
444
  result.export_chunks_jsonl("chunks.jsonl")
398
445
 
446
+ # One file per collection
447
+ result.save("output/")
448
+ # → output/chunks.parquet, blocks.parquet, tables.parquet, metadata.json
449
+ # (+ charts.parquet when the document has charts)
450
+ # Falls back to .csv unless the [parquet] extra is installed.
451
+
399
452
  # Render as Markdown or plain text
400
453
  md = result.export_to_markdown(include_tables=True)
401
454
  txt = result.export_to_text()
@@ -404,6 +457,9 @@ txt = result.export_to_text()
404
457
  df = result.chunks_df()
405
458
  ```
406
459
 
460
+ Only `result.json` round-trips — the collection and directory forms write flat rows
461
+ without the heading hierarchy, so `ParseResult.load()` can't read them back.
462
+
407
463
  ### Debug mode
408
464
 
409
465
  ```python
@@ -435,6 +491,96 @@ pip install 'docslicer[ocr]'
435
491
 
436
492
  ---
437
493
 
494
+ ## MCP server
495
+
496
+ DocSlicer ships an [MCP](https://modelcontextprotocol.io) server, so LLM clients
497
+ (Claude Desktop, Claude Code, Cursor, …) can parse and read documents directly.
498
+
499
+ ```bash
500
+ pip install 'docslicer[mcp]'
501
+ docslicer-mcp # stdio — what desktop clients launch
502
+ docslicer-mcp --transport http --port 8000
503
+ ```
504
+
505
+ Register it with a client by adding to its MCP config:
506
+
507
+ ```jsonc
508
+ {
509
+ "mcpServers": {
510
+ "docslicer": {
511
+ "command": "docslicer-mcp",
512
+ "env": { "DOCSLICER_MCP_ROOT": "/Users/you/Documents" }
513
+ }
514
+ }
515
+ }
516
+ ```
517
+
518
+ ### How it works
519
+
520
+ A parsed document is far larger than a model's context window, so the server
521
+ never returns one in a single call. `parse` registers the document and hands
522
+ back a `doc_id` handle plus a heading outline. Every other tool takes that
523
+ handle and returns a bounded slice — the model pulls in only what it needs.
524
+
525
+ | Tool | Returns |
526
+ | --- | --- |
527
+ | `parse` | `doc_id` handle, title, page count, heading outline |
528
+ | `get_outline` | The outline again, for when it scrolls out of context |
529
+ | `read` | The text under one or more headings, named from the outline |
530
+ | `search` | Headings to `read`, ranked, each with a snippet |
531
+ | `to_markdown` | Writes the whole document to disk; returns the path |
532
+
533
+ Every outline line carries what reading it would cost:
534
+
535
+ ```
536
+ - Financial statements ~48k
537
+ - Note 14 — Segment reporting ~900
538
+ - Note 15 — Income taxes ~2.1k
539
+ ```
540
+
541
+ That figure is the same estimate `read` reports back, so a budget made from the
542
+ outline holds when it is spent. Sizes are cumulative — a parent never costs less
543
+ than the children beneath it — which is what makes "descend or just read it" a
544
+ decision the model can make before spending the context rather than after.
545
+
546
+ `read` takes heading text exactly as the outline prints it. Where a heading
547
+ appears twice, prefixing any ancestor disambiguates it (`"Notes > Revenue"`);
548
+ the full chain is never required. Returned text is interleaved with `[Page X]`
549
+ markers using the document's own page labels (`S-23`, `iv`), so a quotation can
550
+ be cited to the page it actually came from rather than to wherever its section
551
+ began.
552
+
553
+ `search` is the fallback for when the outline does not settle the question —
554
+ headings that name nothing useful (`Note 14`, `Item 7A`), or a figure buried in
555
+ a table no heading mentions. It combines a whole-word literal match with BM25
556
+ over the chunks, and returns *places*, not answers: each hit is a heading to
557
+ pass to `read`. Query terms that appear nowhere in the document are reported
558
+ back, so a query that scored well on one rare word can be recognised as the bad
559
+ query it was.
560
+
561
+ `to_markdown` is the escape hatch for when the user wants the document itself
562
+ rather than an answer drawn from it. It writes to disk and returns a path, so
563
+ nothing enters the model's context and document size stops mattering.
564
+
565
+ Parsed results are cached on disk, so re-parsing the same file with the same
566
+ options is free. The cache key includes the file's size and mtime — edit the
567
+ document and the next `parse` re-parses it automatically.
568
+
569
+ ### Configuration
570
+
571
+ | Variable | Effect |
572
+ | --- | --- |
573
+ | `DOCSLICER_MCP_ROOT` | Restrict file sources **and** written output to this directory tree |
574
+ | `DOCSLICER_MCP_ALLOW_URLS` | Set to `0` to reject `http(s)` sources |
575
+ | `DOCSLICER_MCP_CACHE` | Where parsed results are persisted (default `~/.cache/docslicer-mcp`) |
576
+ | `DOCSLICER_MCP_CACHE_MAX_MB` | Cache size ceiling, oldest pruned first (default `2048`; `0` disables) |
577
+
578
+ Set `DOCSLICER_MCP_ROOT` when exposing the server to anything but yourself —
579
+ without it, any readable path on the machine is parseable, and `to_markdown`
580
+ can write anywhere the server process can.
581
+
582
+ ---
583
+
438
584
  ## Format-specific functions
439
585
 
440
586
  If you know the format upfront and want explicit failure on unexpected input, use the
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "docslicer"
7
- version = "0.2.0"
7
+ version = "0.2.1"
8
8
  description = "Deterministic hierarchical document parser and chunker"
9
9
  readme = "README.md"
10
10
  license = "AGPL-3.0-only"
@@ -65,6 +65,12 @@ crypto = [
65
65
  parquet = [
66
66
  "pyarrow>=18.0",
67
67
  ]
68
+ mcp = [
69
+ "mcp>=2.0",
70
+ # The server reports token costs the model budgets against, so it counts
71
+ # exactly rather than estimating. Falls back to chars/4 if unusable.
72
+ "tiktoken>=0.10",
73
+ ]
68
74
  dev = [
69
75
  "pytest>=8.0",
70
76
  "pytest-asyncio>=0.24",
@@ -72,6 +78,7 @@ dev = [
72
78
 
73
79
  [project.scripts]
74
80
  docslicer = "docslicer.cli:main"
81
+ docslicer-mcp = "docslicer.mcp.server:main"
75
82
 
76
83
  [tool.pytest.ini_options]
77
84
  testpaths = ["tests"]
@@ -13,7 +13,8 @@ class ParseConfig:
13
13
  """User-facing configuration for a parse.
14
14
 
15
15
  Chunking: ``max_chunk_size`` / ``optimal_chunk_size`` / ``min_chunk_size``
16
- bound chunk length (in characters, or tokens when ``exact_tokens=True``);
16
+ bound chunk length in characters (always — ``exact_tokens`` changes how the
17
+ resulting chunks are *counted*, not where they are cut);
17
18
  ``chunking`` toggles chunking entirely and ``merge_small_chunks`` folds
18
19
  undersized chunks into neighbours. ``table_representation`` selects how
19
20
  tables are serialized ("markdown", "jsonl", or "melted"). ``extra_fields``
@@ -64,6 +64,7 @@ def _resolve_metadata(
64
64
  source_url=source_url,
65
65
  file_size_bytes=file_size_bytes,
66
66
  is_password_protected=bool(discovered.get("is_password_protected", False)),
67
+ renderer=discovered.get("renderer"),
67
68
  page_count=int(discovered.get("page_count") or 0),
68
69
  page_width=discovered.get("page_width"),
69
70
  page_height=discovered.get("page_height"),
@@ -435,8 +436,11 @@ def _build_result(
435
436
  doc_token_count: int | None = None
436
437
  doc_token_count_exact = False
437
438
  if not df_chunks.empty and "token_count" in df_chunks.columns:
439
+ from .shared.step_08_chunk_builder import token_encoder
438
440
  doc_token_count = int(df_chunks["token_count"].fillna(0).sum())
439
- doc_token_count_exact = bool(config.exact_tokens)
441
+ # Requesting exact counts is not getting them: tiktoken may be missing, or
442
+ # unable to fetch its vocabulary. Report what the counter actually did.
443
+ doc_token_count_exact = bool(config.exact_tokens) and token_encoder() is not None
440
444
 
441
445
  metadata = _resolve_metadata(
442
446
  discovered_metadata, source_url, source_filename, file_size_bytes, run_id, df_blocks,
@@ -502,24 +502,81 @@ def _collect_block_ids(node: HierarchyNode, recursive: bool) -> set[str]:
502
502
  return ids
503
503
 
504
504
 
505
+ def _split_table_row(line: str) -> list[str]:
506
+ """Split one pipe-table line into stripped cells, dropping the outer pipes."""
507
+ cells = line.strip().split("|")
508
+ if cells and cells[0].strip() == "":
509
+ cells = cells[1:]
510
+ if cells and cells[-1].strip() == "":
511
+ cells = cells[:-1]
512
+ return [c.strip() for c in cells]
513
+
514
+
515
+ def _is_sep_cell(cell: str) -> bool:
516
+ return bool(cell) and all(c in "-:" for c in cell)
517
+
518
+
519
+ def _is_sep_row(row: list[str]) -> bool:
520
+ return any(row) and all(_is_sep_cell(c) for c in row if c)
521
+
522
+
523
+ def _gfm_normalize_table(markdown: str) -> str:
524
+ """Rewrite a table so it renders under GitHub-flavored Markdown.
525
+
526
+ ``_format_table_markdown`` is faithful to the source grid: it draws the
527
+ separator under the *last* header row, and omits it entirely for tables with
528
+ no header cells. GFM instead requires exactly one header row followed by one
529
+ separator, so:
530
+
531
+ * multiple header rows collapse into one, joined per column with " > "
532
+ (the convention ``_format_table_melted`` already uses for header paths), and
533
+ * headerless tables gain a blank header row, which keeps every source row
534
+ in the body rather than promoting one to a header the detector rejected.
535
+
536
+ Text that is not a pipe table passes through untouched.
537
+ """
538
+ lines = markdown.strip().splitlines()
539
+ if not lines or any(l.strip() and not l.strip().startswith("|") for l in lines):
540
+ return markdown
541
+
542
+ rows = [_split_table_row(l) for l in lines if l.strip().startswith("|")]
543
+ if not rows:
544
+ return markdown
545
+
546
+ n_cols = max(len(r) for r in rows)
547
+ for row in rows:
548
+ while len(row) < n_cols:
549
+ row.append("")
550
+
551
+ sep = ["---"] * n_cols
552
+ first_sep = next((i for i, row in enumerate(rows) if _is_sep_row(row)), None)
553
+
554
+ if first_sep is None:
555
+ rows = [[""] * n_cols, sep] + rows
556
+ elif first_sep == 0:
557
+ rows = [[""] * n_cols] + rows
558
+ elif first_sep > 1:
559
+ header = []
560
+ for col in range(n_cols):
561
+ parts: list[str] = []
562
+ for row in rows[:first_sep]:
563
+ # A rowspan header cell repeats down the header rows — "Region >
564
+ # Region" carries no more than "Region" does.
565
+ if row[col] and (not parts or parts[-1] != row[col]):
566
+ parts.append(row[col])
567
+ header.append(" > ".join(parts))
568
+ rows = [header, sep] + rows[first_sep + 1:]
569
+
570
+ return "\n".join("| " + " | ".join(row) + " |" for row in rows)
571
+
572
+
505
573
  def _prettify_table(markdown: str) -> str:
506
574
  """Reformat a markdown table so pipe characters are vertically aligned."""
507
575
  lines = markdown.strip().splitlines()
508
576
  if not lines:
509
577
  return markdown
510
578
 
511
- def _split_row(line: str) -> list[str]:
512
- cells = line.strip().split("|")
513
- if cells and cells[0].strip() == "":
514
- cells = cells[1:]
515
- if cells and cells[-1].strip() == "":
516
- cells = cells[:-1]
517
- return [c.strip() for c in cells]
518
-
519
- def _is_sep(cell: str) -> bool:
520
- return bool(cell) and all(c in "-:" for c in cell)
521
-
522
- rows = [_split_row(l) for l in lines if l.strip().startswith("|")]
579
+ rows = [_split_table_row(l) for l in lines if l.strip().startswith("|")]
523
580
  if not rows:
524
581
  return markdown
525
582
 
@@ -528,7 +585,7 @@ def _prettify_table(markdown: str) -> str:
528
585
  while len(row) < n_cols:
529
586
  row.append("")
530
587
 
531
- sep_indices = {i for i, row in enumerate(rows) if all(_is_sep(c) for c in row if c)}
588
+ sep_indices = {i for i, row in enumerate(rows) if _is_sep_row(row)}
532
589
  col_widths = [3] * n_cols
533
590
  for i, row in enumerate(rows):
534
591
  if i not in sep_indices:
@@ -664,8 +721,16 @@ class ParseResult:
664
721
  include_toc: bool = True,
665
722
  include_furniture: bool = True,
666
723
  prettify: bool = True,
724
+ gfm_tables: bool = True,
667
725
  ) -> str:
668
- """Render the document as Markdown using blocks as the source of truth."""
726
+ """Render the document as Markdown using blocks as the source of truth.
727
+
728
+ ``gfm_tables`` rewrites tables to the one-header-row-plus-separator shape
729
+ GitHub-flavored Markdown renderers require. Pass ``False`` for the faithful
730
+ grid — multiple header rows, and no separator for tables that genuinely
731
+ have no header — which is what ``Table.markdown`` and the block text always
732
+ carry regardless of this flag.
733
+ """
669
734
  _HEADING_ROLES = {
670
735
  "heading", "toc_heading", "exhibit_heading", "hybrid_heading_paragraph",
671
736
  }
@@ -704,6 +769,8 @@ class ParseResult:
704
769
  if include_tables:
705
770
  table = tables_by_id.get(block.table_ids[0]) if block.table_ids else None
706
771
  raw = table.markdown if table else text
772
+ if gfm_tables:
773
+ raw = _gfm_normalize_table(raw)
707
774
  parts.append(_prettify_table(raw) if prettify else raw)
708
775
  elif block.type == "chart":
709
776
  if text: