langparse 0.1.0__tar.gz → 0.1.0rc2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. {langparse-0.1.0/langparse.egg-info → langparse-0.1.0rc2}/PKG-INFO +6 -58
  2. {langparse-0.1.0 → langparse-0.1.0rc2}/README.md +3 -55
  3. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/__init__.py +1 -16
  4. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/workbook.py +1 -43
  5. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/cli.py +1 -44
  6. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc_engine.py +3 -27
  7. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru.py +4 -28
  8. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/simple.py +1 -8
  9. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/errors.py +1 -19
  10. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/excel_parser.py +3 -15
  11. langparse-0.1.0rc2/langparse/py.typed +0 -0
  12. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/batch_service.py +6 -92
  13. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/parse_service.py +31 -86
  14. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/types.py +0 -2
  15. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/__init__.py +0 -4
  16. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/adapters.py +32 -70
  17. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/assembly.py +2 -22
  18. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/evaluator.py +2 -18
  19. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/schema.py +2 -14
  20. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/rendering.py +4 -10
  21. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/types.py +1 -8
  22. {langparse-0.1.0 → langparse-0.1.0rc2/langparse.egg-info}/PKG-INFO +6 -58
  23. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/SOURCES.txt +1 -28
  24. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/requires.txt +2 -2
  25. {langparse-0.1.0 → langparse-0.1.0rc2}/pyproject.toml +3 -4
  26. langparse-0.1.0rc2/tests/test_errors.py +29 -0
  27. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_benchmark.py +2 -2
  28. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_evaluator.py +0 -17
  29. langparse-0.1.0/MANIFEST.in +0 -6
  30. langparse-0.1.0/SKILLS.md +0 -22
  31. langparse-0.1.0/docs/CHUNKING.md +0 -51
  32. langparse-0.1.0/docs/RELEASE_0.1.0.md +0 -38
  33. langparse-0.1.0/docs/WORKBOOK_BUNDLE.md +0 -42
  34. langparse-0.1.0/examples/agent_ingestion.py +0 -40
  35. langparse-0.1.0/langparse/chunkers/__init__.py +0 -12
  36. langparse-0.1.0/langparse/chunkers/registry.py +0 -38
  37. langparse-0.1.0/langparse/chunkers/text.py +0 -96
  38. langparse-0.1.0/langparse/progress.py +0 -77
  39. langparse-0.1.0/langparse/workbooks/bundle-v1.schema.json +0 -71
  40. langparse-0.1.0/langparse/workbooks/bundle.py +0 -341
  41. langparse-0.1.0/langparse/workbooks/lineage.py +0 -117
  42. langparse-0.1.0/langparse/workbooks/objects.py +0 -229
  43. langparse-0.1.0/langparse/workbooks/quality/bundle.py +0 -53
  44. langparse-0.1.0/langparse/workbooks/quality/facts.py +0 -142
  45. langparse-0.1.0/langparse/workbooks/reference_types.py +0 -73
  46. langparse-0.1.0/langparse/workbooks/references.py +0 -178
  47. langparse-0.1.0/skills/langparse/SKILL.md +0 -38
  48. langparse-0.1.0/tests/fixtures/workbook-bundle-v1.json +0 -11
  49. langparse-0.1.0/tests/test_batch_progress.py +0 -224
  50. langparse-0.1.0/tests/test_chunk_strategies.py +0 -131
  51. langparse-0.1.0/tests/test_errors.py +0 -60
  52. langparse-0.1.0/tests/test_pdf_progress.py +0 -105
  53. langparse-0.1.0/tests/test_progress.py +0 -126
  54. langparse-0.1.0/tests/test_workbook_bundle.py +0 -138
  55. langparse-0.1.0/tests/test_workbook_fact_quality.py +0 -124
  56. langparse-0.1.0/tests/test_workbook_lineage.py +0 -141
  57. langparse-0.1.0/tests/test_workbook_objects.py +0 -196
  58. {langparse-0.1.0 → langparse-0.1.0rc2}/LICENSE +0 -0
  59. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/autoparser.py +0 -0
  60. {langparse-0.1.0/langparse/core → langparse-0.1.0rc2/langparse/chunkers}/__init__.py +0 -0
  61. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/blocks.py +0 -0
  62. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/profiles.py +0 -0
  63. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/chunkers/semantic.py +0 -0
  64. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/config.py +0 -0
  65. {langparse-0.1.0/langparse/parsers → langparse-0.1.0rc2/langparse/core}/__init__.py +0 -0
  66. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/chunker.py +0 -0
  67. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/engine.py +0 -0
  68. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/parser.py +0 -0
  69. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/core/rendering.py +0 -0
  70. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/__init__.py +0 -0
  71. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/__init__.py +0 -0
  72. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/__init__.py +0 -0
  73. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/layout_recognizer.py +0 -0
  74. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/model_loader.py +0 -0
  75. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/ocr.py +0 -0
  76. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/operators.py +0 -0
  77. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/pdf_parser.py +0 -0
  78. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/postprocess.py +0 -0
  79. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/recognizer.py +0 -0
  80. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/rendering.py +0 -0
  81. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/table_structure_recognizer.py +0 -0
  82. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/tokenizer.py +0 -0
  83. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/deepdoc/utils.py +0 -0
  84. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru_client.py +0 -0
  85. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/mineru_service.py +0 -0
  86. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/ocr.py +0 -0
  87. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/other.py +0 -0
  88. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/engines/pdf/vision_llm.py +0 -0
  89. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/logging.py +0 -0
  90. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/metrics.py +0 -0
  91. /langparse-0.1.0/langparse/py.typed → /langparse-0.1.0rc2/langparse/parsers/__init__.py +0 -0
  92. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/docx_parser.py +0 -0
  93. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/markdown_parser.py +0 -0
  94. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/pdf_parser.py +0 -0
  95. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/registry.py +0 -0
  96. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/parsers/sniff.py +0 -0
  97. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/__init__.py +0 -0
  98. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/benchmark_service.py +0 -0
  99. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/fidelity.py +0 -0
  100. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/output_paths.py +0 -0
  101. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/quality.py +0 -0
  102. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/workbook_ambiguity_benchmark.py +0 -0
  103. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/services/workbook_quality_benchmark.py +0 -0
  104. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/blocks.py +0 -0
  105. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/classification.py +0 -0
  106. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/continuation.py +0 -0
  107. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/__init__.py +0 -0
  108. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/evaluator.py +0 -0
  109. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/evaluation/schema.py +0 -0
  110. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/labels.py +0 -0
  111. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/__init__.py +0 -0
  112. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/cache.py +0 -0
  113. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/config.py +0 -0
  114. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/contract.py +0 -0
  115. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/disambiguation.py +0 -0
  116. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/openai_adapter.py +0 -0
  117. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/policy.py +0 -0
  118. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/ports.py +0 -0
  119. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/pricing.py +0 -0
  120. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/modeling/types.py +0 -0
  121. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/quality/__init__.py +0 -0
  122. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/regions.py +0 -0
  123. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse/workbooks/tables.py +0 -0
  124. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/dependency_links.txt +0 -0
  125. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/entry_points.txt +0 -0
  126. {langparse-0.1.0 → langparse-0.1.0rc2}/langparse.egg-info/top_level.txt +0 -0
  127. {langparse-0.1.0 → langparse-0.1.0rc2}/setup.cfg +0 -0
  128. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_autoparser.py +0 -0
  129. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_batch_service.py +0 -0
  130. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_benchmark_service.py +0 -0
  131. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_blocks.py +0 -0
  132. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunk_pipeline.py +0 -0
  133. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunker.py +0 -0
  134. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_chunker_sizing.py +0 -0
  135. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_cli.py +0 -0
  136. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_config.py +0 -0
  137. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_content_sniffing.py +0 -0
  138. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_credential_isolation.py +0 -0
  139. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_engine.py +0 -0
  140. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_model_loader.py +0 -0
  141. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_rendering.py +0 -0
  142. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_deepdoc_tokenizer.py +0 -0
  143. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_engine_registry.py +0 -0
  144. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_evaluation_strictness.py +0 -0
  145. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_logical_parser.py +0 -0
  146. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_model_cli.py +0 -0
  147. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_model_modes.py +0 -0
  148. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_excel_structural_parser.py +0 -0
  149. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_fidelity.py +0 -0
  150. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_logging.py +0 -0
  151. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_metrics.py +0 -0
  152. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_mineru_engine.py +0 -0
  153. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_mineru_live_integration.py +0 -0
  154. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_ocr.py +0 -0
  155. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_ooxml_adapter.py +0 -0
  156. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_optional_dependency_boundaries.py +0 -0
  157. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parse_service.py +0 -0
  158. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parser_results.py +0 -0
  159. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_parsers.py +0 -0
  160. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_pdf_parser.py +0 -0
  161. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_quality.py +0 -0
  162. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_quota_fail_closed.py +0 -0
  163. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_real_sample_regression.py +0 -0
  164. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_result_envelope.py +0 -0
  165. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_ambiguity_benchmark.py +0 -0
  166. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_ambiguity_evaluator.py +0 -0
  167. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly.py +0 -0
  168. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly_blocks.py +0 -0
  169. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_assembly_modeling.py +0 -0
  170. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_blocks.py +0 -0
  171. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_chunk_profiles.py +0 -0
  172. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_chunker.py +0 -0
  173. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_classification.py +0 -0
  174. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_continuation.py +0 -0
  175. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_disambiguation.py +0 -0
  176. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_evaluation_schema.py +0 -0
  177. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_config.py +0 -0
  178. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_contract.py +0 -0
  179. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_policy.py +0 -0
  180. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_pricing.py +0 -0
  181. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_model_safety_drills.py +0 -0
  182. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_openai_adapter.py +0 -0
  183. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_quality_schema.py +0 -0
  184. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_real_layouts.py +0 -0
  185. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_region_assessment.py +0 -0
  186. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_regions.py +0 -0
  187. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_rendering.py +0 -0
  188. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_tables.py +0 -0
  189. {langparse-0.1.0 → langparse-0.1.0rc2}/tests/test_workbook_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: langparse
3
- Version: 0.1.0
3
+ Version: 0.1.0rc2
4
4
  Summary: A developer-friendly document parsing toolkit with precise, source-grounded Excel understanding.
5
5
  Author-email: syw2014 <syw2014@gmail.com>
6
6
  License-Expression: Apache-2.0
@@ -40,7 +40,7 @@ Provides-Extra: deepdoc
40
40
  Requires-Dist: pdfplumber>=0.11.10; extra == "deepdoc"
41
41
  Requires-Dist: opencv-python-headless>=4.9.0; extra == "deepdoc"
42
42
  Requires-Dist: onnxruntime>=1.17.0; extra == "deepdoc"
43
- Requires-Dist: pypdf>=6.19.0; extra == "deepdoc"
43
+ Requires-Dist: pypdf>=6.15.0; extra == "deepdoc"
44
44
  Requires-Dist: huggingface_hub>=0.20.0; extra == "deepdoc"
45
45
  Requires-Dist: scikit-learn>=1.3.0; extra == "deepdoc"
46
46
  Requires-Dist: shapely>=2.0.0; extra == "deepdoc"
@@ -58,7 +58,7 @@ Requires-Dist: rapidocr_onnxruntime; extra == "all"
58
58
  Requires-Dist: mineru<4,>=3.4; extra == "all"
59
59
  Requires-Dist: opencv-python-headless; extra == "all"
60
60
  Requires-Dist: onnxruntime; extra == "all"
61
- Requires-Dist: pypdf>=6.19.0; extra == "all"
61
+ Requires-Dist: pypdf>=6.15.0; extra == "all"
62
62
  Requires-Dist: huggingface_hub; extra == "all"
63
63
  Requires-Dist: scikit-learn; extra == "all"
64
64
  Requires-Dist: shapely; extra == "all"
@@ -112,59 +112,13 @@ where LangParse goes deliberately deeper.
112
112
 
113
113
  ---
114
114
 
115
- ## Parsing progress (unreleased)
116
-
117
- Pass `progress_callback` to `AutoParser`, `ParseService`, `PDFParser`, or
118
- `BatchParseService.run`. The callback receives immutable `ProgressEvent` objects:
119
-
120
- ```python
121
- from langparse import AutoParser
122
-
123
- result = AutoParser.parse_result(
124
- "budget.xlsx",
125
- progress_callback=lambda event: print(event.phase, event.state, event.percent),
126
- )
127
- ```
128
-
129
- Events contain `source`, `phase`, `state`, `completed_units`, `total_units`, `unit`,
130
- `percent`, and `message`. Percent is phase-local (0–100), not an ETA; unknown
131
- values remain `None`. Only a `file` event with `completed`/`failed` marks the end
132
- of parsing and requested chunking, before any caller-side export. A completed
133
- parse can still have partial quality diagnostics.
134
-
135
- Simple PDF reports page counts; DeepDoc forwards its internal weighted-stage or
136
- page-batch progress; MinerU reports preparation, remote parsing, and rendering
137
- without an internal percentage. Excel reports extraction, assembly (including
138
- optional model disambiguation), and rendering. Direct `ExcelParser` calls also
139
- emit those stages; the facade/service supplies the file lifecycle.
140
-
141
- Batch events use `phase="batch"`, `source=""`, and `unit="files"`. Notifications
142
- follow completion order while returned results retain the batch's sorted file order. Success,
143
- failure, and skip all count as finished files. Batch completion follows report
144
- writing and does not mean every file succeeded. Each `batch_item` event names
145
- the source and reports `completed`/`failed`/`skipped` after that item's output
146
- write. Empty batches report 0/0 with
147
- unknown percent. Ordinary callback exceptions are logged by type and isolated
148
- from parsing. Callbacks are synchronous: keep them brief. One batch serializes
149
- its callbacks, which can run on different threads; callers coordinate callbacks
150
- shared across independent runs. There is no task store, background execution,
151
- cancellation, heartbeat, or HTTP query endpoint.
152
-
153
- CLI `--progress` writes line-oriented updates to stderr, leaving stdout intact:
154
-
155
- ```bash
156
- langparse parse budget.xlsx --format json --progress > result.json
157
- langparse parse docs/ --batch --output-dir out --progress
158
- ```
159
-
160
115
  ## Project status
161
116
 
162
- The latest GitHub release is `0.1.0`. Core multi-format parsing,
117
+ The current release candidate is `0.1.0rc2`. Core multi-format parsing,
163
118
  structured OOXML workbooks, semantic chunking, batch processing, quality checks,
164
119
  and CI are available today. LangParse remains pre-1.0; see
165
120
  [docs/PROGRESS.md](docs/PROGRESS.md) for the module-by-module source of truth and
166
- known gaps. See the
167
- [release scope](docs/RELEASE_0.1.0.md).
121
+ known gaps.
168
122
 
169
123
  ## Why LangParse?
170
124
 
@@ -214,7 +168,7 @@ views derived from it, not replacements for it.
214
168
  Install the current release candidate:
215
169
 
216
170
  ```bash
217
- pip install "langparse==0.1.0"
171
+ pip install --pre "langparse==0.1.0rc2"
218
172
  ```
219
173
 
220
174
  Install only the optional capabilities you need:
@@ -782,9 +736,3 @@ See [CHANGELOG.md](CHANGELOG.md) ([中文](CHANGELOG_cn.md)) for release notes a
782
736
 
783
737
  ## License
784
738
  This project is licensed under the [Apache 2.0 License](https://www.apache.org/licenses/LICENSE-2.0).
785
-
786
- ### v0.1.0 capabilities
787
-
788
- - [Workbook Bundle 与查询](docs/WORKBOOK_BUNDLE.md)
789
- - [分块策略与 CLI](docs/CHUNKING.md)
790
- - [Agent Skill 与接入示例](SKILLS.md)
@@ -33,59 +33,13 @@ where LangParse goes deliberately deeper.
33
33
 
34
34
  ---
35
35
 
36
- ## Parsing progress (unreleased)
37
-
38
- Pass `progress_callback` to `AutoParser`, `ParseService`, `PDFParser`, or
39
- `BatchParseService.run`. The callback receives immutable `ProgressEvent` objects:
40
-
41
- ```python
42
- from langparse import AutoParser
43
-
44
- result = AutoParser.parse_result(
45
- "budget.xlsx",
46
- progress_callback=lambda event: print(event.phase, event.state, event.percent),
47
- )
48
- ```
49
-
50
- Events contain `source`, `phase`, `state`, `completed_units`, `total_units`, `unit`,
51
- `percent`, and `message`. Percent is phase-local (0–100), not an ETA; unknown
52
- values remain `None`. Only a `file` event with `completed`/`failed` marks the end
53
- of parsing and requested chunking, before any caller-side export. A completed
54
- parse can still have partial quality diagnostics.
55
-
56
- Simple PDF reports page counts; DeepDoc forwards its internal weighted-stage or
57
- page-batch progress; MinerU reports preparation, remote parsing, and rendering
58
- without an internal percentage. Excel reports extraction, assembly (including
59
- optional model disambiguation), and rendering. Direct `ExcelParser` calls also
60
- emit those stages; the facade/service supplies the file lifecycle.
61
-
62
- Batch events use `phase="batch"`, `source=""`, and `unit="files"`. Notifications
63
- follow completion order while returned results retain the batch's sorted file order. Success,
64
- failure, and skip all count as finished files. Batch completion follows report
65
- writing and does not mean every file succeeded. Each `batch_item` event names
66
- the source and reports `completed`/`failed`/`skipped` after that item's output
67
- write. Empty batches report 0/0 with
68
- unknown percent. Ordinary callback exceptions are logged by type and isolated
69
- from parsing. Callbacks are synchronous: keep them brief. One batch serializes
70
- its callbacks, which can run on different threads; callers coordinate callbacks
71
- shared across independent runs. There is no task store, background execution,
72
- cancellation, heartbeat, or HTTP query endpoint.
73
-
74
- CLI `--progress` writes line-oriented updates to stderr, leaving stdout intact:
75
-
76
- ```bash
77
- langparse parse budget.xlsx --format json --progress > result.json
78
- langparse parse docs/ --batch --output-dir out --progress
79
- ```
80
-
81
36
  ## Project status
82
37
 
83
- The latest GitHub release is `0.1.0`. Core multi-format parsing,
38
+ The current release candidate is `0.1.0rc2`. Core multi-format parsing,
84
39
  structured OOXML workbooks, semantic chunking, batch processing, quality checks,
85
40
  and CI are available today. LangParse remains pre-1.0; see
86
41
  [docs/PROGRESS.md](docs/PROGRESS.md) for the module-by-module source of truth and
87
- known gaps. See the
88
- [release scope](docs/RELEASE_0.1.0.md).
42
+ known gaps.
89
43
 
90
44
  ## Why LangParse?
91
45
 
@@ -135,7 +89,7 @@ views derived from it, not replacements for it.
135
89
  Install the current release candidate:
136
90
 
137
91
  ```bash
138
- pip install "langparse==0.1.0"
92
+ pip install --pre "langparse==0.1.0rc2"
139
93
  ```
140
94
 
141
95
  Install only the optional capabilities you need:
@@ -703,9 +657,3 @@ See [CHANGELOG.md](CHANGELOG.md) ([中文](CHANGELOG_cn.md)) for release notes a
703
657
 
704
658
  ## License
705
659
  This project is licensed under the [Apache 2.0 License](https://www.apache.org/licenses/LICENSE-2.0).
706
-
707
- ### v0.1.0 capabilities
708
-
709
- - [Workbook Bundle 与查询](docs/WORKBOOK_BUNDLE.md)
710
- - [分块策略与 CLI](docs/CHUNKING.md)
711
- - [Agent Skill 与接入示例](SKILLS.md)
@@ -3,14 +3,7 @@ from importlib.metadata import version as _distribution_version
3
3
  __version__ = _distribution_version("langparse")
4
4
 
5
5
  from langparse.autoparser import AutoParser
6
- from langparse.chunkers import (
7
- FixedTokenChunker,
8
- SemanticChunker,
9
- SlidingWindowChunker,
10
- available_chunkers,
11
- create_chunker,
12
- register_chunker,
13
- )
6
+ from langparse.chunkers.semantic import SemanticChunker
14
7
  from langparse.core.chunker import BaseChunker
15
8
  from langparse.core.parser import BaseParser
16
9
  from langparse.metrics import BatchItemResult, BatchRunResult, ParseMetrics
@@ -18,7 +11,6 @@ from langparse.parsers.docx_parser import DocxParser
18
11
  from langparse.parsers.excel_parser import ExcelParser
19
12
  from langparse.parsers.markdown_parser import MarkdownParser
20
13
  from langparse.parsers.pdf_parser import PDFParser
21
- from langparse.progress import ProgressCallback, ProgressEvent
22
14
  from langparse.types import (
23
15
  Chunk,
24
16
  Document,
@@ -42,14 +34,7 @@ __all__ = [
42
34
  "DocxParser",
43
35
  "ExcelParser",
44
36
  "SemanticChunker",
45
- "FixedTokenChunker",
46
- "SlidingWindowChunker",
47
- "available_chunkers",
48
- "create_chunker",
49
- "register_chunker",
50
37
  "ParseMetrics",
51
- "ProgressCallback",
52
- "ProgressEvent",
53
38
  "BatchItemResult",
54
39
  "BatchRunResult",
55
40
  ]
@@ -1,7 +1,6 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  from collections.abc import Callable
4
- from dataclasses import asdict
5
4
 
6
5
  from openpyxl.utils import get_column_letter, range_boundaries
7
6
 
@@ -19,7 +18,6 @@ from langparse.workbooks.types import (
19
18
  LogicalTable,
20
19
  MatrixBlock,
21
20
  MatrixHeader,
22
- SourceRef,
23
21
  TableContinuation,
24
22
  TextBlock,
25
23
  TextLine,
@@ -115,18 +113,6 @@ class WorkbookStructuralChunker:
115
113
  sheet_snapshot.visibility if sheet_snapshot is not None else sheet_ir.visibility
116
114
  )
117
115
  chunk.metadata["hidden_row_numbers"] = sorted(referenced_rows & hidden_rows)
118
- if self.policy.analysis_records:
119
- lineage = workbook_ir.lineage
120
- dependencies = []
121
- for source in chunk.metadata["source_ranges"]:
122
- name, cell_range = source.rsplit("!", 1)
123
- for edge in (
124
- lineage.dependencies_of(SourceRef(name, cell_range)) if lineage else []
125
- ):
126
- payload = asdict(edge)
127
- if payload not in dependencies:
128
- dependencies.append(payload)
129
- chunk.structured_payload["dependencies"] = dependencies
130
116
 
131
117
  def _validate_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
132
118
  workbook_ir = parsed.structure
@@ -155,12 +141,7 @@ class WorkbookStructuralChunker:
155
141
  for chunk in chunks:
156
142
  source_ranges = chunk.metadata["source_ranges"]
157
143
  for source_range in source_ranges:
158
- if chunk.metadata["chunk_type"] in {"chart", "image"}:
159
- from langparse.workbooks.objects import validate_object_source
160
-
161
- validate_object_source(workbook_ir.snapshot, source_range)
162
- else:
163
- _source_range_is_valid(workbook_ir.snapshot, source_range)
144
+ _source_range_is_valid(workbook_ir.snapshot, source_range)
164
145
  if chunk.metadata["chunk_type"] != "table_rows":
165
146
  continue
166
147
  payload = chunk.structured_payload
@@ -186,29 +167,6 @@ class WorkbookStructuralChunker:
186
167
  block: WorkbookBlock,
187
168
  chunk_index_offset: int,
188
169
  ) -> list[Chunk]:
189
- if block.kind in {"chart", "image"}:
190
- from langparse.workbooks.objects import render_object
191
-
192
- content = render_object(block)
193
- metadata = document_metadata(parsed)
194
- metadata.update(
195
- {
196
- "chunk_type": block.kind,
197
- "chunk_index": chunk_index_offset,
198
- "sheet_name": sheet_name,
199
- "sheet_ordinal": sheet_ordinal,
200
- "source_ranges": [ref.key for ref in block.source_refs],
201
- "object_id": block.block_id,
202
- "oversized": self.length_function(content) > self.max_chunk_size,
203
- }
204
- )
205
- return [
206
- Chunk(
207
- content=content,
208
- metadata=metadata,
209
- structured_payload={"object": block.metadata["object"]},
210
- )
211
- ]
212
170
  if block.logical_table is not None:
213
171
  return self._chunk_logical_table(
214
172
  parsed,
@@ -6,9 +6,7 @@ from collections.abc import Sequence
6
6
  from pathlib import Path
7
7
 
8
8
  from langparse import __version__
9
- from langparse.chunkers.registry import available_chunkers
10
9
  from langparse.errors import classify_exception
11
- from langparse.progress import ProgressEvent
12
10
  from langparse.services.batch_service import BatchParseService
13
11
  from langparse.services.benchmark_service import BenchmarkService
14
12
  from langparse.services.parse_service import ParseService
@@ -45,27 +43,18 @@ def build_parser():
45
43
  parse_cmd.add_argument("--model-source", default=None)
46
44
  parse_cmd.add_argument("--auto-install-runtime", action="store_true")
47
45
  parse_cmd.add_argument("--runtime-package", default=None)
48
- parse_cmd.add_argument("--format", default="markdown", help="markdown, json, or workbook-json")
46
+ parse_cmd.add_argument("--format", default="markdown")
49
47
  parse_cmd.add_argument("--batch", action="store_true")
50
48
  parse_cmd.add_argument("--output", default=None)
51
49
  parse_cmd.add_argument("--output-dir", default=None)
52
50
  parse_cmd.add_argument("--max-workers", type=int, default=None)
53
51
  parse_cmd.add_argument("--skip-existing", action="store_true")
54
52
  parse_cmd.add_argument("--metrics", action="store_true")
55
- parse_cmd.add_argument("--progress", action="store_true", help="write progress to stderr")
56
53
  parse_cmd.add_argument(
57
54
  "--chunk",
58
55
  action="store_true",
59
56
  help="semantically chunk the parsed document and include chunks in the output",
60
57
  )
61
- parse_cmd.add_argument("--chunk-strategy", choices=available_chunkers(), default=None)
62
- parse_cmd.add_argument(
63
- "--chunk-size",
64
- type=int,
65
- default=None,
66
- help="strategy size budget (lexical tokens for fixed-token, characters otherwise)",
67
- )
68
- parse_cmd.add_argument("--chunk-overlap", type=int, default=None)
69
58
  parse_cmd.add_argument(
70
59
  "--chunk-profile",
71
60
  choices=["retrieval", "analysis"],
@@ -251,22 +240,6 @@ def _run(args, parser) -> int:
251
240
  if value is not None and value is not False
252
241
  }
253
242
  chunk_kwargs = {"chunk_profile": args.chunk_profile} if args.chunk else {}
254
- if any(
255
- value is not None for value in (args.chunk_strategy, args.chunk_size, args.chunk_overlap)
256
- ):
257
- if not args.chunk:
258
- parser.error("chunk strategy/size/overlap options require --chunk")
259
- chunk_kwargs["chunk_strategy"] = args.chunk_strategy or "semantic"
260
- chunk_kwargs["chunk_options"] = {
261
- key: value
262
- for key, value in {
263
- "max_chunk_size": args.chunk_size,
264
- "overlap": args.chunk_overlap,
265
- }.items()
266
- if value is not None
267
- }
268
- if args.progress:
269
- parse_kwargs["progress_callback"] = _print_progress
270
243
 
271
244
  if args.batch:
272
245
  # One implementation regardless of flags. Without --output-dir the run
@@ -309,21 +282,5 @@ def _run(args, parser) -> int:
309
282
  return 0
310
283
 
311
284
 
312
- def _print_progress(event: ProgressEvent) -> None:
313
- counts = ""
314
- if event.completed_units is not None:
315
- total = event.total_units if event.total_units is not None else "?"
316
- counts = f" {event.completed_units}/{total} {event.unit or ''}"
317
- percent = f" {event.percent:.0f}%" if event.percent is not None else ""
318
- # Escape embedded line breaks/control characters from untrusted filenames/messages.
319
- source = repr(event.source) if event.source else "batch"
320
- message = f" {event.message!r}" if event.message else ""
321
- print(
322
- f"langparse: {source} {event.phase} {event.state}{counts}{percent}{message}",
323
- file=sys.stderr,
324
- flush=True,
325
- )
326
-
327
-
328
285
  if __name__ == "__main__":
329
286
  raise SystemExit(main())
@@ -6,7 +6,6 @@ from typing import Any
6
6
  from langparse.core.engine import PageResult
7
7
  from langparse.engines.pdf.simple import BasePDFEngine
8
8
  from langparse.logging import get_logger
9
- from langparse.progress import ProgressCallback, ProgressReporter
10
9
  from langparse.types import ParsedDocumentResult
11
10
 
12
11
  logger = get_logger(__name__)
@@ -98,15 +97,7 @@ class DeepDocEngine(BasePDFEngine):
98
97
  logger.warning("Skipping OCR-applied classification for %s: %s", file_path, exc)
99
98
  return {}
100
99
 
101
- def process_document(
102
- self,
103
- file_path: Path,
104
- *,
105
- progress_callback: ProgressCallback | None = None,
106
- **kwargs: Any,
107
- ) -> ParsedDocumentResult:
108
- reporter = ProgressReporter(str(file_path), progress_callback)
109
- reporter.emit("preparing", message="Preparing DeepDoc parser")
100
+ def process_document(self, file_path: Path, **kwargs: Any) -> ParsedDocumentResult:
110
101
  try:
111
102
  from langparse.engines.pdf.deepdoc.rendering import render_pages
112
103
  except ImportError as exc:
@@ -117,19 +108,8 @@ class DeepDocEngine(BasePDFEngine):
117
108
  with self._parser_lock:
118
109
  if self._parser is None:
119
110
  self._parser = self._build_parser()
120
- if progress_callback is None:
121
- boxes = self._parser.parse_into_bboxes(str(file_path))
122
- else:
123
- last_percent = 0.0
124
-
125
- def on_progress(progress, message=""):
126
- nonlocal last_percent
127
- last_percent = max(last_percent, min(100.0, max(0.0, progress * 100)))
128
- reporter.emit("parsing", percent=last_percent, message=message)
129
-
130
- boxes = self._parser.parse_into_bboxes(str(file_path), callback=on_progress)
111
+ boxes = self._parser.parse_into_bboxes(str(file_path))
131
112
 
132
- reporter.emit("rendering")
133
113
  ocr_pages = self._classify_ocr_pages(file_path)
134
114
  pages = render_pages(boxes, ocr_pages=ocr_pages)
135
115
  return ParsedDocumentResult(
@@ -146,11 +126,7 @@ class DeepDocEngine(BasePDFEngine):
146
126
  },
147
127
  )
148
128
 
149
- def process(
150
- self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
151
- ) -> Iterator[PageResult]:
152
- if progress_callback is not None:
153
- kwargs["progress_callback"] = progress_callback
129
+ def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
154
130
  parsed = self.process_document(file_path, **kwargs)
155
131
  for page in parsed.pages:
156
132
  yield PageResult(
@@ -6,7 +6,6 @@ from langparse.core.engine import PageResult
6
6
  from langparse.engines.pdf.mineru_client import MinerUClient
7
7
  from langparse.engines.pdf.mineru_service import MinerUServiceManager
8
8
  from langparse.engines.pdf.simple import BasePDFEngine
9
- from langparse.progress import ProgressCallback, ProgressReporter
10
9
  from langparse.types import ParsedDocumentResult, ParsedElement, ParsedPageResult
11
10
 
12
11
  _SENSITIVE_OPTION_KEYS = frozenset(
@@ -152,35 +151,16 @@ class MinerUEngine(BasePDFEngine):
152
151
  def _create_service_manager(self) -> MinerUServiceManager:
153
152
  return MinerUServiceManager(**self._build_service_config())
154
153
 
155
- def _run_mineru(
156
- self,
157
- file_path: Path,
158
- runtime_config: dict[str, Any],
159
- progress_callback: ProgressCallback | None = None,
160
- ) -> list[dict[str, Any]]:
161
- reporter = ProgressReporter(str(file_path), progress_callback)
154
+ def _run_mineru(self, file_path: Path, runtime_config: dict[str, Any]) -> list[dict[str, Any]]:
162
155
  manager = self._create_service_manager()
163
156
  with manager.running_service() as base_url:
164
157
  client = self._create_client(base_url)
165
- reporter.emit("parsing", message="Waiting for MinerU response")
166
158
  return client.parse_file(file_path, runtime_config)
167
159
 
168
- def process_document(
169
- self,
170
- file_path: Path,
171
- *,
172
- progress_callback: ProgressCallback | None = None,
173
- **kwargs: Any,
174
- ) -> ParsedDocumentResult:
175
- reporter = ProgressReporter(str(file_path), progress_callback)
176
- reporter.emit("preparing", message="Preparing MinerU service")
160
+ def process_document(self, file_path: Path, **kwargs: Any) -> ParsedDocumentResult:
177
161
  self._ensure_runtime()
178
162
  runtime_config = self._build_runtime_config(**kwargs)
179
- if progress_callback is None:
180
- raw_pages = self._run_mineru(file_path, runtime_config)
181
- else:
182
- raw_pages = self._run_mineru(file_path, runtime_config, progress_callback)
183
- reporter.emit("rendering")
163
+ raw_pages = self._run_mineru(file_path, runtime_config)
184
164
  pages = [
185
165
  ParsedPageResult(
186
166
  page_number=item["page_number"],
@@ -241,11 +221,7 @@ class MinerUEngine(BasePDFEngine):
241
221
  },
242
222
  )
243
223
 
244
- def process(
245
- self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
246
- ) -> Iterator[PageResult]:
247
- if progress_callback is not None:
248
- kwargs["progress_callback"] = progress_callback
224
+ def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
249
225
  parsed = self.process_document(file_path, **kwargs)
250
226
  for page in parsed.pages:
251
227
  yield PageResult(
@@ -10,7 +10,6 @@ from langparse.engines.pdf.ocr import (
10
10
  needs_ocr,
11
11
  ocr_page_text,
12
12
  )
13
- from langparse.progress import ProgressCallback, ProgressReporter
14
13
 
15
14
 
16
15
  class BasePDFEngine(BaseEngine):
@@ -50,10 +49,7 @@ class SimplePDFEngine(BasePDFEngine):
50
49
  # guarantee -- correctness over throughput on an already slow path.
51
50
  self._ocr_lock = threading.Lock()
52
51
 
53
- def process(
54
- self, file_path: Path, *, progress_callback: ProgressCallback | None = None, **kwargs
55
- ) -> Iterator[PageResult]:
56
- reporter = ProgressReporter(str(file_path), progress_callback)
52
+ def process(self, file_path: Path, **kwargs) -> Iterator[PageResult]:
57
53
  try:
58
54
  import pdfplumber
59
55
  except ImportError:
@@ -76,9 +72,6 @@ class SimplePDFEngine(BasePDFEngine):
76
72
  if table_markdown:
77
73
  markdown_content = "\n\n".join([text, "\n".join(table_markdown)]).strip()
78
74
 
79
- reporter.emit(
80
- "parsing", completed_units=i + 1, total_units=len(pdf.pages), unit="pages"
81
- )
82
75
  yield PageResult(
83
76
  page_number=i + 1,
84
77
  markdown_content=markdown_content,
@@ -1,10 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
- import re
4
3
  from dataclasses import dataclass
5
4
  from enum import Enum
6
- from subprocess import TimeoutExpired
7
- from urllib.error import URLError
8
5
 
9
6
 
10
7
  class ErrorType(str, Enum):
@@ -45,7 +42,7 @@ def classify_exception(exc: BaseException) -> ClassifiedError:
45
42
  return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
46
43
  if "unable to start local mineru-api" in lowered:
47
44
  return ClassifiedError(ErrorType.ENGINE_UNAVAILABLE, message)
48
- if _is_timeout(exc) or re.search(r"\btimed out\b", lowered):
45
+ if "timed out" in lowered or "timeout" in lowered:
49
46
  return ClassifiedError(ErrorType.ENGINE_TIMEOUT, message)
50
47
  if "ocr" in lowered and "unavailable" in lowered:
51
48
  return ClassifiedError(ErrorType.OCR_UNAVAILABLE, message)
@@ -53,18 +50,3 @@ def classify_exception(exc: BaseException) -> ClassifiedError:
53
50
  return ClassifiedError(ErrorType.TABLE_EXTRACTION_FAILED, message)
54
51
 
55
52
  return ClassifiedError(ErrorType.PARSE_FAILED, message)
56
-
57
-
58
- def _is_timeout(exc: BaseException) -> bool:
59
- """Recognize standard timeout types through explicit backend wrappers."""
60
- seen: set[int] = set()
61
- current: BaseException | None = exc
62
- while current is not None and id(current) not in seen:
63
- seen.add(id(current))
64
- if isinstance(current, (TimeoutError, TimeoutExpired)):
65
- return True
66
- if isinstance(current, URLError) and isinstance(current.reason, BaseException):
67
- current = current.reason
68
- else:
69
- current = current.__cause__
70
- return False
@@ -3,7 +3,6 @@ from pathlib import Path
3
3
 
4
4
  from langparse.core.parser import BaseParser
5
5
  from langparse.parsers.sniff import looks_like_ole_binary, looks_like_zip_ooxml
6
- from langparse.progress import ProgressCallback, ProgressReporter
7
6
  from langparse.types import ParsedDocumentResult, ParsedElement, ParseDiagnostics, ParsedPageResult
8
7
  from langparse.workbooks.modeling import (
9
8
  RequiredWorkbookDisambiguationError,
@@ -65,14 +64,11 @@ class ExcelParser(BaseParser):
65
64
  WorkbookDisambiguation.off() if disambiguation is None else disambiguation
66
65
  )
67
66
 
68
- def parse_result(
69
- self, file_path: str | Path, *, progress_callback: ProgressCallback | None = None, **kwargs
70
- ) -> ParsedDocumentResult:
67
+ def parse_result(self, file_path: str | Path, **kwargs) -> ParsedDocumentResult:
71
68
  path = self._resolve_existing_path(file_path)
72
- reporter = ProgressReporter(path, progress_callback)
73
69
 
74
70
  if looks_like_zip_ooxml(path):
75
- return self._parse_ooxml(path, reporter)
71
+ return self._parse_ooxml(path)
76
72
 
77
73
  try:
78
74
  import pandas as pd
@@ -86,7 +82,6 @@ class ExcelParser(BaseParser):
86
82
  # otherwise be handed to the wrong pandas reader. Content decides:
87
83
  # a real workbook is either a ZIP-OOXML or legacy-OLE container;
88
84
  # anything else is read as delimited text regardless of its label.
89
- reporter.emit("extracting")
90
85
  is_legacy_workbook = looks_like_ole_binary(path)
91
86
  if is_legacy_workbook:
92
87
  sheets = pd.read_excel(path, sheet_name=None)
@@ -110,7 +105,6 @@ class ExcelParser(BaseParser):
110
105
  )
111
106
  }
112
107
 
113
- reporter.emit("rendering")
114
108
  pages = [
115
109
  self._page_for_sheet(index + 1, sheet_name, frame)
116
110
  for index, (sheet_name, frame) in enumerate(sheets.items())
@@ -138,7 +132,7 @@ class ExcelParser(BaseParser):
138
132
  ),
139
133
  )
140
134
 
141
- def _parse_ooxml(self, path: Path, reporter: ProgressReporter) -> ParsedDocumentResult:
135
+ def _parse_ooxml(self, path: Path) -> ParsedDocumentResult:
142
136
  try:
143
137
  from langparse.workbooks.adapters import OOXMLWorkbookAdapter
144
138
  from langparse.workbooks.assembly import assemble_baseline, assemble_workbook
@@ -152,9 +146,7 @@ class ExcelParser(BaseParser):
152
146
  "Install with `pip install langparse[excel]`."
153
147
  ) from None
154
148
 
155
- reporter.emit("extracting")
156
149
  snapshot = OOXMLWorkbookAdapter().snapshot(path)
157
- reporter.emit("assembling")
158
150
  try:
159
151
  structure, diagnostics = assemble_workbook(
160
152
  snapshot,
@@ -168,10 +160,6 @@ class ExcelParser(BaseParser):
168
160
  diagnostics.warnings.append(
169
161
  f"Semantic workbook assembly failed; retained raw-grid fallback: {type(exc).__name__}"
170
162
  )
171
- from langparse.workbooks.objects import attach_objects
172
-
173
- attach_objects(snapshot, structure, diagnostics)
174
- reporter.emit("rendering")
175
163
  pages = compatibility_pages(snapshot, structure)
176
164
  markdown = render_workbook_markdown(snapshot, structure)
177
165
  return ParsedDocumentResult(
File without changes