bantamkit 0.29.0__tar.gz → 0.29.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (199) hide show
  1. {bantamkit-0.29.0 → bantamkit-0.29.2}/PKG-INFO +2 -2
  2. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/__init__.py +1 -1
  3. bantamkit-0.29.2/src/bantamkit/docmanifest.py +99 -0
  4. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/docread.py +334 -25
  5. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/evalrun.py +17 -21
  6. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/mcpserver.py +118 -25
  7. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/__main__.py +25 -0
  8. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/store.py +67 -2
  9. bantamkit-0.29.2/tests/data/platform-assumption-baseline.json +38 -0
  10. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/docread_fixtures.py +15 -1
  11. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_docread.py +69 -3
  12. bantamkit-0.29.2/tests/test_docread_ceilings.py +675 -0
  13. bantamkit-0.29.2/tests/test_document_manifest_parity.py +401 -0
  14. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_eventlog.py +15 -4
  15. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_hostinstall.py +67 -0
  16. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory.py +294 -1
  17. bantamkit-0.29.2/tests/test_platform_assumption_gate.py +268 -0
  18. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_served_tool_count_records.py +112 -3
  19. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_statusline.py +38 -6
  20. {bantamkit-0.29.0 → bantamkit-0.29.2}/.gitignore +0 -0
  21. {bantamkit-0.29.0 → bantamkit-0.29.2}/README.md +0 -0
  22. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/contracts/default.yaml +0 -0
  23. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/manifest.yaml +0 -0
  24. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/HISTORY.md +0 -0
  25. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/README.md +0 -0
  26. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/docs/architecture.md +0 -0
  27. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/docs/runbook.md +0 -0
  28. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/issues/142-settlement-timeout.md +0 -0
  29. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/patches/0009-retry-budget.patch +0 -0
  30. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/__init__.py +0 -0
  31. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/config.py +0 -0
  32. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/errors.py +0 -0
  33. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/posting.py +0 -0
  34. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/registry.py +0 -0
  35. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/report.py +0 -0
  36. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/retry.py +0 -0
  37. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/settle.py +0 -0
  38. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/validate.py +0 -0
  39. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/tests/test_posting.py +0 -0
  40. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/tests/test_settle.py +0 -0
  41. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-error-contract.yaml +0 -0
  42. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-handler-map.yaml +0 -0
  43. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-patch-before-after.yaml +0 -0
  44. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-retry-attempts.yaml +0 -0
  45. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-settlement-config.yaml +0 -0
  46. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-symbol-home.yaml +0 -0
  47. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-trace-blame.yaml +0 -0
  48. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-unread-key.yaml +0 -0
  49. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-137.yaml +0 -0
  50. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-359.yaml +0 -0
  51. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-372.yaml +0 -0
  52. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-11764.yaml +0 -0
  53. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-4137.yaml +0 -0
  54. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-8022.yaml +0 -0
  55. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-137.yaml +0 -0
  56. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-261.yaml +0 -0
  57. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-388.yaml +0 -0
  58. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/fixtures/.gitkeep +0 -0
  59. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/fixtures/catalog.json +0 -0
  60. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/perturbations/task-completion.yaml +0 -0
  61. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/.gitkeep +0 -0
  62. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-contact.yaml +0 -0
  63. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-invoice.yaml +0 -0
  64. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-order.yaml +0 -0
  65. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-schedule.yaml +0 -0
  66. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-versions.yaml +0 -0
  67. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/nav-prod-port.yaml +0 -0
  68. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/nav-release-bundle.yaml +0 -0
  69. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-audit-retention.yaml +0 -0
  70. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-cache-ttl.yaml +0 -0
  71. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-db-port.yaml +0 -0
  72. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-deploy.yaml +0 -0
  73. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-env-endpoint.yaml +0 -0
  74. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-oncall-rotation.yaml +0 -0
  75. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-oncall.yaml +0 -0
  76. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-org-quota.yaml +0 -0
  77. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-owner.yaml +0 -0
  78. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-basket-total.yaml +0 -0
  79. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-cheapest.yaml +0 -0
  80. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-compare.yaml +0 -0
  81. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-gadget-value.yaml +0 -0
  82. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-stock-total.yaml +0 -0
  83. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-total.yaml +0 -0
  84. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/profiles/default.yaml +0 -0
  85. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/profiles/patient.yaml +0 -0
  86. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/.gitkeep +0 -0
  87. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/code-quality.yaml +0 -0
  88. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/grounded-completion.yaml +0 -0
  89. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/task-completion.yaml +0 -0
  90. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/schemas/shiftwork-checkpoint.json +0 -0
  91. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/.gitkeep +0 -0
  92. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/file-graph.md +0 -0
  93. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/memory.md +0 -0
  94. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/.gitkeep +0 -0
  95. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/bantamkit_read.json +0 -0
  96. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/bantamkit_status.json +0 -0
  97. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/build_identity.json +0 -0
  98. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/document_list.json +0 -0
  99. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/document_read.json +0 -0
  100. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/file_graph.json +0 -0
  101. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_compact.json +0 -0
  102. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_recall.json +0 -0
  103. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_save.json +0 -0
  104. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_clock_in.json +0 -0
  105. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_clock_out.json +0 -0
  106. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_status.json +0 -0
  107. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/skill_audit.json +0 -0
  108. {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/validate_json.json +0 -0
  109. {bantamkit-0.29.0 → bantamkit-0.29.2}/hatch_build.py +0 -0
  110. {bantamkit-0.29.0 → bantamkit-0.29.2}/pyproject.toml +0 -0
  111. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/agent.py +0 -0
  112. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/assets.py +0 -0
  113. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/budget.py +0 -0
  114. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/client.py +0 -0
  115. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/contract.py +0 -0
  116. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/criticreplay.py +0 -0
  117. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/critique.py +0 -0
  118. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/eventlog.py +0 -0
  119. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/filegraph.py +0 -0
  120. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/hostinstall.py +0 -0
  121. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/loopguard.py +0 -0
  122. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/mcpreport.py +0 -0
  123. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/__init__.py +0 -0
  124. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/component.py +0 -0
  125. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/divergence.py +0 -0
  126. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/layers.py +0 -0
  127. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/pdfread.py +0 -0
  128. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/profile.py +0 -0
  129. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/shiftwork.py +0 -0
  130. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/skillaudit.py +0 -0
  131. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/statusline.py +0 -0
  132. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/structured.py +0 -0
  133. {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/textutil.py +0 -0
  134. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/cli_exit_status_probe.py +0 -0
  135. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/conftest.py +0 -0
  136. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/bad-crc.docx +0 -0
  137. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/charref-4301-digits.html +0 -0
  138. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/charset-table.json +0 -0
  139. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/compression-method-9.docx +0 -0
  140. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/corrupt-deflate.docx +0 -0
  141. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/encrypted-member.docx +0 -0
  142. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/encrypted-mimetype.odt +0 -0
  143. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/eszett-cell-ref.xlsx +0 -0
  144. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/internal-dtd-entity.docx +0 -0
  145. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/rfc2231-charset.eml +0 -0
  146. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/rfc822-nested-twice.eml +0 -0
  147. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/unicode-digit-shared-string.xlsx +0 -0
  148. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/x-uuencode.eml +0 -0
  149. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/f8404ab-perturbation-baseline.json +0 -0
  150. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/served-tool-surface.json +0 -0
  151. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/perturbation_baseline_harness.py +0 -0
  152. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/rbp16_effect_probe.py +0 -0
  153. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/rbp18_payload_probe.py +0 -0
  154. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_adapter.py +0 -0
  155. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_agent.py +0 -0
  156. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_amendguard.py +0 -0
  157. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_bantamkit_read_tool.py +0 -0
  158. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_budget.py +0 -0
  159. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_build_identity.py +0 -0
  160. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_client.py +0 -0
  161. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_compaction_corpus_survey.py +0 -0
  162. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_conformance.py +0 -0
  163. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_contract_fanout.py +0 -0
  164. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_criticreplay.py +0 -0
  165. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_critique.py +0 -0
  166. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_doc_commands_gate.py +0 -0
  167. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_setup.py +0 -0
  168. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_tasks.py +0 -0
  169. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_tools.py +0 -0
  170. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_encoding_gate.py +0 -0
  171. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_evalrun.py +0 -0
  172. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_field_program_gates.py +0 -0
  173. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_field_programs.py +0 -0
  174. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_filegraph.py +0 -0
  175. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_ladder_statistics.py +0 -0
  176. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_launcher_which.py +0 -0
  177. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_layers.py +0 -0
  178. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_loopguard.py +0 -0
  179. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcp_endpoint.py +0 -0
  180. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpdrift.py +0 -0
  181. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpreport.py +0 -0
  182. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpserver.py +0 -0
  183. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_compact_tool.py +0 -0
  184. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_component.py +0 -0
  185. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_divergence.py +0 -0
  186. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_layers.py +0 -0
  187. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_store_tripwire.py +0 -0
  188. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mutmatrix.py +0 -0
  189. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_newline_gate.py +0 -0
  190. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_packaging.py +0 -0
  191. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_pdfread.py +0 -0
  192. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_pinharness_ledger.py +0 -0
  193. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_shiftwork.py +0 -0
  194. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_skillaudit.py +0 -0
  195. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_status_surface.py +0 -0
  196. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_structured.py +0 -0
  197. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_thread_exception_gate.py +0 -0
  198. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_tool_manifest.py +0 -0
  199. {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_version_agreement.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.5
1
+ Metadata-Version: 2.4
2
2
  Name: bantamkit
3
- Version: 0.29.0
3
+ Version: 0.29.2
4
4
  Summary: bantamweight tooling — harness primitives that lift small-model agents
5
5
  License-Expression: MIT
6
6
  Requires-Python: >=3.11
@@ -29,4 +29,4 @@ from bantamkit.structured import StructuredOutputError, extract_json, structured
29
29
  # file as its dynamic version source, so the wheel's metadata and the string the MCP
30
30
  # server advertises are the same committed bytes, and neither is a function of when
31
31
  # someone last ran `pip`.
32
- __version__ = "0.29.0"
32
+ __version__ = "0.29.2"
@@ -0,0 +1,99 @@
1
+ """One rendering of a `docread.Document` as a manifest — for every caller that has one.
2
+
3
+ Register entries (b) and (h), `docs/roadmap-toolbox.md` row 8. The entry dict
4
+ `{document, kind, index, part, row_count, rows, omissions}` that `contract.document_manifest`
5
+ takes was built by hand in `evalrun._document_tools` and again in `mcpserver.bantamkit_read`,
6
+ with no shared helper and no test that the two produced the same bytes for the same file. They
7
+ did not: the eval harness answered a zero-row part with `document_offset_past_end`, whose
8
+ sentence reads `numbered 0 to -1`, while the MCP server answered `"<part>" in <path> has no
9
+ rows`. One runtime disagreeing with itself about a file is not a two-runtime divergence and
10
+ costs no `docs/porting.md` row — it just had to stop.
11
+
12
+ **Why this is its own module and not a function on either caller.** `evalrun` is the
13
+ measurement harness and `mcpserver` is Layer 5; making either import the other to reach a
14
+ shared renderer would put the eval harness behind the MCP SDK's import, or the server behind
15
+ `yaml` and the whole eval config surface. `docread` itself is the other candidate and is the
16
+ better one on paper — but the entry dict is the *contract layer's* input shape, not the
17
+ reader's, and `docread` is deliberately ignorant of who renders it. So the adapter between the
18
+ two sits here: it imports `docread` for nothing at all (it reads attributes off whatever it is
19
+ handed) and `contract` for the sentences, which is the direction the layer rule allows.
20
+
21
+ The sentences are `contract`'s, with ONE exception this module inherited rather than chose:
22
+ `document_no_rows` builds its text here, exactly as `mcpserver` built it inline before. It is
23
+ not in `assets/contracts/default.yaml` beside every other model-facing sentence in this
24
+ repository, and the port spells it a third time in `runtime-ts/src/mcp/server.ts`. Moving it
25
+ into the contract asset is a two-runtime contract change and is registered, not done here.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ from collections.abc import Iterable
31
+ from typing import Any
32
+
33
+ from bantamkit.contract import document_error, document_manifest
34
+
35
+ __all__ = [
36
+ "document_no_rows",
37
+ "manifest_entries",
38
+ "package_entries",
39
+ "render_manifest",
40
+ ]
41
+
42
+
43
+ def manifest_entries(document: str, doc: Any) -> list[dict]:
44
+ """The PART-grain entries for one document, keyed by the label the caller uses.
45
+
46
+ `document` is the caller's name for the file — a path on the MCP surface, a fixture name
47
+ in the eval harness — and it is the only thing about this rendering that legitimately
48
+ differs between the two. Everything else is a fact about the bytes.
49
+ """
50
+ return [
51
+ {
52
+ "document": document,
53
+ "kind": doc.kind,
54
+ "index": part.index,
55
+ "part": part.name,
56
+ "row_count": part.row_count,
57
+ "rows": part.rows,
58
+ "omissions": [o.as_dict() for o in part.omissions],
59
+ }
60
+ for part in doc.parts
61
+ ]
62
+
63
+
64
+ def package_entries(document: str, doc: Any) -> list[dict]:
65
+ """The DOCUMENT-grain entries: what the container holds that belongs to no part.
66
+
67
+ Empty when the document has no such omissions, because `contract.document_manifest`
68
+ renders an entry without them byte-for-byte as it did before the disclosure existed —
69
+ which is what keeps the committed `document-read` rows where they are.
70
+ """
71
+ if not doc.omissions:
72
+ return []
73
+ return [{"document": document, "omissions": [o.as_dict() for o in doc.omissions]}]
74
+
75
+
76
+ def render_manifest(documents: Iterable[tuple[str, Any]]) -> str:
77
+ """The manifest for one or many (label, `Document`) pairs, as one observation."""
78
+ parts: list[dict] = []
79
+ package: list[dict] = []
80
+ for document, doc in documents:
81
+ parts.extend(manifest_entries(document, doc))
82
+ package.extend(package_entries(document, doc))
83
+ return document_manifest(parts, package)
84
+
85
+
86
+ def document_no_rows(part: str, document: str) -> str:
87
+ """The refusal for a part that has no rows at all — NOT an offset past the end.
88
+
89
+ `document_offset_past_end` over a zero-row part prints `numbered 0 to -1`, a range with
90
+ no members, and it says "offset N is past the end" about an offset of 0, which is not
91
+ past anything. No offset can be in range here, so the reply states that fact instead.
92
+
93
+ This exact string is pinned ON THE WIRE for both runtimes by
94
+ `tools/conformance/suites/wire.mjs` (`read: id 12`), which asserts it as a literal on each
95
+ side precisely so that both sides drifting back together would still fail. Changing it is
96
+ a two-runtime change plus that case; it is why this is the sentence that survived (b)/(h)
97
+ rather than the eval harness's.
98
+ """
99
+ return document_error(f'"{part}" in {document} has no rows')
@@ -210,15 +210,37 @@ OMIT_UNMAPPED = "unmapped-text"
210
210
  # votes on a 4096-byte head, so `extract` routinely meets bytes that verdict never saw, and
211
211
  # those bytes can contradict it. The prefix that IS text is content; the rest is COUNTED.
212
212
  OMIT_UNREAD_TAIL = "unread-tail"
213
- # Bytes past `TEXT_MAX_BYTES`, the ceiling this reader states for one plain-text file. Not a
214
- # property of the file -- a limit this reader imposes -- which is exactly why it is declared
215
- # as a count instead of applied in silence.
213
+ # What a ceiling THIS READER imposes kept out of the rows. Not a property of the file, which
214
+ # is exactly why it is declared as a count instead of applied in silence. Four ceilings carry
215
+ # it, and each one names its own number in `what`: bytes past `TEXT_MAX_BYTES` for a plain-text
216
+ # file, for the markup of an `.html`/`.mhtml`, and for what `textutil` converted a `.doc` or
217
+ # an `.rtf` into — the same constant three times, because "how much of a file this reader
218
+ # reads" is one number and not one per container — and rows past `XLSX_MAX_TEXT_BYTES` for a
219
+ # workbook, where the thing that runs away is the RENDERING rather than the file.
220
+ #
221
+ # The principle overclaimed until review round 5 (H3), and the correction is worth stating
222
+ # because the sentence is what stopped anyone looking: `.doc` and `.rtf` had NO ceiling of any
223
+ # kind — `extract_textutil` rendered every byte of the converter's stdout — and `.docx` had
224
+ # none either, measured at 40,000,000 bytes of text out of a 181,289-byte file, 2.38x
225
+ # `TEXT_MAX_BYTES`, with both omission tuples empty.
226
+ #
227
+ # The fifth ceiling is deliberately NOT one of these, and that is the honest amendment rather
228
+ # than a fifth token: `ZIP_MEMBER_MAX_BYTES` refuses instead of disclosing, because half an
229
+ # XML member is not a smaller XML member. So an OOXML container is bounded by what this reader
230
+ # will PARSE and says so by refusing; every other container is bounded by what it will READ
231
+ # and says so by counting. `.docx` needs no rendering budget on top of that: the text a
232
+ # `<w:t>` walk produces can never exceed the bytes of the part it walked.
216
233
  OMIT_SIZE_CAP = "size-cap"
217
234
  # A worksheet cell whose own `r` reference could not place it: not letters-then-digits, or a
218
235
  # column past the last one the format has. The cell's TEXT is in the rows, at its XML
219
236
  # position; what the rows do not carry is the column the file asked for. Counted per reason,
220
237
  # with the columns it landed in, exactly as `number-format` is counted per format code.
221
238
  OMIT_UNPLACED_CELL = "unplaced-cell"
239
+ # A worksheet cell a LATER cell in the same row and column replaced. Last-wins is what both
240
+ # runtimes do and what a writer's own second cell means, so the reading is kept; what was not
241
+ # kept was the disclosure. Counted with the columns it happened in, like every other cell this
242
+ # reader could not put in the rows.
243
+ OMIT_DUPLICATE_CELL = "duplicate-cell"
222
244
 
223
245
 
224
246
  class DocumentReadError(BantamError):
@@ -621,6 +643,33 @@ _UNREADABLE_OPTIONAL = (
621
643
  ) # fmt: skip
622
644
 
623
645
 
646
+ # How much ONE MEMBER of a zip this reader will decompress and parse. The ceiling on the
647
+ # PARSE, which is a different door from `XLSX_MAX_TEXT_BYTES`: that one bounds the text a
648
+ # workbook renders, and it never stands in front of this, because `_read` decompresses the
649
+ # member whole and `_parse` builds a tree from it before the first row is rendered.
650
+ #
651
+ # MEASURED 2026-09-06 on this machine, through `docread.extract` at the shipped constants,
652
+ # on a sheet of N `<row><c t="inlineStr"><is><t>x</t></is></c></row>` deflated at level 9:
653
+ # 400,000 rows are 58,097 bytes on disk and peaked at 342.2 MB (5,890x) with NO omission;
654
+ # 1,000,000 rows are 143,658 bytes and peaked at 838.8 MB (5,839x). Linear and unbounded —
655
+ # the memory is the `Element` tree, not the text, so a budget over the rendering could not
656
+ # see it. At this ceiling the same shape parses in 0.74 s and peaks at 264.3 MB, which is a
657
+ # worst case rather than no case at all.
658
+ #
659
+ # 16 MiB, the number `TEXT_MAX_BYTES` states, because "how much of a file this reader reads"
660
+ # is one number — but under its OWN NAME, because it bounds a member of a container and not a
661
+ # file on a disk, and a bar that varies one must not be varying the other. MEASURED the same
662
+ # day over `~/Downloads`, `~/Documents/Claude/Projects` and `~/Documents` (pruned as J25
663
+ # prunes them): 25 OOXML/ODF packages, the largest single XML member among them 4,283,286
664
+ # bytes — 3.9x under this — and the largest `word/document.xml` 159,976 bytes, 105x under it.
665
+ #
666
+ # A REFUSAL and not a truncation, which is the one place this module departs from
667
+ # "disclose, never truncate": half an XML document is not a smaller XML document, and a tree
668
+ # built from a severed member would carry text that is not what the file says. So the reader
669
+ # stops and names the member, the number the file declares and its own ceiling.
670
+ ZIP_MEMBER_MAX_BYTES = 16 * 1024 * 1024
671
+
672
+
624
673
  def _read(zf: zipfile.ZipFile, name: str, path: Path) -> bytes:
625
674
  """One member this reader cannot do without, or a refusal in the reader's own words.
626
675
 
@@ -642,8 +691,29 @@ def _read(zf: zipfile.ZipFile, name: str, path: Path) -> bytes:
642
691
  different facts about the file (a stored checksum that lies, a deflate stream that is
643
692
  not one), and it is the only part of the sentence the Node port cannot print from the
644
693
  same words — the ruling in docs/porting.md quotes it.
694
+
695
+ And a fifth, which is not zipfile's: a member that decompresses past
696
+ `ZIP_MEMBER_MAX_BYTES`. The gate is the UNCOMPRESSED SIZE the central directory declares,
697
+ read before anything is decompressed, and it is deliberately the cheapest possible check —
698
+ a member that says it is 46 MB costs no inflate at all to refuse.
699
+
700
+ Those are the attacker's bytes, so the question is what a lying declaration buys, and the
701
+ answer was MEASURED rather than assumed: `zipfile.ZipExtFile` clamps its own output to
702
+ `ZipInfo.file_size` and checks the CRC of what it produced, so a member declaring 10 bytes
703
+ while holding 100,000 yields 10 bytes and `BadZipFile: Bad CRC-32` — declaring LOW is a
704
+ damaged file, not a way past the ceiling, and declaring HIGH is what this refuses. That
705
+ clamp is CPython's and not the format's: the Node port walks the archive with its own
706
+ reader, and if that reader does not clamp it owes the property a ceiling on the real read.
707
+ The property is the contract; the mechanism is not.
645
708
  """
646
709
  try:
710
+ declared = zf.getinfo(name).file_size
711
+ if declared > ZIP_MEMBER_MAX_BYTES:
712
+ raise DocumentReadError(
713
+ f"{path.name} is a zip but its {name} declares {declared} bytes uncompressed, "
714
+ f"past the {ZIP_MEMBER_MAX_BYTES} bytes this reader parses, "
715
+ "so this reader cannot parse it"
716
+ )
647
717
  return zf.read(name)
648
718
  except KeyError:
649
719
  sample = ", ".join(sorted(zf.namelist())[:8]) or "(empty archive)"
@@ -866,6 +936,35 @@ _MAX_COLUMN_LETTERS = 3
866
936
  # into the manifest with no bound at all.
867
937
  UNPLACED_SHAPE = "the column of a cell whose reference is not letters then digits"
868
938
  UNPLACED_RANGE = "the column of a cell past XFD, the last column the format has"
939
+ # What a cell loses when a later cell in the same row claims its column. Fixed for the same
940
+ # reason the two above are: the reference is the file's bytes and the column letter is this
941
+ # reader's own, so only the letter travels — in `where`, bounded by `XLSX_MAX_COLUMNS`.
942
+ DUPLICATE_CELL = "the text of a cell a later cell in the same row and column replaced"
943
+
944
+ # How much text ONE WORKBOOK may materialise, all sheets together. `XLSX_MAX_COLUMNS` bounds
945
+ # a ROW and nothing bounded the document, which is a ceiling with a hole in it: the width is
946
+ # bought a row at a time. MEASURED 2026-09-06 on both runtimes: 20,000 rows each holding one
947
+ # `XFD1` cell deflate to 53,967 bytes and materialise 327,680,000 bytes in 9.02 s — 6,072x,
948
+ # out of a file small enough to mail.
949
+ #
950
+ # 16 MiB, the same figure `TEXT_MAX_BYTES` carries and for the same kind of reason, but under
951
+ # its OWN NAME because the two bound different things: bytes read off a disk there, bytes
952
+ # rendered out of a container here, and a bar that varies one must not be varying the other.
953
+ # MEASURED 2026-09-06 over `~/Downloads`, `~/Documents/Claude/Projects` and `~/Documents`
954
+ # (pruned as J25 prunes them): 15 `.xlsx`, the largest 8,664,227 bytes on disk, and the
955
+ # largest RENDERING among them 754,520 bytes — 22x under this budget, so no real workbook on
956
+ # this machine meets it.
957
+ #
958
+ # It bounds the RENDERING and nothing else. The sentence that stood here said it bounded the
959
+ # rendering "because the file is already bounded", and that was FALSE for the whole life of
960
+ # this constant: `_read` decompressed a member whole and `_parse` built a tree from it, both
961
+ # before the first row was rendered and both outside this budget, so 58,097 bytes on disk
962
+ # peaked at 342.2 MB with no omission and this ceiling never saw it (review round 5, H1).
963
+ # What bounds the file is `ZIP_MEMBER_MAX_BYTES`, one door earlier; this bounds what comes out
964
+ # of it. The shortfall is disclosed as `OMIT_SIZE_CAP` counting the rows no sheet rendered: a
965
+ # row is what a caller addresses, and a byte count of text that was never built would be a
966
+ # number this reader cannot honestly produce.
967
+ XLSX_MAX_TEXT_BYTES = 16 * 1024 * 1024
869
968
 
870
969
 
871
970
  def _column(ref: str | None, fallback: int) -> tuple[int, str]:
@@ -947,8 +1046,29 @@ def _cell_text(cell: ET.Element, shared: list[str]) -> str:
947
1046
  return raw # numbers, cached formula strings (`str`), errors (`e`): stored form, verbatim
948
1047
 
949
1048
 
1049
+ @dataclass
1050
+ class _TextBudget:
1051
+ """How much rendered text one WORKBOOK may still materialise, and what it cost to stop.
1052
+
1053
+ One of these is made per `extract_xlsx` call and handed to every sheet, which is the whole
1054
+ point: a budget made per sheet would let an N-sheet workbook materialise N budgets, and the
1055
+ input this ceiling exists for is one sheet of 20,000 rows anyway.
1056
+
1057
+ `total` counts every `<row>` the document declares, rendered or not, because the omission
1058
+ has to say what the rows it did render are a fraction OF. Counting them costs a walk of
1059
+ XML that is already parsed and bounded by the file; rendering them is what does not.
1060
+ """
1061
+
1062
+ remaining: int
1063
+ total: int = 0
1064
+ dropped: int = 0
1065
+
1066
+
950
1067
  def _sheet_rows(
951
- root: ET.Element, shared: list[str], date_styles: tuple[str, ...] = ()
1068
+ root: ET.Element,
1069
+ shared: list[str],
1070
+ date_styles: tuple[str, ...] = (),
1071
+ budget: _TextBudget | None = None,
952
1072
  ) -> tuple[tuple[str, ...], tuple[Omission, ...]]:
953
1073
  """The rendered rows of one sheet, and a count of what the rendering did not carry.
954
1074
 
@@ -959,19 +1079,40 @@ def _sheet_rows(
959
1079
  A cell `_column` cannot place is one of those omissions and NOT a refusal (review round 4,
960
1080
  M2): it keeps its text at its XML position and loses only the column the file asked for.
961
1081
  Counted per reason and rendered after the format codes, so the order of this tuple is
962
- blank rows, then number formats by code, then unplaced cells by reason.
1082
+ blank rows, then number formats by code, then unplaced cells by reason, then the cells a
1083
+ later cell in the same row and column replaced.
1084
+
1085
+ A DUPLICATE is a cell whose column already holds text from a cell earlier in the same row.
1086
+ Last-wins is kept — it is what both runtimes do and what a writer's own second cell means
1087
+ — and the earlier cell's text is disclosed rather than dropped in silence, which is the
1088
+ only part of that behaviour nobody chose.
1089
+
1090
+ `budget` is the DOCUMENT's, not this sheet's: the width of one row is already bounded by
1091
+ `XLSX_MAX_COLUMNS` and the height of a workbook was not, so the row is where the ceiling
1092
+ has to bite. A row is skipped whole rather than cut in half — half a row is a row this
1093
+ reader cannot vouch for, which is the same rule `extract_text` follows at its own cap —
1094
+ so the overshoot is at most one row, itself bounded at `XLSX_MAX_COLUMNS` fields.
963
1095
  """
1096
+ if budget is None:
1097
+ budget = _TextBudget(XLSX_MAX_TEXT_BYTES)
964
1098
  rows = []
965
1099
  blank = 0
966
1100
  dated: dict[str, dict[int, int]] = {}
967
1101
  unplaced: dict[str, dict[int, int]] = {}
1102
+ duplicated: dict[int, int] = {}
968
1103
  for row in root.iter(NS_S + "row"):
1104
+ budget.total += 1
1105
+ if budget.remaining <= 0:
1106
+ budget.dropped += 1
1107
+ continue
969
1108
  cells: dict[int, str] = {}
970
1109
  for position, cell in enumerate(row.iter(NS_S + "c")):
971
1110
  text = _clean(_cell_text(cell, shared))
972
1111
  if not text:
973
1112
  continue
974
1113
  column, unplaceable = _column(cell.get("r"), position)
1114
+ if column in cells:
1115
+ duplicated[column] = duplicated.get(column, 0) + 1
975
1116
  cells[column] = text
976
1117
  if unplaceable:
977
1118
  unplaced.setdefault(unplaceable, {})
@@ -986,6 +1127,7 @@ def _sheet_rows(
986
1127
  dated[code][column] = dated[code].get(column, 0) + 1
987
1128
  width = max(cells) + 1 if cells else 0
988
1129
  line = "\t".join(cells.get(i, "") for i in range(width))
1130
+ budget.remaining -= len(line.encode())
989
1131
  if not line:
990
1132
  blank += 1
991
1133
  rows.append(line)
@@ -1012,6 +1154,15 @@ def _sheet_rows(
1012
1154
  what=reason,
1013
1155
  )
1014
1156
  )
1157
+ if duplicated:
1158
+ omissions.append(
1159
+ Omission(
1160
+ OMIT_DUPLICATE_CELL,
1161
+ sum(duplicated.values()),
1162
+ where=tuple(_letter(c) for c in sorted(duplicated)),
1163
+ what=DUPLICATE_CELL,
1164
+ )
1165
+ )
1015
1166
  return tuple(rows), tuple(omissions)
1016
1167
 
1017
1168
 
@@ -1051,7 +1202,15 @@ def _media_omission(media: dict[str, int]) -> tuple[Omission, ...]:
1051
1202
 
1052
1203
 
1053
1204
  def extract_xlsx(path: str | Path) -> Document:
1205
+ """Every declared sheet, under ONE `XLSX_MAX_TEXT_BYTES` budget for the whole workbook.
1206
+
1207
+ The budget is the document's, so it is made here and not in `_sheet_rows`, and the
1208
+ shortfall is disclosed at the document's grain for the same reason — it is not a fact
1209
+ about the sheet the budget happened to run out on. The cap is stated BEFORE the media
1210
+ tally, because the media a reader met is a count of what it met underneath the cap.
1211
+ """
1054
1212
  path = Path(path)
1213
+ budget = _TextBudget(XLSX_MAX_TEXT_BYTES)
1055
1214
  with _open(path) as zf:
1056
1215
  shared = _shared_strings(zf, path)
1057
1216
  date_styles = _date_formats(zf)
@@ -1061,10 +1220,20 @@ def extract_xlsx(path: str | Path) -> Document:
1061
1220
  if target is None:
1062
1221
  raise DocumentReadError(f"sheet {name!r} has no resolvable worksheet part")
1063
1222
  sheet = _parse(_read(zf, target, path), target, path)
1064
- rows, omissions = _sheet_rows(sheet, shared, date_styles)
1223
+ rows, omissions = _sheet_rows(sheet, shared, date_styles, budget)
1065
1224
  omissions = _media_omission(_anchored_media(zf, target, media)) + omissions
1066
1225
  parts.append(Part(name=name, index=index, rows=rows, omissions=omissions))
1067
- return Document(kind="xlsx", parts=tuple(parts), omissions=_media_omission(media))
1226
+ capped: tuple[Omission, ...] = ()
1227
+ if budget.dropped:
1228
+ capped = (
1229
+ Omission(
1230
+ OMIT_SIZE_CAP,
1231
+ budget.dropped,
1232
+ what=f"{budget.total} rows in this workbook; this reader renders "
1233
+ f"{XLSX_MAX_TEXT_BYTES} bytes of cell text",
1234
+ ),
1235
+ )
1236
+ return Document(kind="xlsx", parts=tuple(parts), omissions=capped + _media_omission(media))
1068
1237
 
1069
1238
 
1070
1239
  def extract_docx(path: str | Path) -> Document:
@@ -1136,22 +1305,50 @@ def _cap_charrefs(text: str) -> str:
1136
1305
 
1137
1306
 
1138
1307
  def _with_bounded_unescape(name: str):
1139
- """`HTMLParser.<name>`, as the library wrote it, with `unescape` bound to the capped one.
1308
+ """`HTMLParser.<name>` with `unescape` bound to the capped one, or `None` if it cannot be.
1140
1309
 
1141
1310
  The two methods that call `unescape` (`goahead` on text, `parse_starttag` on attribute
1142
1311
  values) look it up in `html.parser`'s globals; a copy of the code object with one entry
1143
1312
  of that namespace replaced is the same loop calling the same helpers, and nothing else
1144
1313
  in the process — no other parser, not `html.unescape` itself — sees the change.
1314
+
1315
+ CONTAINED, `docs/roadmap-toolbox.md` row 8 entry (u). This depends on three properties of
1316
+ a CPython private method at once — the name existing, `unescape` resolving as a MODULE
1317
+ GLOBAL, and no closure — and it is called in a class body, so before this guard a single
1318
+ changed property raised `AttributeError` (or `TypeError`) at MODULE IMPORT and the MCP
1319
+ server did not start. `pyproject.toml` declares `requires-python = ">=3.11"` and CI
1320
+ measures 3.11 and 3.12, so every interpreter from 3.13 up is permitted and none is
1321
+ measured; an interpreter this package says it supports may not be able to make it fail
1322
+ to import. Measured on this machine's CPython 3.12.13: `goahead` has `'unescape' in
1323
+ co_names` True and `co_freevars ()`, `parse_starttag` the same.
1324
+
1325
+ All three are checked rather than caught, because the failure that is NOT an exception is
1326
+ the dangerous one: a method that resolves `unescape` some other way would take this
1327
+ rebinding silently and go on calling the uncapped `html.unescape`. `None` here is what
1328
+ `html_rows` reads to fall back, and `HTML_UNESCAPE_BOUNDED` is what makes that visible.
1145
1329
  """
1146
- method = getattr(html.parser.HTMLParser, name)
1330
+ method = getattr(html.parser.HTMLParser, name, None)
1331
+ code = getattr(method, "__code__", None)
1332
+ if code is None or "unescape" not in code.co_names or code.co_freevars:
1333
+ return None
1147
1334
  return types.FunctionType(
1148
- method.__code__,
1335
+ code,
1149
1336
  {**vars(html.parser), "unescape": lambda text: html.unescape(_cap_charrefs(text))},
1150
1337
  name,
1151
1338
  method.__defaults__,
1152
1339
  )
1153
1340
 
1154
1341
 
1342
+ # The rebindings that could be built on THIS interpreter, and whether both of them could.
1343
+ # `html_rows` reads the flag, a test can force it, and nothing about the fallback is silent.
1344
+ _BOUNDED_UNESCAPE = {
1345
+ name: bound
1346
+ for name in ("goahead", "parse_starttag")
1347
+ if (bound := _with_bounded_unescape(name)) is not None
1348
+ }
1349
+ HTML_UNESCAPE_BOUNDED = len(_BOUNDED_UNESCAPE) == 2
1350
+
1351
+
1155
1352
  class _HtmlText(html.parser.HTMLParser):
1156
1353
  """HTML to rows. Block markup ends a row, `<td>`/`<th>` separate fields with a tab.
1157
1354
 
@@ -1171,8 +1368,9 @@ class _HtmlText(html.parser.HTMLParser):
1171
1368
  # `handle_entityref` changes the library's chunking — `&#65b` is handed over as `&#`
1172
1369
  # and `65b`, and an `&#` with no `;` anywhere after it makes the parser emit the rest
1173
1370
  # of the document, tags included, as data at `close()`.
1174
- goahead = _with_bounded_unescape("goahead")
1175
- parse_starttag = _with_bounded_unescape("parse_starttag") # attribute values, likewise
1371
+ # The rebindings are attached AFTER the class statement, from `_BOUNDED_UNESCAPE`, so an
1372
+ # interpreter that has neither method still produces a class — see `_with_bounded_unescape`
1373
+ # and `html_rows` for what happens then.
1176
1374
 
1177
1375
  def __init__(self) -> None:
1178
1376
  super().__init__(convert_charrefs=True)
@@ -1218,8 +1416,22 @@ class _HtmlText(html.parser.HTMLParser):
1218
1416
  self._flush()
1219
1417
 
1220
1418
 
1419
+ for _name, _bound in _BOUNDED_UNESCAPE.items():
1420
+ setattr(_HtmlText, _name, _bound)
1421
+
1422
+
1221
1423
  def html_rows(markup: str) -> tuple[str, ...]:
1222
- """Rendered rows of one HTML fragment. A pure function of the string: no I/O, no host."""
1424
+ """Rendered rows of one HTML fragment. A pure function of the string: no I/O, no host.
1425
+
1426
+ When the per-chunk binding could not be built on this interpreter, the WHOLE MARKUP is
1427
+ capped first — the pre-H1 shape, kept as the fallback. It is measurably worse and it is
1428
+ measurably not a crash: H1 refused it as the default because `&#<4301 digits>;` inside
1429
+ `<xmp>` is CDATA the parser never unescapes, so this rewrite turns it into U+FFFD where
1430
+ the bounded binding keeps the digits. A reader that answers slightly differently beats a
1431
+ package that will not import, and the difference is one a test can see.
1432
+ """
1433
+ if not HTML_UNESCAPE_BOUNDED:
1434
+ markup = _cap_charrefs(markup)
1223
1435
  parser = _HtmlText()
1224
1436
  parser.feed(markup)
1225
1437
  parser.close()
@@ -1251,6 +1463,30 @@ def _decoded_body(part: email.message.Message) -> str:
1251
1463
  return payload.decode("utf-8", errors="replace")
1252
1464
 
1253
1465
 
1466
+ def _read_to_ceiling(path: Path) -> tuple[bytes, tuple[Omission, ...]]:
1467
+ """A file's bytes up to `TEXT_MAX_BYTES`, and the `OMIT_SIZE_CAP` for what is past it.
1468
+
1469
+ ONE ceiling and one sentence for every container that holds markup, because "how much of
1470
+ a file this reader reads" is a fact about the reader and not about the suffix: a 1 GB
1471
+ `.txt` stopped at `TEXT_MAX_BYTES` and counted the rest while a 1 GB `.html` was held
1472
+ whole (`docs/roadmap-toolbox.md` row 8, entry (k)). The sentence is `extract_text`'s,
1473
+ to the byte, so a caller cannot tell from it which reader hit the cap.
1474
+
1475
+ Where `extract_text` also cuts back to the last line break, this does not: markup is not
1476
+ a line-oriented format, a half-open tag is not a claim about content the way half a line
1477
+ is, and `html.parser` closes what the file left open without inventing text for it.
1478
+ """
1479
+ size = path.stat().st_size
1480
+ with path.open("rb") as handle:
1481
+ raw = handle.read(TEXT_MAX_BYTES + 1)
1482
+ if len(raw) <= TEXT_MAX_BYTES:
1483
+ return raw, ()
1484
+ raw = raw[:TEXT_MAX_BYTES]
1485
+ dropped = size - TEXT_MAX_BYTES
1486
+ what = f"{size} bytes on disk; this reader reads {TEXT_MAX_BYTES}"
1487
+ return raw, (Omission(OMIT_SIZE_CAP, dropped, size=dropped, what=what),)
1488
+
1489
+
1254
1490
  def extract_mhtml(path: str | Path) -> Document:
1255
1491
  """A MIME message / MHTML archive: every text part, in message order.
1256
1492
 
@@ -1261,8 +1497,12 @@ def extract_mhtml(path: str | Path) -> Document:
1261
1497
  `-format html` pass left `signature` split as `s= ignature`.
1262
1498
  """
1263
1499
  path = Path(path)
1264
- with path.open("rb") as handle:
1265
- message = email.message_from_binary_file(handle, policy=email.policy.default)
1500
+ # BOUNDED, entry (k): `message_from_binary_file` reads the handle to EOF, so a 1 GB
1501
+ # `.mht` was held whole — the same hole `extract_html` had, in its own spelling. The
1502
+ # message is parsed from the bytes this reader will admit to having read; a truncated
1503
+ # MIME message is one `email` still walks, and what it could not see is COUNTED.
1504
+ raw, capped = _read_to_ceiling(path)
1505
+ message = email.message_from_bytes(raw, policy=email.policy.default)
1266
1506
  bodies: list[tuple[str, str]] = []
1267
1507
  skipped: dict[str, int] = {}
1268
1508
  skipped_bytes = 0
@@ -1286,7 +1526,7 @@ def extract_mhtml(path: str | Path) -> Document:
1286
1526
  rows = html_rows(body) if subtype == "html" else _plain_rows(body)
1287
1527
  name = "document" if len(bodies) == 1 else f"part{index}"
1288
1528
  parts.append(Part(name=name, index=index, rows=rows))
1289
- omissions = ()
1529
+ omissions: tuple[Omission, ...] = ()
1290
1530
  if skipped:
1291
1531
  omissions = (
1292
1532
  Omission(
@@ -1296,18 +1536,29 @@ def extract_mhtml(path: str | Path) -> Document:
1296
1536
  what=", ".join(sorted(skipped)),
1297
1537
  ),
1298
1538
  )
1539
+ # The cap first: the media tally counts what this reader met UNDERNEATH it, so a caller
1540
+ # who reads that number without the cap above it has read a lower bound as a total.
1299
1541
  return _nonempty(
1300
- Document(kind="mhtml", parts=tuple(parts), omissions=omissions),
1542
+ Document(kind="mhtml", parts=tuple(parts), omissions=capped + omissions),
1301
1543
  path,
1302
1544
  "no text/html or text/plain part carried any text",
1303
1545
  )
1304
1546
 
1305
1547
 
1306
1548
  def extract_html(path: str | Path) -> Document:
1549
+ """Markup to rows, up to `TEXT_MAX_BYTES` of it, saying how much it did not read.
1550
+
1551
+ BOUNDED, entry (k): this did `path.read_bytes()`, so a 1 GB `.html` was materialised whole
1552
+ where a 1 GB `.txt` had stopped at the cap and counted the rest since J10.
1553
+ """
1307
1554
  path = Path(path)
1308
- raw = path.read_bytes()
1555
+ raw, capped = _read_to_ceiling(path)
1309
1556
  markup = raw.decode("utf-8", errors="replace")
1310
- doc = Document(kind="html", parts=(Part(name="document", index=0, rows=html_rows(markup)),))
1557
+ doc = Document(
1558
+ kind="html",
1559
+ parts=(Part(name="document", index=0, rows=html_rows(markup)),),
1560
+ omissions=capped,
1561
+ )
1311
1562
  return _nonempty(doc, path, "its markup carried no text outside script and style")
1312
1563
 
1313
1564
 
@@ -1321,13 +1572,45 @@ def _nonempty(doc: Document, path: Path, why: str) -> Document:
1321
1572
  `.xlsx` is deliberately exempt: a declared-but-empty sheet is a real part with no rows and
1322
1573
  J10's row counts are committed measurements. Here there is no such thing — a `.doc` that
1323
1574
  renders nothing is a `.doc` this reader did not read.
1575
+
1576
+ THE REFUSAL CARRIES THE OMISSIONS, review round 5 (H2). `extract_html` and `extract_mhtml`
1577
+ build the `Document` with `capped` in `omissions` and hand it here, and here it raises: the
1578
+ refusal kept the reader's verdict about the content and threw away the ceiling that
1579
+ produced that verdict. MEASURED on a 16,777,291-byte `.html` whose 16 MiB `<script>`
1580
+ comment is followed by one visible sentence — `its markup carried no text outside script
1581
+ and style ... it is not an empty document`, about a document that carries text, from a
1582
+ read that stopped 75 bytes short of it. The same on a `.mht` past the ceiling, where the
1583
+ media tally went with it.
1584
+
1585
+ A reader is allowed to refuse. It is not allowed to state a false fact about a file, and
1586
+ `why` is a fact about the file only when the whole file was read. Under a cap it is scoped
1587
+ to the part that was read and the unread bytes are named, so a caller who would have
1588
+ stopped looking has the one number that tells it not to.
1324
1589
  """
1325
1590
  if any(part.rows for part in doc.parts):
1326
1591
  return doc
1327
- raise DocumentReadError(
1328
- f"cannot read {path.name}: it is a {doc.kind} container but {why}, so this reader has "
1329
- "no text for it — it is not an empty document"
1330
- )
1592
+ capped = next((o for o in doc.omissions if o.subject == OMIT_SIZE_CAP), None)
1593
+ media = next((o for o in doc.omissions if o.subject == OMIT_MEDIA), None)
1594
+ if capped is None:
1595
+ said = (
1596
+ f"cannot read {path.name}: it is a {doc.kind} container but {why}, so this reader "
1597
+ "has no text for it — it is not an empty document"
1598
+ )
1599
+ else:
1600
+ # The cap's own `what` is quoted rather than rebuilt, so the sentence and the omission
1601
+ # can never state different numbers, and so the clause says whose bytes were counted:
1602
+ # a file's on disk here, `textutil`'s output there.
1603
+ said = (
1604
+ f"cannot read {path.name}: it is a {doc.kind} container but {why} in the part "
1605
+ f"this reader read ({capped.what}) — the {capped.count} bytes it did not read "
1606
+ "may carry text"
1607
+ )
1608
+ if media is not None:
1609
+ said += (
1610
+ f", and it holds {media.count} embedded part(s) ({media.what}) this reader "
1611
+ "renders no text for"
1612
+ )
1613
+ raise DocumentReadError(said)
1331
1614
 
1332
1615
 
1333
1616
  # --------------------------------------------------------------- plain text, in no container
@@ -1492,6 +1775,29 @@ def _textutil_type(exe: str, path: Path) -> str:
1492
1775
  return ""
1493
1776
 
1494
1777
 
1778
+ def _cap_converted(stdout: bytes) -> tuple[bytes, tuple[Omission, ...]]:
1779
+ """`textutil`'s output up to `TEXT_MAX_BYTES`, and the `OMIT_SIZE_CAP` for what is past it.
1780
+
1781
+ THE FOURTH CEILING, review round 5 (H3). `.doc` and `.rtf` had none at all: every byte the
1782
+ converter wrote was rendered, so the principle stated at `OMIT_SIZE_CAP` — "how much of a
1783
+ file this reader reads" is one number and not one per container — was contradicted two
1784
+ containers over by the same module.
1785
+
1786
+ The number is `TEXT_MAX_BYTES`, and `what` names WHOSE bytes were counted rather than
1787
+ borrowing `_read_to_ceiling`'s sentence: these are the converter's, not the file's, and a
1788
+ caller must not read this omission as a statement about the `.rtf` on disk. What is NOT
1789
+ bounded here is the memory: `_textutil_run` captures the whole of a subprocess's stdout
1790
+ before this sees a byte of it, so this bounds what the reader RENDERS and the host process
1791
+ still decides how much it wrote. Saying so is the point — a comment that claimed otherwise
1792
+ is exactly what (v) shipped.
1793
+ """
1794
+ if len(stdout) <= TEXT_MAX_BYTES:
1795
+ return stdout, ()
1796
+ dropped = len(stdout) - TEXT_MAX_BYTES
1797
+ what = f"{len(stdout)} bytes {TEXTUTIL} produced; this reader reads {TEXT_MAX_BYTES}"
1798
+ return stdout[:TEXT_MAX_BYTES], (Omission(OMIT_SIZE_CAP, dropped, size=dropped, what=what),)
1799
+
1800
+
1495
1801
  def extract_textutil(path: str | Path, kind: str = "doc") -> Document:
1496
1802
  """Convert through `textutil` and render its plain text as rows.
1497
1803
 
@@ -1519,8 +1825,11 @@ def extract_textutil(path: str | Path, kind: str = "doc") -> Document:
1519
1825
  "raw bytes re-encoded, not the document's text"
1520
1826
  )
1521
1827
  done = _textutil_run(exe, path, "-convert", "txt", "-stdout")
1522
- rows = _plain_rows(done.stdout.decode("utf-8", errors="replace"))
1523
- doc = Document(kind=kind, parts=(Part(name="document", index=0, rows=rows),))
1828
+ produced, capped = _cap_converted(done.stdout)
1829
+ rows = _plain_rows(produced.decode("utf-8", errors="replace"))
1830
+ doc = Document(
1831
+ kind=kind, parts=(Part(name="document", index=0, rows=rows),), omissions=capped
1832
+ )
1524
1833
  return _nonempty(doc, path, f"{TEXTUTIL} converted it to no text at all")
1525
1834
 
1526
1835