bantamkit 0.35.2__tar.gz → 0.35.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. {bantamkit-0.35.2 → bantamkit-0.35.3}/PKG-INFO +4 -4
  2. {bantamkit-0.35.2 → bantamkit-0.35.3}/README.md +2 -2
  3. bantamkit-0.35.3/_assets/tools/shiftwork_clock_out.json +150 -0
  4. bantamkit-0.35.3/_assets/tools/shiftwork_plan.json +25 -0
  5. {bantamkit-0.35.2 → bantamkit-0.35.3}/pyproject.toml +6 -1
  6. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/__init__.py +1 -1
  7. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/mcpserver.py +19 -0
  8. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/store.py +2 -2
  9. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/served-tool-surface.json +65 -10
  10. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_mcpserver.py +22 -0
  11. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_tool_manifest.py +48 -1
  12. bantamkit-0.35.2/_assets/tools/shiftwork_clock_out.json +0 -95
  13. bantamkit-0.35.2/_assets/tools/shiftwork_plan.json +0 -25
  14. {bantamkit-0.35.2 → bantamkit-0.35.3}/.gitignore +0 -0
  15. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/contracts/default.yaml +0 -0
  16. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/manifest.yaml +0 -0
  17. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/HISTORY.md +0 -0
  18. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/README.md +0 -0
  19. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/docs/architecture.md +0 -0
  20. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/docs/runbook.md +0 -0
  21. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/issues/142-settlement-timeout.md +0 -0
  22. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/patches/0009-retry-budget.patch +0 -0
  23. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/__init__.py +0 -0
  24. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/config.py +0 -0
  25. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/errors.py +0 -0
  26. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/posting.py +0 -0
  27. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/registry.py +0 -0
  28. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/report.py +0 -0
  29. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/retry.py +0 -0
  30. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/settle.py +0 -0
  31. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/src/ledger/validate.py +0 -0
  32. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/tests/test_posting.py +0 -0
  33. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/repo/tests/test_settle.py +0 -0
  34. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-error-contract.yaml +0 -0
  35. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-handler-map.yaml +0 -0
  36. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-patch-before-after.yaml +0 -0
  37. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-retry-attempts.yaml +0 -0
  38. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-settlement-config.yaml +0 -0
  39. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-symbol-home.yaml +0 -0
  40. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-trace-blame.yaml +0 -0
  41. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/devteam/tasks/dt-unread-key.yaml +0 -0
  42. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-in-137.yaml +0 -0
  43. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-in-359.yaml +0 -0
  44. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-in-372.yaml +0 -0
  45. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-out-11764.yaml +0 -0
  46. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-out-4137.yaml +0 -0
  47. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-large-out-8022.yaml +0 -0
  48. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-small-137.yaml +0 -0
  49. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-small-261.yaml +0 -0
  50. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/document/tasks/doc-small-388.yaml +0 -0
  51. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/fixtures/.gitkeep +0 -0
  52. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/fixtures/catalog.json +0 -0
  53. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/perturbations/task-completion.yaml +0 -0
  54. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/.gitkeep +0 -0
  55. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/extract-contact.yaml +0 -0
  56. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/extract-invoice.yaml +0 -0
  57. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/extract-order.yaml +0 -0
  58. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/extract-schedule.yaml +0 -0
  59. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/extract-versions.yaml +0 -0
  60. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/nav-prod-port.yaml +0 -0
  61. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/nav-release-bundle.yaml +0 -0
  62. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-audit-retention.yaml +0 -0
  63. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-cache-ttl.yaml +0 -0
  64. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-db-port.yaml +0 -0
  65. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-deploy.yaml +0 -0
  66. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-env-endpoint.yaml +0 -0
  67. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-oncall-rotation.yaml +0 -0
  68. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-oncall.yaml +0 -0
  69. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-org-quota.yaml +0 -0
  70. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/recall-owner.yaml +0 -0
  71. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-basket-total.yaml +0 -0
  72. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-cheapest.yaml +0 -0
  73. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-compare.yaml +0 -0
  74. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-gadget-value.yaml +0 -0
  75. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-stock-total.yaml +0 -0
  76. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/evals/tasks/shop-total.yaml +0 -0
  77. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/pricing/default.json +0 -0
  78. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/profiles/default.yaml +0 -0
  79. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/profiles/patient.yaml +0 -0
  80. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/rubrics/.gitkeep +0 -0
  81. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/rubrics/code-quality.yaml +0 -0
  82. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/rubrics/grounded-completion.yaml +0 -0
  83. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/rubrics/task-completion.yaml +0 -0
  84. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/schemas/shiftwork-checkpoint.json +0 -0
  85. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/skills/.gitkeep +0 -0
  86. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/skills/file-graph.md +0 -0
  87. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/skills/memory.md +0 -0
  88. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/.gitkeep +0 -0
  89. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/bantamkit_read.json +0 -0
  90. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/bantamkit_status.json +0 -0
  91. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/build_identity.json +0 -0
  92. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/document_list.json +0 -0
  93. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/document_read.json +0 -0
  94. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/file_graph.json +0 -0
  95. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/memory_compact.json +0 -0
  96. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/memory_dream.json +0 -0
  97. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/memory_recall.json +0 -0
  98. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/memory_save.json +0 -0
  99. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/repo_map.json +0 -0
  100. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/shiftwork_clock_in.json +0 -0
  101. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/shiftwork_status.json +0 -0
  102. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/skill_audit.json +0 -0
  103. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/token_ledger.json +0 -0
  104. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/validate_json.json +0 -0
  105. {bantamkit-0.35.2 → bantamkit-0.35.3}/_assets/tools/work_plan.json +0 -0
  106. {bantamkit-0.35.2 → bantamkit-0.35.3}/hatch_build.py +0 -0
  107. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/agent.py +0 -0
  108. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/assets.py +0 -0
  109. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/budget.py +0 -0
  110. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/client.py +0 -0
  111. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/contract.py +0 -0
  112. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/criticreplay.py +0 -0
  113. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/critique.py +0 -0
  114. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/docmanifest.py +0 -0
  115. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/docread.py +0 -0
  116. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/evalrun.py +0 -0
  117. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/eventlog.py +0 -0
  118. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/filegraph.py +0 -0
  119. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/hostinstall.py +0 -0
  120. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/loopguard.py +0 -0
  121. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/mcpreport.py +0 -0
  122. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/__init__.py +0 -0
  123. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/__main__.py +0 -0
  124. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/component.py +0 -0
  125. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/divergence.py +0 -0
  126. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/dream.py +0 -0
  127. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/memory/layers.py +0 -0
  128. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/pdfread.py +0 -0
  129. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/pricing.py +0 -0
  130. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/profile.py +0 -0
  131. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/repomap.py +0 -0
  132. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/selfupdate.py +0 -0
  133. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/shiftwork.py +0 -0
  134. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/skillaudit.py +0 -0
  135. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/statusline.py +0 -0
  136. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/structured.py +0 -0
  137. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/textutil.py +0 -0
  138. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/tokenledger.py +0 -0
  139. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/updatecheck.py +0 -0
  140. {bantamkit-0.35.2 → bantamkit-0.35.3}/src/bantamkit/workplan.py +0 -0
  141. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/cli_exit_status_probe.py +0 -0
  142. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/conftest.py +0 -0
  143. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/bad-crc.docx +0 -0
  144. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/charref-4301-digits.html +0 -0
  145. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/charset-table.json +0 -0
  146. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/compression-method-9.docx +0 -0
  147. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/corrupt-deflate.docx +0 -0
  148. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/encrypted-member.docx +0 -0
  149. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/encrypted-mimetype.odt +0 -0
  150. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/eszett-cell-ref.xlsx +0 -0
  151. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/internal-dtd-entity.docx +0 -0
  152. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/rfc2231-charset.eml +0 -0
  153. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/rfc822-nested-twice.eml +0 -0
  154. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/unicode-digit-shared-string.xlsx +0 -0
  155. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/docread/x-uuencode.eml +0 -0
  156. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/f8404ab-perturbation-baseline.json +0 -0
  157. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/data/platform-assumption-baseline.json +0 -0
  158. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/docread_fixtures.py +0 -0
  159. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/perturbation_baseline_harness.py +0 -0
  160. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/rbp16_effect_probe.py +0 -0
  161. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/rbp18_payload_probe.py +0 -0
  162. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_adapter.py +0 -0
  163. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_agent.py +0 -0
  164. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_amendguard.py +0 -0
  165. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_bantamkit_gitignore.py +0 -0
  166. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_bantamkit_read_tool.py +0 -0
  167. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_budget.py +0 -0
  168. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_build_identity.py +0 -0
  169. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_client.py +0 -0
  170. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_compaction_corpus_survey.py +0 -0
  171. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_conformance.py +0 -0
  172. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_conformance_harness_resilience.py +0 -0
  173. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_conformance_suite_table_gate.py +0 -0
  174. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_contract_fanout.py +0 -0
  175. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_criticreplay.py +0 -0
  176. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_critique.py +0 -0
  177. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_doc_commands_gate.py +0 -0
  178. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_docread.py +0 -0
  179. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_docread_ceilings.py +0 -0
  180. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_document_manifest_parity.py +0 -0
  181. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_document_setup.py +0 -0
  182. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_document_tasks.py +0 -0
  183. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_document_tools.py +0 -0
  184. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_encoding_gate.py +0 -0
  185. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_evalrun.py +0 -0
  186. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_eventlog.py +0 -0
  187. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_field_program_gates.py +0 -0
  188. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_field_programs.py +0 -0
  189. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_filegraph.py +0 -0
  190. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_hostinstall.py +0 -0
  191. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_install_shape.py +0 -0
  192. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_ladder_statistics.py +0 -0
  193. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_launcher_which.py +0 -0
  194. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_layers.py +0 -0
  195. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_loopguard.py +0 -0
  196. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_mcp_endpoint.py +0 -0
  197. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_mcpdrift.py +0 -0
  198. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_mcpreport.py +0 -0
  199. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory.py +0 -0
  200. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_compact_tool.py +0 -0
  201. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_component.py +0 -0
  202. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_divergence.py +0 -0
  203. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_dream.py +0 -0
  204. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_layers.py +0 -0
  205. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_memory_store_tripwire.py +0 -0
  206. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_mutmatrix.py +0 -0
  207. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_newline_gate.py +0 -0
  208. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_packaging.py +0 -0
  209. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_pdfread.py +0 -0
  210. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_pinharness_ledger.py +0 -0
  211. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_platform_assumption_gate.py +0 -0
  212. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_pricing.py +0 -0
  213. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_repo_map_tool.py +0 -0
  214. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_repomap.py +0 -0
  215. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_selfupdate.py +0 -0
  216. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_served_tool_count_records.py +0 -0
  217. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_shiftwork.py +0 -0
  218. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_skillaudit.py +0 -0
  219. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_status_surface.py +0 -0
  220. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_statusline.py +0 -0
  221. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_structured.py +0 -0
  222. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_thread_exception_gate.py +0 -0
  223. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_tokenledger.py +0 -0
  224. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_updatecheck.py +0 -0
  225. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_version_agreement.py +0 -0
  226. {bantamkit-0.35.2 → bantamkit-0.35.3}/tests/test_workplan.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bantamkit
3
- Version: 0.35.2
3
+ Version: 0.35.3
4
4
  Summary: Memory MCP server for Claude Code, Cursor, VS Code Copilot and Claude Desktop, plus a Python library that lifts small-model agents. Install once, run offline.
5
5
  Project-URL: Homepage, https://github.com/Ink01101011/bantamkit
6
6
  Project-URL: Repository, https://github.com/Ink01101011/bantamkit
@@ -23,7 +23,7 @@ Requires-Dist: hatchling>=1.24; extra == 'dev'
23
23
  Requires-Dist: pytest>=8.0; extra == 'dev'
24
24
  Requires-Dist: ruff>=0.4; extra == 'dev'
25
25
  Provides-Extra: mcp
26
- Requires-Dist: mcp<3,>=2.0; extra == 'mcp'
26
+ Requires-Dist: mcp<2.1,>=2.0; extra == 'mcp'
27
27
  Description-Content-Type: text/markdown
28
28
 
29
29
  # bantamkit — memory MCP server for Claude Code, Cursor, VS Code Copilot and Claude Desktop (Python)
@@ -97,9 +97,9 @@ CPU architecture and Python minor version** (some wheels, such as `pydantic_core
97
97
  one platform only), copy `wheels/` across, and install from it:
98
98
 
99
99
  ```bash
100
- python -m pip download "bantamkit[mcp]==0.35.2" -d wheels
100
+ python -m pip download "bantamkit[mcp]==0.35.3" -d wheels
101
101
  python -m venv <env>
102
- <env>/bin/pip install --no-index --find-links wheels "bantamkit[mcp]==0.35.2"
102
+ <env>/bin/pip install --no-index --find-links wheels "bantamkit[mcp]==0.35.3"
103
103
  <env>/bin/bantamkit-mcp --install cursor
104
104
  ```
105
105
 
@@ -69,9 +69,9 @@ CPU architecture and Python minor version** (some wheels, such as `pydantic_core
69
69
  one platform only), copy `wheels/` across, and install from it:
70
70
 
71
71
  ```bash
72
- python -m pip download "bantamkit[mcp]==0.35.2" -d wheels
72
+ python -m pip download "bantamkit[mcp]==0.35.3" -d wheels
73
73
  python -m venv <env>
74
- <env>/bin/pip install --no-index --find-links wheels "bantamkit[mcp]==0.35.2"
74
+ <env>/bin/pip install --no-index --find-links wheels "bantamkit[mcp]==0.35.3"
75
75
  <env>/bin/bantamkit-mcp --install cursor
76
76
  ```
77
77
 
@@ -0,0 +1,150 @@
1
+ {
2
+ "name": "shiftwork_clock_out",
3
+ "description": "Shift-work clock-out: record a finished unit — set its status, advance plan.cursor, merge handoff_patch, push history_entry onto the 5-entry ring — validating the whole mutated document against the checkpoint schema BEFORE an atomic write (a failure writes nothing and returns result=error). `unit_id` must be `plan.cursor`, the one unit clock_in briefs; any other id is refused (`unit X is not the cursor unit Y`). `handoff_patch` and `history_entry` are both required arguments: `handoff_patch` is shallow-merged into `handoff` and takes only its four keys (pass `{}` to leave it unchanged); `history_entry` needs `unit` and `outcome`; `status` is one of the five unit statuses — a value outside these shapes is refused by the schema check, nothing written. Every success appends one accounting line (unit, role, status, ts, briefed, plus your accounting fields, e.g. tokens/duration_ms/model) to <checkpoint>.log.jsonl, beside the brief lines clock_in writes (a brief line carries `event` and no `status`). `accounting` is optional: null, the default, logs a line with no cost figures at all; a non-null line must fit the accounting shape below. `briefed` is written by the runtime, never taken from you — a self-reported value is overwritten by the measured one: true when a brief was issued for this unit since its LAST clock-out (not ever), so a unit re-run inline after a blocked clock-out reads false. A brief is not a precondition: a unit that was never clocked in still clocks out, with briefed=false. If the checkpoint's job.roles names this unit's role, accounting.model must be one of that role's model identifiers, spelled exactly — a model that is not on the list, or none reported at all, is refused before anything is written; a role job.roles omits, and a checkpoint carrying no job.roles, are unconstrained.",
4
+ "surfaces": [
5
+ "mcp"
6
+ ],
7
+ "parameters": {
8
+ "properties": {
9
+ "checkpoint": {
10
+ "title": "Checkpoint",
11
+ "type": "string"
12
+ },
13
+ "unit_id": {
14
+ "title": "Unit Id",
15
+ "type": "string",
16
+ "description": "The unit being clocked out. Must equal plan.cursor — the single-pointer contract: clock_in briefs only the cursor unit and clock_out accepts only it."
17
+ },
18
+ "status": {
19
+ "title": "Status",
20
+ "type": "string",
21
+ "enum": [
22
+ "todo",
23
+ "in_progress",
24
+ "done",
25
+ "blocked",
26
+ "dropped"
27
+ ],
28
+ "description": "The unit's new status; the same five values the checkpoint schema allows on plan.units[].status."
29
+ },
30
+ "handoff_patch": {
31
+ "title": "Handoff Patch",
32
+ "type": "object",
33
+ "additionalProperties": false,
34
+ "description": "Shallow-merged into `handoff`: only these four keys exist there, and a key outside them is refused by the schema check. Required; `{}` leaves `handoff` unchanged.",
35
+ "properties": {
36
+ "next_action": {
37
+ "type": "string",
38
+ "minLength": 1,
39
+ "description": "Imperative first move — zero re-derivation."
40
+ },
41
+ "open_questions": {
42
+ "type": "array",
43
+ "items": {
44
+ "type": "string",
45
+ "minLength": 1
46
+ },
47
+ "description": "The autonomy switch: empty = driver keeps cycling, non-empty = STOP and ask."
48
+ },
49
+ "do_not": {
50
+ "type": "array",
51
+ "items": {
52
+ "type": "string",
53
+ "minLength": 1
54
+ },
55
+ "description": "Negative space — near-mistakes past sessions made."
56
+ },
57
+ "notes": {
58
+ "type": "string",
59
+ "description": "Free-form prose for the next session, one string; the empty string clears it."
60
+ }
61
+ }
62
+ },
63
+ "history_entry": {
64
+ "title": "History Entry",
65
+ "type": "object",
66
+ "required": [
67
+ "unit",
68
+ "outcome"
69
+ ],
70
+ "additionalProperties": true,
71
+ "description": "Pushed onto the 5-entry `history` ring; `unit` and `outcome` are required, `notes` optional, other keys pass through.",
72
+ "properties": {
73
+ "unit": {
74
+ "type": "string",
75
+ "minLength": 1
76
+ },
77
+ "outcome": {
78
+ "type": "string",
79
+ "minLength": 1
80
+ },
81
+ "notes": {
82
+ "type": "string"
83
+ }
84
+ }
85
+ },
86
+ "accounting": {
87
+ "anyOf": [
88
+ {
89
+ "type": "object",
90
+ "description": "Per-unit cost. Written verbatim, beside ts/unit/role/status, as one line of <checkpoint>.log.jsonl - the ledger the N-sessions experiment reads. The keys named here have fixed meanings so a reader holding only the ledger knows what each number counts, in which unit, and which model produced it; any other key passes through unchanged (a refused key is friction, not safety). Lines written before this schema existed may spell these differently; the schema governs new writes only.",
91
+ "properties": {
92
+ "tokens": {
93
+ "type": "integer",
94
+ "minimum": 0,
95
+ "description": "Tokens the unit consumed as the harness's subagent counter reports them: every class it reports (input, output, cache creation) summed, EXCLUDING cache reads. In a non-null accounting line it is required — a line without it is refused before anything is written; only `accounting: null`, the default, is exempt, and that logs no cost at all. A rounded self-estimate is allowed only if `note` says it is one; a unit whose cost is unknown cannot be compared with any other."
96
+ },
97
+ "cache_read_tokens": {
98
+ "type": "integer",
99
+ "minimum": 0,
100
+ "description": "Tokens the unit's requests served from prompt cache, kept OUT of `tokens` because they dominate a real session's traffic and would swamp the work signal. Optional: omit it when the harness reports no such figure - a written 0 means the unit read nothing from cache."
101
+ },
102
+ "duration_ms": {
103
+ "type": "integer",
104
+ "minimum": 0,
105
+ "description": "Wall-clock from spawning the subagent to its final message, in whole milliseconds. In a non-null line it is required, and it is the only duration key with a defined meaning: older lines carry `duration`, `duration_min` or `duration_s`, which this schema neither reads nor renames."
106
+ },
107
+ "model": {
108
+ "type": "string",
109
+ "description": "The exact identifier of the model the subagent actually ran on (e.g. `claude-sonnet-5`) and nothing else - a caveat about how it was chosen belongs in `note`. Not required here: when the checkpoint's job.roles names this unit's role, clock_out already requires it and refuses a value off that role's list, spelled exactly; a checkpoint with no job.roles leaves it optional. This schema does not change that gate."
110
+ },
111
+ "tool_uses": {
112
+ "type": "integer",
113
+ "minimum": 0,
114
+ "description": "Tool calls the subagent made, as the harness counts them. Optional; it is the denominator that makes `tokens` comparable across units of different size (per-unit cost is linear in tool calls, not in unit length)."
115
+ },
116
+ "note": {
117
+ "type": "string",
118
+ "description": "Free text for what the fields above cannot say: that `tokens` is an estimate, that this is a retry or a second scope under the same unit id, what a harness counter excludes. Anything that is not a model identifier goes here, not in `model`."
119
+ }
120
+ },
121
+ "required": [
122
+ "tokens",
123
+ "duration_ms"
124
+ ],
125
+ "additionalProperties": true
126
+ },
127
+ {
128
+ "type": "null"
129
+ }
130
+ ],
131
+ "default": null,
132
+ "title": "Accounting"
133
+ }
134
+ },
135
+ "required": [
136
+ "checkpoint",
137
+ "unit_id",
138
+ "status",
139
+ "handoff_patch",
140
+ "history_entry"
141
+ ],
142
+ "type": "object",
143
+ "title": "shiftwork_clock_outArguments"
144
+ },
145
+ "output_schema": {
146
+ "type": "object",
147
+ "additionalProperties": true,
148
+ "title": "shiftwork_clock_outDictOutput"
149
+ }
150
+ }
@@ -0,0 +1,25 @@
1
+ {
2
+ "name": "shiftwork_plan",
3
+ "description": "Shift-work plan: read-only batch view of a checkpoint — which units its `depends_on` graph permits to run in parallel. Answers `{\"result\": \"plan\", \"batches\", \"ready\", \"sequence\", \"width\", \"cursor\"}`: `ready` is `batches[0]`, the units whose dependencies are all satisfied, and `width` is the largest batch length, the widest fan-out. Only `cursor` can be clocked: shiftwork_clock_in briefs it and shiftwork_clock_out refuses every other id, so a `ready` member beyond the cursor is what the graph permits, not a unit the clock accepts yet. Units with status `done` or `dropped` are satisfied — dropped from the graph, and edges pointing at them treated as already resolved; `todo`, `in_progress` and `blocked` stay in it. Every unit has priority 0, because the checkpoint schema has no priority field, so order inside a batch is `plan.units` order. Same refusal sentences as `work_plan` for a duplicate id, an unknown dependency or a cycle, and the same refusals `shiftwork_status` gives for a checkpoint it cannot read. It reports what the dependency graph permits; it does not move the cursor, which is echoed unchanged so the single-pointer contract and the batch view can be read side by side. An orchestrator stays responsible for what it actually dispatches: `depends_on` encodes logical order, not file contention. Never mutates.",
4
+ "surfaces": [
5
+ "mcp"
6
+ ],
7
+ "parameters": {
8
+ "properties": {
9
+ "checkpoint": {
10
+ "title": "Checkpoint",
11
+ "type": "string"
12
+ }
13
+ },
14
+ "required": [
15
+ "checkpoint"
16
+ ],
17
+ "type": "object",
18
+ "title": "shiftwork_planArguments"
19
+ },
20
+ "output_schema": {
21
+ "type": "object",
22
+ "additionalProperties": true,
23
+ "title": "shiftwork_planDictOutput"
24
+ }
25
+ }
@@ -69,7 +69,12 @@ dependencies = ["httpx>=0.27", "jsonschema>=4.21", "pyyaml>=6.0"]
69
69
  # without this declaration the bar would silently skip in exactly the environment
70
70
  # CI runs (`pip install -e "runtime-py[dev,mcp]"`). Declared here so it runs.
71
71
  dev = ["pytest>=8.0", "ruff>=0.4", "hatchling>=1.24"]
72
- mcp = ["mcp>=2.0,<3"]
72
+ # Upper bound measured, not guessed: 2.0.0 and 2.0.1 wrap a tool crash as
73
+ # `Error executing tool <name>: <exc>`, which runtime-ts/src/mcp/server.ts mirrors.
74
+ # 2.1.0 added an UnexpectedToolError branch that drops the `: <exc>` suffix, so the
75
+ # crash text stops reaching the client and `--suite workplan` goes red on the
76
+ # reference while the port still carries it. Raise this only together with the port.
77
+ mcp = ["mcp>=2.0,<2.1"]
73
78
 
74
79
  [project.scripts]
75
80
  bantamkit-mcp = "bantamkit.mcpserver:main"
@@ -29,4 +29,4 @@ from bantamkit.structured import StructuredOutputError, extract_json, structured
29
29
  # file as its dynamic version source, so the wheel's metadata and the string the MCP
30
30
  # server advertises are the same committed bytes, and neither is a function of when
31
31
  # someone last ran `pip`.
32
- __version__ = "0.35.2"
32
+ __version__ = "0.35.3"
@@ -1907,6 +1907,25 @@ def build_server(memory: Memory, log: EventLog | None = None) -> Any:
1907
1907
 
1908
1908
  @server.resource("bantamkit://skills/{name}")
1909
1909
  def skill_resource(name: str) -> str:
1910
+ # A skill that pairs with a tool is served on that tool's surfaces and no other. The
1911
+ # pairing is the one `filegraph.py` spells: `load_tool("file_graph")` beside
1912
+ # `load_skill("file-graph")` — the skill's name with `-` for `_`. `file-graph` is the
1913
+ # eval agent's system-prompt snippet telling it to call `file_graph`, a tool whose
1914
+ # asset claims only `agent`; handing that text to an MCP client sends it to a tool
1915
+ # `tools/list` does not carry and `tools/call` refuses as unknown (job60 row 46).
1916
+ # Same sentence and same error code on both runtimes; `tools/conformance/suites/
1917
+ # instructions.mjs` reads every skill on disk over `resources/read` and resolves the
1918
+ # tools it names against `tools/list`.
1919
+ paired = name.replace("-", "_")
1920
+ try:
1921
+ asset = load_tool_asset(paired)
1922
+ except AssetNotFound:
1923
+ asset = None
1924
+ if asset is not None and "mcp" not in asset["surfaces"]:
1925
+ raise ResourceError(
1926
+ f"skill asset {name} is not served here: it pairs with tool {paired}, "
1927
+ f"whose asset claims surfaces {asset['surfaces']}, not mcp"
1928
+ )
1910
1929
  try:
1911
1930
  return load_skill(name)
1912
1931
  except AssetNotFound:
@@ -133,7 +133,7 @@ _FACTS_UNREADABLE = (
133
133
  _ARCHIVE_UNREADABLE = (
134
134
  "an archive that could not be listed is not an empty archive, and answering "
135
135
  "'nothing is archived' here is what makes compaction look like deletion — the "
136
- "facts compact() moved are still on disk under this path"
136
+ "facts that compaction moved are still on disk under this path"
137
137
  )
138
138
  # The same distinction one syscall down, for `restore`, which stats one named path
139
139
  # instead of listing (see `archived()` for why). A refused stat is not an absent file,
@@ -1310,5 +1310,5 @@ class MemoryStore:
1310
1310
  if size > self.index_budget:
1311
1311
  raise MemoryBudgetExceeded(
1312
1312
  f"memory index is {size} bytes, budget is {self.index_budget}: "
1313
- f"run compact() or tersen descriptions"
1313
+ f"compact the store or tersen descriptions"
1314
1314
  )
@@ -187,7 +187,7 @@
187
187
  }
188
188
  },
189
189
  "shiftwork_clock_out": {
190
- "description": "Shift-work clock-out: record a finished unit — set its status, advance plan.cursor, merge handoff_patch, push history_entry onto the 5-entry ring — validating the whole mutated document against the checkpoint schema BEFORE an atomic write (a failure writes nothing and returns result=error). Every success appends one accounting line (unit, role, status, ts, briefed, plus your accounting fields, e.g. tokens/duration_ms/model) to <checkpoint>.log.jsonl, beside the brief lines clock_in writes (a brief line carries `event` and no `status`). `briefed` is written by the runtime, never taken from you — a self-reported value is overwritten by the measured one: true when a brief was issued for this unit since its LAST clock-out (not ever), so a unit re-run inline after a blocked clock-out reads false. It records; it never refuses — a unit that was never clocked in still clocks out, with briefed=false. If the checkpoint's job.roles names this unit's role, accounting.model must be one of that role's model identifiers, spelled exactly — a model that is not on the list, or none reported at all, is refused before anything is written; a role job.roles omits, and a checkpoint carrying no job.roles, are unconstrained.",
190
+ "description": "Shift-work clock-out: record a finished unit — set its status, advance plan.cursor, merge handoff_patch, push history_entry onto the 5-entry ring — validating the whole mutated document against the checkpoint schema BEFORE an atomic write (a failure writes nothing and returns result=error). `unit_id` must be `plan.cursor`, the one unit clock_in briefs; any other id is refused (`unit X is not the cursor unit Y`). `handoff_patch` and `history_entry` are both required arguments: `handoff_patch` is shallow-merged into `handoff` and takes only its four keys (pass `{}` to leave it unchanged); `history_entry` needs `unit` and `outcome`; `status` is one of the five unit statuses — a value outside these shapes is refused by the schema check, nothing written. Every success appends one accounting line (unit, role, status, ts, briefed, plus your accounting fields, e.g. tokens/duration_ms/model) to <checkpoint>.log.jsonl, beside the brief lines clock_in writes (a brief line carries `event` and no `status`). `accounting` is optional: null, the default, logs a line with no cost figures at all; a non-null line must fit the accounting shape below. `briefed` is written by the runtime, never taken from you — a self-reported value is overwritten by the measured one: true when a brief was issued for this unit since its LAST clock-out (not ever), so a unit re-run inline after a blocked clock-out reads false. A brief is not a precondition: a unit that was never clocked in still clocks out, with briefed=false. If the checkpoint's job.roles names this unit's role, accounting.model must be one of that role's model identifiers, spelled exactly — a model that is not on the list, or none reported at all, is refused before anything is written; a role job.roles omits, and a checkpoint carrying no job.roles, are unconstrained.",
191
191
  "inputSchema": {
192
192
  "properties": {
193
193
  "checkpoint": {
@@ -196,21 +196,76 @@
196
196
  },
197
197
  "unit_id": {
198
198
  "title": "Unit Id",
199
- "type": "string"
199
+ "type": "string",
200
+ "description": "The unit being clocked out. Must equal plan.cursor — the single-pointer contract: clock_in briefs only the cursor unit and clock_out accepts only it."
200
201
  },
201
202
  "status": {
202
203
  "title": "Status",
203
- "type": "string"
204
+ "type": "string",
205
+ "enum": [
206
+ "todo",
207
+ "in_progress",
208
+ "done",
209
+ "blocked",
210
+ "dropped"
211
+ ],
212
+ "description": "The unit's new status; the same five values the checkpoint schema allows on plan.units[].status."
204
213
  },
205
214
  "handoff_patch": {
206
- "additionalProperties": true,
207
215
  "title": "Handoff Patch",
208
- "type": "object"
216
+ "type": "object",
217
+ "additionalProperties": false,
218
+ "description": "Shallow-merged into `handoff`: only these four keys exist there, and a key outside them is refused by the schema check. Required; `{}` leaves `handoff` unchanged.",
219
+ "properties": {
220
+ "next_action": {
221
+ "type": "string",
222
+ "minLength": 1,
223
+ "description": "Imperative first move — zero re-derivation."
224
+ },
225
+ "open_questions": {
226
+ "type": "array",
227
+ "items": {
228
+ "type": "string",
229
+ "minLength": 1
230
+ },
231
+ "description": "The autonomy switch: empty = driver keeps cycling, non-empty = STOP and ask."
232
+ },
233
+ "do_not": {
234
+ "type": "array",
235
+ "items": {
236
+ "type": "string",
237
+ "minLength": 1
238
+ },
239
+ "description": "Negative space — near-mistakes past sessions made."
240
+ },
241
+ "notes": {
242
+ "type": "string",
243
+ "description": "Free-form prose for the next session, one string; the empty string clears it."
244
+ }
245
+ }
209
246
  },
210
247
  "history_entry": {
211
- "additionalProperties": true,
212
248
  "title": "History Entry",
213
- "type": "object"
249
+ "type": "object",
250
+ "required": [
251
+ "unit",
252
+ "outcome"
253
+ ],
254
+ "additionalProperties": true,
255
+ "description": "Pushed onto the 5-entry `history` ring; `unit` and `outcome` are required, `notes` optional, other keys pass through.",
256
+ "properties": {
257
+ "unit": {
258
+ "type": "string",
259
+ "minLength": 1
260
+ },
261
+ "outcome": {
262
+ "type": "string",
263
+ "minLength": 1
264
+ },
265
+ "notes": {
266
+ "type": "string"
267
+ }
268
+ }
214
269
  },
215
270
  "accounting": {
216
271
  "anyOf": [
@@ -221,7 +276,7 @@
221
276
  "tokens": {
222
277
  "type": "integer",
223
278
  "minimum": 0,
224
- "description": "Tokens the unit consumed as the harness's subagent counter reports them: every class it reports (input, output, cache creation) summed, EXCLUDING cache reads. Required, because a unit whose cost is unknown cannot be compared with any other; a rounded self-estimate is allowed only if `note` says it is one."
279
+ "description": "Tokens the unit consumed as the harness's subagent counter reports them: every class it reports (input, output, cache creation) summed, EXCLUDING cache reads. In a non-null accounting line it is required — a line without it is refused before anything is written; only `accounting: null`, the default, is exempt, and that logs no cost at all. A rounded self-estimate is allowed only if `note` says it is one; a unit whose cost is unknown cannot be compared with any other."
225
280
  },
226
281
  "cache_read_tokens": {
227
282
  "type": "integer",
@@ -231,7 +286,7 @@
231
286
  "duration_ms": {
232
287
  "type": "integer",
233
288
  "minimum": 0,
234
- "description": "Wall-clock from spawning the subagent to its final message, in whole milliseconds. Required, and the only duration key with a defined meaning: older lines carry `duration`, `duration_min` or `duration_s`, which this schema neither reads nor renames."
289
+ "description": "Wall-clock from spawning the subagent to its final message, in whole milliseconds. In a non-null line it is required, and it is the only duration key with a defined meaning: older lines carry `duration`, `duration_min` or `duration_s`, which this schema neither reads nor renames."
235
290
  },
236
291
  "model": {
237
292
  "type": "string",
@@ -498,7 +553,7 @@
498
553
  }
499
554
  },
500
555
  "shiftwork_plan": {
501
- "description": "Shift-work plan: read-only batch view of a checkpoint — which units its `depends_on` graph permits to run in parallel. Answers `{\"result\": \"plan\", \"batches\", \"ready\", \"sequence\", \"width\", \"cursor\"}`: `ready` is `batches[0]`, the units dispatchable right now, and `width` is the largest batch length, the widest fan-out. Units with status `done` or `dropped` are satisfied — dropped from the graph, and edges pointing at them treated as already resolved; `todo`, `in_progress` and `blocked` stay in it. Every unit has priority 0, because the checkpoint schema has no priority field, so order inside a batch is `plan.units` order. Same refusal sentences as `work_plan` for a duplicate id, an unknown dependency or a cycle, and the same refusals `shiftwork_status` gives for a checkpoint it cannot read. It reports what the dependency graph permits; it does not move the cursor, which is echoed unchanged so the single-pointer contract and the batch view can be read side by side. An orchestrator stays responsible for what it actually dispatches: `depends_on` encodes logical order, not file contention. Never mutates.",
556
+ "description": "Shift-work plan: read-only batch view of a checkpoint — which units its `depends_on` graph permits to run in parallel. Answers `{\"result\": \"plan\", \"batches\", \"ready\", \"sequence\", \"width\", \"cursor\"}`: `ready` is `batches[0]`, the units whose dependencies are all satisfied, and `width` is the largest batch length, the widest fan-out. Only `cursor` can be clocked: shiftwork_clock_in briefs it and shiftwork_clock_out refuses every other id, so a `ready` member beyond the cursor is what the graph permits, not a unit the clock accepts yet. Units with status `done` or `dropped` are satisfied — dropped from the graph, and edges pointing at them treated as already resolved; `todo`, `in_progress` and `blocked` stay in it. Every unit has priority 0, because the checkpoint schema has no priority field, so order inside a batch is `plan.units` order. Same refusal sentences as `work_plan` for a duplicate id, an unknown dependency or a cycle, and the same refusals `shiftwork_status` gives for a checkpoint it cannot read. It reports what the dependency graph permits; it does not move the cursor, which is echoed unchanged so the single-pointer contract and the batch view can be read side by side. An orchestrator stays responsible for what it actually dispatches: `depends_on` encodes logical order, not file contention. Never mutates.",
502
557
  "inputSchema": {
503
558
  "properties": {
504
559
  "checkpoint": {
@@ -231,6 +231,28 @@ def test_missing_resource_errors_name_the_asset(tmp_path):
231
231
  run(scenario())
232
232
 
233
233
 
234
+ def test_a_skill_paired_with_an_agent_only_tool_is_not_served(tmp_path):
235
+ """`file-graph` is `file_graph`'s system-prompt snippet; that tool claims only `agent`.
236
+
237
+ Served over MCP it told the client to call a tool `tools/list` does not carry (job60
238
+ row 46). The pairing is the one `filegraph.py` spells — the skill's name with `-` for
239
+ `_` — and the refusal names the surfaces the tool's asset does claim.
240
+ """
241
+
242
+ async def scenario():
243
+ async with Client(make_server(tmp_path)) as c:
244
+ with pytest.raises(
245
+ MCPError,
246
+ match=(
247
+ r"skill asset file-graph is not served here: it pairs with tool "
248
+ r"file_graph, whose asset claims surfaces \['agent'\], not mcp"
249
+ ),
250
+ ):
251
+ await c.read_resource("bantamkit://skills/file-graph")
252
+
253
+ run(scenario())
254
+
255
+
234
256
  def test_resource_templates_listed(tmp_path):
235
257
  async def scenario():
236
258
  async with Client(make_server(tmp_path)) as c:
@@ -33,7 +33,7 @@ pytest.importorskip("mcp")
33
33
  from mcp import ClientSession # noqa: E402
34
34
  from mcp.client.stdio import StdioServerParameters, stdio_client # noqa: E402
35
35
 
36
- from bantamkit.assets import assets_root, load_tool # noqa: E402
36
+ from bantamkit.assets import assets_root, load_schema, load_tool # noqa: E402
37
37
 
38
38
  SRC = Path(__file__).resolve().parents[1] / "src"
39
39
 
@@ -492,3 +492,50 @@ def test_the_advertised_surface_is_read_from_the_asset_pack_at_startup(tmp_path)
492
492
  assert not _surface_differences(expected, wire), "\n".join(
493
493
  _surface_differences(expected, wire)
494
494
  )
495
+
496
+
497
+ # ---------------------------------------------------------------------------
498
+ # J60-F2: the sub-schemas `shiftwork_clock_out` SERVES are a second copy of what the checkpoint
499
+ # schema ENFORCES at write time. Until job60 the served `inputSchema` said `status: string`,
500
+ # `history_entry: {additionalProperties: true}` and `handoff_patch: {additionalProperties: true}`
501
+ # while the writer refused `finished`, `{}` and any fifth handoff key — a host read one contract
502
+ # and hit another (job59 rows 28/29/30). The served copy exists because both servers serve
503
+ # `asset["parameters"]` verbatim and validate arguments from a signature, never from the asset;
504
+ # so the asset is disclosure, the checkpoint schema is enforcement, and this node is what keeps
505
+ # the two saying the same thing. The running-surface version of the same check is
506
+ # `tools/conformance/suites/instructions.mjs` clause (b).
507
+
508
+
509
+ def test_clock_out_serves_the_sub_schemas_the_checkpoint_writer_enforces():
510
+ served = _asset("shiftwork_clock_out")["parameters"]
511
+ writer = load_schema("shiftwork-checkpoint")["properties"]
512
+ props = served["properties"]
513
+
514
+ status = writer["plan"]["properties"]["units"]["items"]["properties"]["status"]
515
+ assert props["status"]["enum"] == status["enum"]
516
+
517
+ history_item = writer["history"]["items"]
518
+ assert sorted(props["history_entry"]["required"]) == sorted(history_item["required"])
519
+ assert props["history_entry"]["additionalProperties"] is history_item["additionalProperties"]
520
+
521
+ handoff = writer["handoff"]
522
+ assert props["handoff_patch"]["additionalProperties"] is False
523
+ assert handoff["additionalProperties"] is False
524
+ assert set(props["handoff_patch"]["properties"]) == set(handoff["properties"])
525
+ # A patch is a shallow merge: it may omit keys the whole document requires.
526
+ assert "required" not in props["handoff_patch"]
527
+
528
+ # The tool requires both objects, and the description says so in the same sentence that
529
+ # names `handoff_patch` — the row-56 disclosure — rather than leaving `required` to be
530
+ # discovered by a refused call.
531
+ assert {"handoff_patch", "history_entry"} <= set(served["required"])
532
+ description = _asset("shiftwork_clock_out")["description"]
533
+ assert "never refuses" not in description
534
+ assert any(
535
+ "handoff_patch" in sentence and "required" in sentence.lower()
536
+ for sentence in description.replace(";", ".").split(". ")
537
+ )
538
+ assert any(
539
+ "unit_id" in sentence and "cursor" in sentence
540
+ for sentence in description.replace(";", ".").split(". ")
541
+ )
@@ -1,95 +0,0 @@
1
- {
2
- "name": "shiftwork_clock_out",
3
- "description": "Shift-work clock-out: record a finished unit — set its status, advance plan.cursor, merge handoff_patch, push history_entry onto the 5-entry ring — validating the whole mutated document against the checkpoint schema BEFORE an atomic write (a failure writes nothing and returns result=error). Every success appends one accounting line (unit, role, status, ts, briefed, plus your accounting fields, e.g. tokens/duration_ms/model) to <checkpoint>.log.jsonl, beside the brief lines clock_in writes (a brief line carries `event` and no `status`). `briefed` is written by the runtime, never taken from you — a self-reported value is overwritten by the measured one: true when a brief was issued for this unit since its LAST clock-out (not ever), so a unit re-run inline after a blocked clock-out reads false. It records; it never refuses — a unit that was never clocked in still clocks out, with briefed=false. If the checkpoint's job.roles names this unit's role, accounting.model must be one of that role's model identifiers, spelled exactly \u2014 a model that is not on the list, or none reported at all, is refused before anything is written; a role job.roles omits, and a checkpoint carrying no job.roles, are unconstrained.",
4
- "surfaces": [
5
- "mcp"
6
- ],
7
- "parameters": {
8
- "properties": {
9
- "checkpoint": {
10
- "title": "Checkpoint",
11
- "type": "string"
12
- },
13
- "unit_id": {
14
- "title": "Unit Id",
15
- "type": "string"
16
- },
17
- "status": {
18
- "title": "Status",
19
- "type": "string"
20
- },
21
- "handoff_patch": {
22
- "additionalProperties": true,
23
- "title": "Handoff Patch",
24
- "type": "object"
25
- },
26
- "history_entry": {
27
- "additionalProperties": true,
28
- "title": "History Entry",
29
- "type": "object"
30
- },
31
- "accounting": {
32
- "anyOf": [
33
- {
34
- "type": "object",
35
- "description": "Per-unit cost. Written verbatim, beside ts/unit/role/status, as one line of <checkpoint>.log.jsonl - the ledger the N-sessions experiment reads. The keys named here have fixed meanings so a reader holding only the ledger knows what each number counts, in which unit, and which model produced it; any other key passes through unchanged (a refused key is friction, not safety). Lines written before this schema existed may spell these differently; the schema governs new writes only.",
36
- "properties": {
37
- "tokens": {
38
- "type": "integer",
39
- "minimum": 0,
40
- "description": "Tokens the unit consumed as the harness's subagent counter reports them: every class it reports (input, output, cache creation) summed, EXCLUDING cache reads. Required, because a unit whose cost is unknown cannot be compared with any other; a rounded self-estimate is allowed only if `note` says it is one."
41
- },
42
- "cache_read_tokens": {
43
- "type": "integer",
44
- "minimum": 0,
45
- "description": "Tokens the unit's requests served from prompt cache, kept OUT of `tokens` because they dominate a real session's traffic and would swamp the work signal. Optional: omit it when the harness reports no such figure - a written 0 means the unit read nothing from cache."
46
- },
47
- "duration_ms": {
48
- "type": "integer",
49
- "minimum": 0,
50
- "description": "Wall-clock from spawning the subagent to its final message, in whole milliseconds. Required, and the only duration key with a defined meaning: older lines carry `duration`, `duration_min` or `duration_s`, which this schema neither reads nor renames."
51
- },
52
- "model": {
53
- "type": "string",
54
- "description": "The exact identifier of the model the subagent actually ran on (e.g. `claude-sonnet-5`) and nothing else - a caveat about how it was chosen belongs in `note`. Not required here: when the checkpoint's job.roles names this unit's role, clock_out already requires it and refuses a value off that role's list, spelled exactly; a checkpoint with no job.roles leaves it optional. This schema does not change that gate."
55
- },
56
- "tool_uses": {
57
- "type": "integer",
58
- "minimum": 0,
59
- "description": "Tool calls the subagent made, as the harness counts them. Optional; it is the denominator that makes `tokens` comparable across units of different size (per-unit cost is linear in tool calls, not in unit length)."
60
- },
61
- "note": {
62
- "type": "string",
63
- "description": "Free text for what the fields above cannot say: that `tokens` is an estimate, that this is a retry or a second scope under the same unit id, what a harness counter excludes. Anything that is not a model identifier goes here, not in `model`."
64
- }
65
- },
66
- "required": [
67
- "tokens",
68
- "duration_ms"
69
- ],
70
- "additionalProperties": true
71
- },
72
- {
73
- "type": "null"
74
- }
75
- ],
76
- "default": null,
77
- "title": "Accounting"
78
- }
79
- },
80
- "required": [
81
- "checkpoint",
82
- "unit_id",
83
- "status",
84
- "handoff_patch",
85
- "history_entry"
86
- ],
87
- "type": "object",
88
- "title": "shiftwork_clock_outArguments"
89
- },
90
- "output_schema": {
91
- "type": "object",
92
- "additionalProperties": true,
93
- "title": "shiftwork_clock_outDictOutput"
94
- }
95
- }
@@ -1,25 +0,0 @@
1
- {
2
- "name": "shiftwork_plan",
3
- "description": "Shift-work plan: read-only batch view of a checkpoint — which units its `depends_on` graph permits to run in parallel. Answers `{\"result\": \"plan\", \"batches\", \"ready\", \"sequence\", \"width\", \"cursor\"}`: `ready` is `batches[0]`, the units dispatchable right now, and `width` is the largest batch length, the widest fan-out. Units with status `done` or `dropped` are satisfied — dropped from the graph, and edges pointing at them treated as already resolved; `todo`, `in_progress` and `blocked` stay in it. Every unit has priority 0, because the checkpoint schema has no priority field, so order inside a batch is `plan.units` order. Same refusal sentences as `work_plan` for a duplicate id, an unknown dependency or a cycle, and the same refusals `shiftwork_status` gives for a checkpoint it cannot read. It reports what the dependency graph permits; it does not move the cursor, which is echoed unchanged so the single-pointer contract and the batch view can be read side by side. An orchestrator stays responsible for what it actually dispatches: `depends_on` encodes logical order, not file contention. Never mutates.",
4
- "surfaces": [
5
- "mcp"
6
- ],
7
- "parameters": {
8
- "properties": {
9
- "checkpoint": {
10
- "title": "Checkpoint",
11
- "type": "string"
12
- }
13
- },
14
- "required": [
15
- "checkpoint"
16
- ],
17
- "type": "object",
18
- "title": "shiftwork_planArguments"
19
- },
20
- "output_schema": {
21
- "type": "object",
22
- "additionalProperties": true,
23
- "title": "shiftwork_planDictOutput"
24
- }
25
- }
File without changes