memleaf 0.2.30__tar.gz → 0.2.31__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (160) hide show
  1. {memleaf-0.2.30 → memleaf-0.2.31}/CHANGELOG.md +7 -0
  2. {memleaf-0.2.30/src/memleaf.egg-info → memleaf-0.2.31}/PKG-INFO +4 -4
  3. {memleaf-0.2.30 → memleaf-0.2.31}/README.en.md +3 -3
  4. {memleaf-0.2.30 → memleaf-0.2.31}/README.md +3 -3
  5. memleaf-0.2.31/docs/capture-budget-design.md +74 -0
  6. {memleaf-0.2.30 → memleaf-0.2.31}/docs/evidence-retention.md +11 -3
  7. {memleaf-0.2.30 → memleaf-0.2.31}/docs/general-processing.md +9 -6
  8. {memleaf-0.2.30 → memleaf-0.2.31}/pyproject.toml +1 -1
  9. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/__init__.py +1 -1
  10. memleaf-0.2.31/src/memleaf/evidence_budget.py +294 -0
  11. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/hermes_provider/__init__.py +28 -19
  12. memleaf-0.2.31/src/memleaf/hermes_provider/evidence_budget.py +294 -0
  13. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/hermes_provider/plugin.yaml +1 -1
  14. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/host_runtime.py +18 -3
  15. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/installer.py +1 -1
  16. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/mcp_server.py +1 -1
  17. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/provenance.py +61 -53
  18. {memleaf-0.2.30 → memleaf-0.2.31/src/memleaf.egg-info}/PKG-INFO +4 -4
  19. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf.egg-info/SOURCES.txt +4 -0
  20. memleaf-0.2.31/tests/test_evidence_budget.py +381 -0
  21. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_evidence_retention_policy.py +2 -2
  22. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_general_tool_provenance.py +2 -2
  23. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_processing_contract_v026.py +6 -6
  24. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_pypi_install.py +1 -0
  25. {memleaf-0.2.30 → memleaf-0.2.31}/IMPLEMENTATION_PLAN.md +0 -0
  26. {memleaf-0.2.30 → memleaf-0.2.31}/LICENSE +0 -0
  27. {memleaf-0.2.30 → memleaf-0.2.31}/MANIFEST.in +0 -0
  28. {memleaf-0.2.30 → memleaf-0.2.31}/RELEASE_CHECKLIST.md +0 -0
  29. {memleaf-0.2.30 → memleaf-0.2.31}/docs/config-migrations.md +0 -0
  30. {memleaf-0.2.30 → memleaf-0.2.31}/docs/core-refactor.md +0 -0
  31. {memleaf-0.2.30 → memleaf-0.2.31}/docs/gate-evidence-boundary.md +0 -0
  32. {memleaf-0.2.30 → memleaf-0.2.31}/docs/hermes-mcp-runtime.md +0 -0
  33. {memleaf-0.2.30 → memleaf-0.2.31}/docs/performance.md +0 -0
  34. {memleaf-0.2.30 → memleaf-0.2.31}/docs/v0.2.26-processing-status.md +0 -0
  35. {memleaf-0.2.30 → memleaf-0.2.31}/examples/README.md +0 -0
  36. {memleaf-0.2.30 → memleaf-0.2.31}/examples/basic_usage.py +0 -0
  37. {memleaf-0.2.30 → memleaf-0.2.31}/examples/live_processing_acceptance.py +0 -0
  38. {memleaf-0.2.30 → memleaf-0.2.31}/examples/mcp_stdio.ndjson +0 -0
  39. {memleaf-0.2.30 → memleaf-0.2.31}/install.ps1 +0 -0
  40. {memleaf-0.2.30 → memleaf-0.2.31}/install.sh +0 -0
  41. {memleaf-0.2.30 → memleaf-0.2.31}/setup.cfg +0 -0
  42. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/__main__.py +0 -0
  43. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/adapters/__init__.py +0 -0
  44. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/adapters/antigravity.py +0 -0
  45. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/adapters/base.py +0 -0
  46. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/adapters/codex.py +0 -0
  47. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/adapters/hermes.py +0 -0
  48. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/admission.py +0 -0
  49. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/budget.py +0 -0
  50. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/capture.py +0 -0
  51. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/cli.py +0 -0
  52. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/compaction.py +0 -0
  53. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/config.py +0 -0
  54. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/credentials.py +0 -0
  55. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/evidence_policy.py +0 -0
  56. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/frontmatter.py +0 -0
  57. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/hermes_provider/README.md +0 -0
  58. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/hermes_runtime.py +0 -0
  59. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/host_events.py +0 -0
  60. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/inbox.py +0 -0
  61. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/index.py +0 -0
  62. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/inspection.py +0 -0
  63. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/__init__.py +0 -0
  64. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/base.py +0 -0
  65. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/claude_compatible.py +0 -0
  66. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/gemini.py +0 -0
  67. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/openai_compatible.py +0 -0
  68. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/llm/router.py +0 -0
  69. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/locking.py +0 -0
  70. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/memory_commit.py +0 -0
  71. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/memory_planner.py +0 -0
  72. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/memory_writer.py +0 -0
  73. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/model_discovery.py +0 -0
  74. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/model_execution.py +0 -0
  75. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/models.py +0 -0
  76. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/native_index.py +0 -0
  77. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/native_registration.py +0 -0
  78. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/planning_context.py +0 -0
  79. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/process_common.py +0 -0
  80. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/process_journal.py +0 -0
  81. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/process_owner.py +0 -0
  82. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/processing.py +0 -0
  83. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/prompts.py +0 -0
  84. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/recording_policy.py +0 -0
  85. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/redaction.py +0 -0
  86. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/retention.py +0 -0
  87. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/retrieval.py +0 -0
  88. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/retrieval_gate.py +0 -0
  89. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/scope_maintenance.py +0 -0
  90. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/scope_state.py +0 -0
  91. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/service.py +0 -0
  92. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/source_policy.py +0 -0
  93. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/state_layout.py +0 -0
  94. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/turn_audit.py +0 -0
  95. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/turn_plan.py +0 -0
  96. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/update_coordinator.py +0 -0
  97. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/validation.py +0 -0
  98. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf/vault.py +0 -0
  99. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf.egg-info/dependency_links.txt +0 -0
  100. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf.egg-info/entry_points.txt +0 -0
  101. {memleaf-0.2.30 → memleaf-0.2.31}/src/memleaf.egg-info/top_level.txt +0 -0
  102. {memleaf-0.2.30 → memleaf-0.2.31}/tests/__init__.py +0 -0
  103. {memleaf-0.2.30 → memleaf-0.2.31}/tests/semantic_fixtures.py +0 -0
  104. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_admission_noise.py +0 -0
  105. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_codex_install.py +0 -0
  106. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_codex_native_cli.py +0 -0
  107. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_config_migrations_v028.py +0 -0
  108. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_context_budget.py +0 -0
  109. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_credential_safety.py +0 -0
  110. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_cross_host_acceptance.py +0 -0
  111. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_cross_turn_dedupe_regressions.py +0 -0
  112. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_email_actionable_coverage.py +0 -0
  113. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_extraction_quality_regressions.py +0 -0
  114. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_general_evidence_admission.py +0 -0
  115. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_global_todo_acceptance.py +0 -0
  116. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_global_todo_query_no_write.py +0 -0
  117. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_global_todo_retrieval.py +0 -0
  118. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_hermes_native_registration.py +0 -0
  119. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_hermes_provider.py +0 -0
  120. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_hermes_runtime_install.py +0 -0
  121. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_hermes_stdio_transport.py +0 -0
  122. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_host_events.py +0 -0
  123. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_host_runtime_contract.py +0 -0
  124. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_inspection_state_v028.py +0 -0
  125. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_install.py +0 -0
  126. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_long_run_hygiene.py +0 -0
  127. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_maintenance_v2.py +0 -0
  128. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_model_discovery.py +0 -0
  129. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_model_owned_fields.py +0 -0
  130. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_phase2_model_decisions.py +0 -0
  131. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_process_owner_locking.py +0 -0
  132. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_retrieval_gate.py +0 -0
  133. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_retrieval_v2.py +0 -0
  134. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_session_lineage.py +0 -0
  135. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_shared_memory_refactor.py +0 -0
  136. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_source_neutral_todos_v028.py +0 -0
  137. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_a.py +0 -0
  138. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b1.py +0 -0
  139. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b2a.py +0 -0
  140. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b2b.py +0 -0
  141. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3a_commit.py +0 -0
  142. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3a_contract.py +0 -0
  143. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3b_native_context.py +0 -0
  144. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3b_native_index.py +0 -0
  145. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3b_scope.py +0 -0
  146. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3c_retrieval.py +0 -0
  147. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_b3d_scope_maintenance.py +0 -0
  148. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_c1_mcp.py +0 -0
  149. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_c2_init.py +0 -0
  150. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_stage_c3_packaging.py +0 -0
  151. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_state_layout_v028.py +0 -0
  152. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_todo_state_recovery.py +0 -0
  153. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_update_target_recovery.py +0 -0
  154. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_upgrade_preserves_vault.py +0 -0
  155. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v023_scope_correction.py +0 -0
  156. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v2_gate_limits.py +0 -0
  157. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v2_host_flow.py +0 -0
  158. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v2_mcp_flow.py +0 -0
  159. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v2_nomatch_semantics.py +0 -0
  160. {memleaf-0.2.30 → memleaf-0.2.31}/tests/test_v2_search_gate_acceptance.py +0 -0
@@ -2,6 +2,13 @@
2
2
 
3
3
  All notable changes to memleaf are documented here.
4
4
 
5
+ ## 0.2.31 — 2026-09-07
6
+
7
+ - Unify Core and the copied Hermes provider on one idempotent UTF-8 evidence budget: at most 64 source records, 32 KiB per body, and 128 KiB of aggregate body text. Complete matched sources above the former 2,000-character/eight-record limits remain available within those bounds.
8
+ - Keep loss diagnostics bounded to 64 marker identities plus one aggregate marker. Per-record, aggregate, and marker overflow remains explicit incomplete evidence and cannot authorize a write or successful cleanup.
9
+ - Preserve metadata/off, attachment, redaction, and call-ID behavior, and consume metadata-mode pending evidence after a successful capture using the same effective policy on pending and inbox sides.
10
+ - Add provider-copy, capture-to-process, loss-defer, marker-capacity and lifecycle regressions. Deterministic tests do not claim real-model semantic quality, real-session replay, or customer acceptance.
11
+
5
12
  ## 0.2.30 — 2026-09-07
6
13
 
7
14
  - Add a physical evidence projection for the Gate: the complete local inventory remains available for provenance, replay and audit, while only user-origin units and complete external observations become bindable model evidence. Assistant synthesis, retrieved memory, incomplete observations and metadata-only records remain context or unresolved ledger state.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: memleaf
3
- Version: 0.2.30
3
+ Version: 0.2.31
4
4
  Summary: A local-first Markdown memory core for AI agents
5
5
  Author: memleaf contributors
6
6
  License-Expression: MIT
@@ -23,8 +23,8 @@ Dynamic: license-file
23
23
 
24
24
  [English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
25
25
 
26
- > **版本:0.2.30。**
27
- > 核心库、Vault、stdio MCP Server、初始化 CLI、模型路由、提炼流程、受控检索协议和宿主适配器已经实现。本版将完整证据清单与 Gate 可绑定的 physical evidence projection 分开,统一 `candidates`、`coverage`、`evidence_bindings` 协议,保留 legacy candidate-only 的 exact/bound compatibility,对未决证据安全收口并提供 allowlisted `evidence_check` 诊断,同时保持 source-neutral 语义、Markdown 唯一事实源和无 SQLite 运行时依赖。真实模型语义效果仍需结合本地模型和代表性样本验收。
26
+ > **版本:0.2.31。**
27
+ > 核心库、Vault、stdio MCP Server、初始化 CLI、模型路由、提炼流程、受控检索协议和宿主适配器已经实现。本版统一 Core 与 Hermes Provider 的 UTF-8 证据留存预算:最多 64 条 source records、每条正文 32 KiB、正文总量 128 KiB,并保留有界 loss markers 与幂等规范化;同时将 metadata 模式下成功捕获的超限 pending 证据按相同有效策略消费。Gate 继续使用 physical evidence projection、`candidates`、`coverage`、`evidence_bindings` 和 allowlisted `evidence_check` 诊断,保持 source-neutral 语义、Markdown 唯一事实源和无 SQLite 运行时依赖。真实模型语义效果仍需结合本地模型和代表性样本验收。
28
28
  > **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
29
29
 
30
30
  ## 项目定位
@@ -490,7 +490,7 @@ MIT,见 [LICENSE](LICENSE)。
490
490
  *Your memories, in files you own.*
491
491
 
492
492
 
493
- ## 通用处理与只读验收(0.2.30)
493
+ ## 通用处理与只读验收(0.2.31)
494
494
 
495
495
  邮件、日历、工单、文件、浏览器与普通对话共用证据准入、覆盖检查和写入路径。
496
496
  自动摘要只能使用获准引用的原文;助手复述和旧记忆回读不能单独授权新增写入。
@@ -4,8 +4,8 @@
4
4
 
5
5
  [中文](README.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
6
6
 
7
- > **Version: 0.2.30.**
8
- > The core library, Vault, stdio MCP server, initialization CLI, model routing, memory extraction, controlled retrieval protocol, and host adapters are implemented. This release separates the complete evidence inventory from the Gate's bindable physical evidence projection, unifies the `candidates`, `coverage`, and `evidence_bindings` contract, preserves legacy candidate-only exact/bound compatibility, keeps unresolved evidence safe, and adds allowlisted `evidence_check` diagnostics while preserving source-neutral semantics, Markdown as the sole source of truth, and zero SQLite runtime dependencies. Real-model semantics still require local acceptance with the selected model and representative inputs.
7
+ > **Version: 0.2.31.**
8
+ > The core library, Vault, stdio MCP server, initialization CLI, model routing, memory extraction, controlled retrieval protocol, and host adapters are implemented. This release unifies Core and the Hermes provider on an idempotent UTF-8 evidence budget: at most 64 source records, 32 KiB per body, and 128 KiB of aggregate body text, with bounded loss markers; metadata-mode pending evidence from an oversized observation is consumed after successful capture under the same effective policy. The Gate still uses the physical evidence projection, the `candidates`, `coverage`, and `evidence_bindings` contract, and allowlisted `evidence_check` diagnostics while preserving source-neutral semantics, Markdown as the sole source of truth, and zero SQLite runtime dependencies. Real-model semantics still require local acceptance with the selected model and representative inputs.
9
9
  > **The current release supports Hermes and Codex.** Antigravity is not detected, installed, or configured.
10
10
 
11
11
  ## Project scope
@@ -474,7 +474,7 @@ MIT; see [LICENSE](LICENSE).
474
474
  *Your memories, in files you own.*
475
475
 
476
476
 
477
- ## General processing and read-only inspection (0.2.30)
477
+ ## General processing and read-only inspection (0.2.31)
478
478
 
479
479
  Dialogue, calendars, tickets, files, web results and other tools share the evidence, coverage and write path.
480
480
  Models interpret semantics; Core validates physical provenance and exact original quotations.
@@ -4,8 +4,8 @@
4
4
 
5
5
  [English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
6
6
 
7
- > **版本:0.2.30。**
8
- > 核心库、Vault、stdio MCP Server、初始化 CLI、模型路由、提炼流程、受控检索协议和宿主适配器已经实现。本版将完整证据清单与 Gate 可绑定的 physical evidence projection 分开,统一 `candidates`、`coverage`、`evidence_bindings` 协议,保留 legacy candidate-only 的 exact/bound compatibility,对未决证据安全收口并提供 allowlisted `evidence_check` 诊断,同时保持 source-neutral 语义、Markdown 唯一事实源和无 SQLite 运行时依赖。真实模型语义效果仍需结合本地模型和代表性样本验收。
7
+ > **版本:0.2.31。**
8
+ > 核心库、Vault、stdio MCP Server、初始化 CLI、模型路由、提炼流程、受控检索协议和宿主适配器已经实现。本版统一 Core 与 Hermes Provider 的 UTF-8 证据留存预算:最多 64 条 source records、每条正文 32 KiB、正文总量 128 KiB,并保留有界 loss markers 与幂等规范化;同时将 metadata 模式下成功捕获的超限 pending 证据按相同有效策略消费。Gate 继续使用 physical evidence projection、`candidates`、`coverage`、`evidence_bindings` 和 allowlisted `evidence_check` 诊断,保持 source-neutral 语义、Markdown 唯一事实源和无 SQLite 运行时依赖。真实模型语义效果仍需结合本地模型和代表性样本验收。
9
9
  > **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
10
10
 
11
11
  ## 项目定位
@@ -471,7 +471,7 @@ MIT,见 [LICENSE](LICENSE)。
471
471
  *Your memories, in files you own.*
472
472
 
473
473
 
474
- ## 通用处理与只读验收(0.2.30)
474
+ ## 通用处理与只读验收(0.2.31)
475
475
 
476
476
  邮件、日历、工单、文件、浏览器与普通对话共用证据准入、覆盖检查和写入路径。
477
477
  自动摘要只能使用获准引用的原文;助手复述和旧记忆回读不能单独授权新增写入。
@@ -0,0 +1,74 @@
1
+ # Tool evidence capture budget
2
+
3
+ ## Problem and ownership
4
+
5
+ Capture permissions and capture capacity are separate contracts. `metadata`
6
+ deliberately removes source bodies; `off` removes observations. Neither setting
7
+ can produce new external facts for extraction. Enabling `bounded` authorizes
8
+ bounded body retention, but does not reconstruct previously discarded data.
9
+
10
+ The previous implementation also applied independent eight-record and
11
+ 2,000-character limits in the Hermes adapter, provenance normalization and
12
+ pending-state processing. Early discovery and skill results could consume the
13
+ entire record allowance before later source results arrived. Larger plain-text
14
+ results were truncated before Core could account for their complete content.
15
+ This is an ingestion-capacity problem; changing Gate semantics cannot recover
16
+ those bytes.
17
+
18
+ ## Required implementation contract
19
+
20
+ - Keep one shared, deterministic budget implementation for Core and the copied
21
+ Hermes provider. The provider must continue to work without importing Core
22
+ from the host's Python environment.
23
+ - Bound record count, each body and aggregate body size. Publish the limits as
24
+ engineering capacity, not a guarantee of complete capture for arbitrary turns.
25
+ The implementation target is 64 source records, 32 KiB of UTF-8 text per body
26
+ and 128 KiB of total body text. Omission markers have a separate allowance of
27
+ 64 call identities and one aggregate marker. They do not consume source-record
28
+ capacity or count themselves as newly lost observations on a later normalization
29
+ pass. Marker bodies are fixed diagnostics, never retained source excerpts.
30
+ - Preserve complete, matched source results within those limits. Do not rank
31
+ business topics or tool names, infer facts from assistant text, split arbitrary
32
+ stdout into invented document records, or promote truncated prefixes to
33
+ complete observations.
34
+ - Normalize retained evidence idempotently. Cache, capture, inbox read and
35
+ planning must not each discard another part of an already bounded inventory.
36
+ - Keep explicit omission/incompleteness accounting when a limit is exceeded.
37
+ Policy-authorized but incomplete observations remain unresolved and cannot
38
+ authorize a write or successful source cleanup.
39
+ - Apply `metadata`, `off`, attachment exclusions and redaction before persistent
40
+ writes. A capacity increase must not silently change capture permission.
41
+ - Keep the existing Gate/Summarize semantic responsibility and Markdown storage
42
+ model. Larger source capacity is not a claim of successful model extraction.
43
+
44
+ ## Acceptance
45
+
46
+ Use synthetic host messages shaped like an ordinary discovery/read turn: early
47
+ tool discovery, skills and memory search followed by nine later results, with
48
+ individual bodies ranging from hundreds to tens of thousands of characters.
49
+ The old path loses all nine later results; the revised path must retain complete
50
+ results when the full input is within its declared capacity.
51
+
52
+ Verify the same evidence across adapter, pending cache, capture and inbox read;
53
+ then verify a supported later-source candidate can pass the actual processing
54
+ path in an isolated Vault. Check record/body/aggregate overflow separately,
55
+ including repeated reads, accurate loss accounting, no write from incomplete
56
+ evidence, and no cleanup of unresolved turns. Retain permission, redaction,
57
+ cross-turn isolation and standalone installation coverage.
58
+
59
+ Tests must use synthetic data and a deterministic backend. Live model semantics,
60
+ changes to a user's capture permission, installation and release are separate
61
+ operations with separately reported results.
62
+
63
+ ## Model capacity remains a separate limit
64
+
65
+ The capture limits bound retained source data, not the final prompt or model
66
+ token count. Evidence annotations, conversation context, retrieved context and
67
+ the output allowance also consume model capacity. The current `llm.context_window`
68
+ setting is not a pre-call tokenizer check. A source inventory within the capture
69
+ budget can still exceed a selected model's usable context.
70
+
71
+ Do not silently drop evidence to make that call succeed. Existing model-failure
72
+ handling must preserve the failed turn and prevent cleanup. Automatic batching
73
+ or model-specific token accounting needs its own design and is outside this
74
+ capture-budget change.
@@ -17,12 +17,20 @@ legacy boolean (false/absent -> metadata, true -> bounded). An explicit new mode
17
17
  wins over the legacy boolean. Invalid strings and non-boolean flags fail closed.
18
18
  Loading config does not rewrite it. Normal saves make the effective mode explicit.
19
19
 
20
- The inherited limits remain eight records, 2,000 characters per body, and 320 per
21
- metadata field. Oversized eligible evidence remains incomplete; its prefix is
22
- never promoted into a complete fact. Policy-excluded observations are marked
20
+ The shared capture budget allows 64 source records, 32 KiB of UTF-8 text per body,
21
+ 128 KiB of total body text, and 320 characters per metadata field. Loss markers
22
+ have a separate allowance of 64 call identities plus one aggregate marker, with
23
+ fixed diagnostic bodies. Reapplying normalization does not reduce an
24
+ already bounded inventory or count existing omissions again. Oversized eligible
25
+ evidence remains incomplete; its prefix is never promoted into a complete fact.
26
+ Policy-excluded observations are marked
23
27
  retention=metadata, have no content, and are not retried as missing evidence.
24
28
  No later relaxation recreates discarded original content.
25
29
 
30
+ These are source-retention limits, not a guarantee that every configured model
31
+ can process the resulting prompt. See [capture budget design](capture-budget-design.md)
32
+ for the ingestion boundary, model-capacity limitation and acceptance contract.
33
+
26
34
  Document/attachment bodies require include_attachments=true and bounded mode.
27
35
  HostRuntime and the standalone Hermes adapter classify structural file arguments
28
36
  using the same tested contract (path/file_path/file_id/attachment_id/file URI,
@@ -80,12 +80,15 @@ labels them NO_CHANGE. Incomplete turns retain their source instead of being
80
80
  cleaned after the usual grace period. Scope-filtered retries may revisit them;
81
81
  no endless automatic model retry or extra external tool call is introduced.
82
82
 
83
- Tool evidence is bounded to eight records with at most 2,000 content characters
84
- per captured result record and 320-character metadata fields, with redaction at
85
- Core capture. Large unambiguous top-level record collections retain complete
86
- records and enclosing context within that budget. Per-record provenance takes
87
- precedence over common source metadata. An overflow slot reports omitted
88
- records. Arbitrary large prose is not split into falsely complete facts;
83
+ Tool evidence uses one shared budget: 64 source records, at most 32 KiB of UTF-8
84
+ body text per record and 128 KiB of total body text, plus 320-character metadata
85
+ fields, with redaction at Core capture. Cache and inbox normalization are
86
+ idempotent under this budget. Large unambiguous top-level record collections
87
+ retain complete records and enclosing context within that budget. Per-record provenance takes
88
+ precedence over common source metadata. Separate loss markers report omitted
89
+ records, with at most 64 call identities plus one aggregate marker. These markers
90
+ have fixed diagnostic bodies and cannot supply external facts.
91
+ Arbitrary large prose is not split into falsely complete facts;
89
92
  unsupported/incomplete content needs a supported complete source excerpt or a
90
93
  later source input. Switching from `metadata` to `bounded` does not reconstruct
91
94
  body content that was never captured; bounded retention can still produce an
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "memleaf"
7
- version = "0.2.30"
7
+ version = "0.2.31"
8
8
  description = "A local-first Markdown memory core for AI agents"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -1,6 +1,6 @@
1
1
  """Local-first Markdown memory core for AI agents."""
2
2
 
3
- __version__ = "0.2.30"
3
+ __version__ = "0.2.31"
4
4
 
5
5
  from .config import DEFAULT_CONFIG, default_config, load_config, save_config
6
6
  from .frontmatter import FrontmatterError, dump_frontmatter, dump_yaml, load_yaml, parse_frontmatter
@@ -0,0 +1,294 @@
1
+ """Pure standard-library evidence capture budget primitives.
2
+
3
+ The Core package and the copied Hermes provider use this module as the one
4
+ shared retention boundary. Budgets are measured in UTF-8 bytes for bodies;
5
+ the loss marker is metadata and is deliberately outside both budgets.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Any, Mapping, Sequence
12
+
13
+
14
+ DEFAULT_MAX_RECORDS = 64
15
+ DEFAULT_MAX_RECORD_BYTES = 32 * 1024
16
+ DEFAULT_MAX_TOTAL_BYTES = 128 * 1024
17
+
18
+ _OVERFLOW_TOOL = "evidence.inventory"
19
+ _OVERFLOW_CALL_IDS = frozenset({"overflow", "retention-overflow"})
20
+ _LOSS_MARKER_CONTENT = "Additional tool observations exceeded the capture budget."
21
+ _MARKER_FIELDS = frozenset({
22
+ "call_id", "execution_status", "schema_version", "retention",
23
+ })
24
+
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class EvidenceBudget:
29
+ """The bounded evidence shape shared by Core and the Hermes copy."""
30
+
31
+ max_records: int = DEFAULT_MAX_RECORDS
32
+ max_record_bytes: int = DEFAULT_MAX_RECORD_BYTES
33
+ max_total_bytes: int = DEFAULT_MAX_TOTAL_BYTES
34
+
35
+ def __post_init__(self) -> None:
36
+ for field in ("max_records", "max_record_bytes", "max_total_bytes"):
37
+ value = getattr(self, field)
38
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
39
+ raise ValueError(f"invalid evidence budget {field}")
40
+
41
+
42
+ DEFAULT_BUDGET = EvidenceBudget()
43
+
44
+
45
+ def utf8_size(value: str) -> int:
46
+ """Return the exact UTF-8 byte size of one body string."""
47
+
48
+ if not isinstance(value, str):
49
+ raise TypeError("evidence body must be text")
50
+ return len(value.encode("utf-8"))
51
+
52
+
53
+ def truncate_utf8(value: str, limit: int) -> tuple[str, bool]:
54
+ """Safely truncate text to ``limit`` UTF-8 bytes at a codepoint boundary."""
55
+
56
+ if not isinstance(value, str):
57
+ raise TypeError("evidence body must be text")
58
+ if isinstance(limit, bool) or not isinstance(limit, int) or limit < 0:
59
+ raise ValueError("invalid UTF-8 byte limit")
60
+ encoded = value.encode("utf-8")
61
+ if len(encoded) <= limit:
62
+ return value, False
63
+ # ``ignore`` removes at most the incomplete final codepoint created by the
64
+ # byte slice. It never removes a complete codepoint before the boundary.
65
+ return encoded[:limit].decode("utf-8", "ignore"), True
66
+
67
+
68
+ def is_overflow_marker(value: Mapping[str, Any]) -> bool:
69
+ """Recognize the canonical and legacy loss markers without body matching."""
70
+
71
+ if not isinstance(value, Mapping):
72
+ return False
73
+ if value.get("tool_name") == _OVERFLOW_TOOL:
74
+ call_id = value.get("call_id")
75
+ if call_id in _OVERFLOW_CALL_IDS:
76
+ return True
77
+ return value.get("record_id") == "overflow" and (
78
+ value.get("omitted_count") is not None or value.get("omitted_bytes") is not None
79
+ )
80
+
81
+
82
+ def marker_count(value: Mapping[str, Any]) -> int:
83
+ """Return the audited number of observations represented by a marker."""
84
+
85
+ raw = value.get("omitted_count", "0")
86
+ if isinstance(raw, str) and raw.isascii() and raw.isdigit():
87
+ return int(raw)
88
+ if isinstance(raw, int) and not isinstance(raw, bool) and raw >= 0:
89
+ return raw
90
+ return 0
91
+
92
+
93
+ def marker_bytes(value: Mapping[str, Any]) -> int:
94
+ """Return the audited number of omitted body bytes represented by a marker."""
95
+
96
+ raw = value.get("omitted_bytes", "0")
97
+ if isinstance(raw, str) and raw.isascii() and raw.isdigit():
98
+ return int(raw)
99
+ if isinstance(raw, int) and not isinstance(raw, bool) and raw >= 0:
100
+ return raw
101
+ return 0
102
+
103
+
104
+ def logical_observation_count(value: Mapping[str, Any]) -> int:
105
+ """Count one ordinary record or the observations named by a loss marker."""
106
+
107
+ return marker_count(value) if is_overflow_marker(value) else 1
108
+
109
+
110
+ def _body_bytes(value: Mapping[str, Any]) -> int:
111
+ body = value.get("content")
112
+ return utf8_size(body) if isinstance(body, str) else 0
113
+
114
+
115
+ def _marker_from(
116
+ markers: Sequence[Mapping[str, Any]],
117
+ *,
118
+ omitted_count: int,
119
+ omitted_bytes: int,
120
+ ) -> dict[str, str]:
121
+ """Merge loss markers while preserving the first marker's audit identity."""
122
+
123
+ marker: dict[str, str] = {}
124
+ if markers:
125
+ first = markers[0]
126
+ for field in _MARKER_FIELDS:
127
+ value = first.get(field)
128
+ if isinstance(value, str) and value:
129
+ marker[field] = value[:320]
130
+ marker["call_id"] = str(first.get("call_id") or "overflow")[:320]
131
+ else:
132
+ marker["call_id"] = "overflow"
133
+ total_count = sum(marker_count(item) for item in markers) + omitted_count
134
+ total_bytes = sum(marker_bytes(item) for item in markers) + omitted_bytes
135
+ # A marker is always an untrusted inventory diagnostic, even if a caller
136
+ # supplied an ``overflow`` row with authority-looking fields. Keep only a
137
+ # small fixed identity and force the non-authoritative state.
138
+ marker["tool_name"] = _OVERFLOW_TOOL
139
+ marker["record_id"] = "overflow"
140
+ marker["kind"] = "unknown"
141
+ marker["result_status"] = "truncated"
142
+ marker["completeness"] = "partial"
143
+ marker["omitted_count"] = str(total_count)
144
+ if total_bytes or any("omitted_bytes" in item for item in markers):
145
+ marker["omitted_bytes"] = str(total_bytes)
146
+ # Metadata mode deliberately removes the body. Do not recreate one on a
147
+ # later normalization pass; bounded markers use one fixed diagnostic body.
148
+ if markers and any("content" in item for item in markers) or omitted_count or omitted_bytes:
149
+ marker["content"] = _LOSS_MARKER_CONTENT
150
+ return marker
151
+
152
+
153
+ def apply_evidence_budget(
154
+ value: Sequence[Mapping[str, Any]],
155
+ *,
156
+ budget: EvidenceBudget = DEFAULT_BUDGET,
157
+ ) -> list[dict[str, str]]:
158
+ """Apply one idempotent record/body/aggregate evidence budget.
159
+
160
+ The input is expected to contain already validated records. This helper
161
+ only owns budget accounting: ordinary metadata stays untouched, body text
162
+ is bounded in UTF-8 bytes, and loss markers are excluded from both record
163
+ and body budgets. Passing its output through again is stable.
164
+ """
165
+
166
+ if not isinstance(value, (list, tuple)):
167
+ raise ValueError("tool evidence must be a list")
168
+ if not isinstance(budget, EvidenceBudget):
169
+ raise TypeError("budget must be an EvidenceBudget")
170
+
171
+ markers: list[Mapping[str, Any]] = []
172
+ records: list[dict[str, str]] = []
173
+ for raw in value:
174
+ if not isinstance(raw, Mapping):
175
+ continue
176
+ if is_overflow_marker(raw):
177
+ markers.append(raw)
178
+ else:
179
+ records.append(dict(raw))
180
+
181
+ omitted_count = 0
182
+ omitted_bytes = 0
183
+ if len(records) > budget.max_records:
184
+ for row in records[budget.max_records :]:
185
+ omitted_count += logical_observation_count(row)
186
+ omitted_bytes += _body_bytes(row)
187
+ records = records[: budget.max_records]
188
+
189
+ retained: list[dict[str, str]] = []
190
+ remaining = budget.max_total_bytes
191
+ for row in records:
192
+ body = row.get("content")
193
+ if not isinstance(body, str):
194
+ retained.append(row)
195
+ continue
196
+ # Core historically strips evidence fields before persistence. Keep
197
+ # that canonical boundary in the shared helper so a provider pass and
198
+ # a Core pass cannot disagree at a trailing-space truncation edge.
199
+ body = body.strip()
200
+ if not body:
201
+ row.pop("content", None)
202
+ retained.append(row)
203
+ continue
204
+ original_bytes = utf8_size(body)
205
+ body, per_record_loss = truncate_utf8(body, budget.max_record_bytes)
206
+ if per_record_loss:
207
+ canonical_body = body.strip()
208
+ body = canonical_body
209
+ row["content"] = body
210
+ row["result_status"] = "truncated"
211
+ row["completeness"] = "partial"
212
+ omitted_bytes += original_bytes - utf8_size(body)
213
+ if not body:
214
+ omitted_count += logical_observation_count(row)
215
+ continue
216
+
217
+ body_bytes = utf8_size(body)
218
+ if body_bytes <= remaining:
219
+ row["content"] = body
220
+ retained.append(row)
221
+ remaining -= body_bytes
222
+ continue
223
+
224
+ if remaining > 0:
225
+ body, aggregate_loss = truncate_utf8(body, remaining)
226
+ canonical_body = body.strip()
227
+ body = canonical_body
228
+ if not body:
229
+ omitted_count += logical_observation_count(row)
230
+ omitted_bytes += body_bytes
231
+ continue
232
+ row["content"] = body
233
+ if aggregate_loss:
234
+ row["result_status"] = "truncated"
235
+ row["completeness"] = "partial"
236
+ omitted_bytes += body_bytes - utf8_size(body)
237
+ retained.append(row)
238
+ remaining = 0
239
+ else:
240
+ omitted_count += logical_observation_count(row)
241
+ omitted_bytes += body_bytes
242
+
243
+ if markers or omitted_count or omitted_bytes:
244
+ # Keep distinct host call markers distinct so a later repeated hook
245
+ # can be deduplicated by ``call_id``. Markers with the same identity
246
+ # are merged. Marker identities have their own fixed allowance: keep
247
+ # at most one row per ordinary record slot and summarize every excess
248
+ # marker into one global row. This prevents an unbounded stream of
249
+ # distinct host call IDs from bypassing the capture budget.
250
+ groups: dict[tuple[Any, Any], list[Mapping[str, Any]]] = {}
251
+ for marker in markers:
252
+ key = (marker.get("call_id"), marker.get("record_id"))
253
+ groups.setdefault(key, []).append(marker)
254
+ global_groups: list[Mapping[str, Any]] = []
255
+ call_groups: list[list[Mapping[str, Any]]] = []
256
+ for key, group in groups.items():
257
+ if key[0] in _OVERFLOW_CALL_IDS:
258
+ global_groups.extend(group)
259
+ else:
260
+ call_groups.append(group)
261
+ retained_groups = call_groups[:budget.max_records]
262
+ dropped_groups = call_groups[budget.max_records:]
263
+ dropped_count = sum(marker_count(item) for group in dropped_groups for item in group)
264
+ dropped_bytes = sum(marker_bytes(item) for group in dropped_groups for item in group)
265
+ marker_rows = [
266
+ _marker_from(group, omitted_count=0, omitted_bytes=0)
267
+ for group in retained_groups
268
+ ]
269
+ if global_groups or dropped_groups or omitted_count or omitted_bytes:
270
+ marker_rows.append(
271
+ _marker_from(
272
+ global_groups,
273
+ omitted_count=dropped_count + omitted_count,
274
+ omitted_bytes=dropped_bytes + omitted_bytes,
275
+ )
276
+ )
277
+ retained.extend(marker_rows)
278
+ return retained
279
+
280
+
281
+ __all__ = [
282
+ "DEFAULT_BUDGET",
283
+ "DEFAULT_MAX_RECORD_BYTES",
284
+ "DEFAULT_MAX_RECORDS",
285
+ "DEFAULT_MAX_TOTAL_BYTES",
286
+ "EvidenceBudget",
287
+ "apply_evidence_budget",
288
+ "is_overflow_marker",
289
+ "logical_observation_count",
290
+ "marker_bytes",
291
+ "marker_count",
292
+ "truncate_utf8",
293
+ "utf8_size",
294
+ ]
@@ -9,12 +9,14 @@ same local ``~/.memleaf`` vault as the standalone MCP server.
9
9
  from __future__ import annotations
10
10
 
11
11
  import json
12
+ import importlib.util
12
13
  import logging
13
14
  import os
14
15
  import re
15
16
  import queue
16
17
  import shutil
17
18
  import subprocess
19
+ import sys
18
20
  import threading
19
21
  import time
20
22
  from collections import OrderedDict, deque
@@ -24,6 +26,23 @@ from typing import Any, Deque, Dict, List, Mapping, Optional, Tuple
24
26
 
25
27
  from agent.memory_provider import MemoryProvider, RecallStatus
26
28
 
29
+ try:
30
+ from .evidence_budget import apply_evidence_budget
31
+ except (ImportError, ValueError):
32
+ # Hermes installs this file as a standalone plugin directory. Load the
33
+ # adjacent copied module directly so the provider never imports Core (or
34
+ # Hermes' package initializer) just to apply the capture budget.
35
+ _BUDGET_SPEC = importlib.util.spec_from_file_location(
36
+ "_memleaf_hermes_evidence_budget",
37
+ Path(__file__).with_name("evidence_budget.py"),
38
+ )
39
+ if _BUDGET_SPEC is None or _BUDGET_SPEC.loader is None:
40
+ raise ImportError("Hermes evidence budget module is unavailable")
41
+ _BUDGET_MODULE = importlib.util.module_from_spec(_BUDGET_SPEC)
42
+ sys.modules[_BUDGET_SPEC.name] = _BUDGET_MODULE
43
+ _BUDGET_SPEC.loader.exec_module(_BUDGET_MODULE)
44
+ apply_evidence_budget = _BUDGET_MODULE.apply_evidence_budget
45
+
27
46
  logger = logging.getLogger(__name__)
28
47
 
29
48
  _DEFAULT_VAULT = "~/.memleaf"
@@ -1179,8 +1198,9 @@ def _bounded_current_tool_evidence(messages: Optional[List[Dict[str, Any]]], *,
1179
1198
  """Match current-turn results strictly by call ID, not tool name/order.
1180
1199
 
1181
1200
  Kept standard-library-only: Hermes can load this copied provider while the
1182
- core runs in a separate environment. Core capture redacts and validates
1183
- these bounded records again before persistence.
1201
+ core runs in a separate environment. The adjacent shared budget module is
1202
+ the only body/record boundary; Core redacts and validates these records
1203
+ again before persistence using the same idempotent rule.
1184
1204
  """
1185
1205
  if not isinstance(messages, list):
1186
1206
  return []
@@ -1220,12 +1240,11 @@ def _bounded_current_tool_evidence(messages: Optional[List[Dict[str, Any]]], *,
1220
1240
  return None
1221
1241
  if not text or "\x00" in text:
1222
1242
  return None
1223
- partial = len(text) > 2000
1224
1243
  item = {"tool_name": name[:320], "call_id": cid[:320], "kind": kind,
1225
1244
  "execution_status": "error" if execution_error else "success",
1226
- "completeness": "partial" if partial else "complete", "schema_version": "2",
1227
- "result_status": "truncated" if partial else "error" if execution_error else "success",
1228
- "content": text[:2000],
1245
+ "completeness": "complete", "schema_version": "2",
1246
+ "result_status": "error" if execution_error else "success",
1247
+ "content": text,
1229
1248
  "source_type": "document" if _has_document_arguments(call.get("arguments")) else "tool_result"}
1230
1249
  if isinstance(value, Mapping):
1231
1250
  for key in ("record_id", "title", "message_id", "subject", "sender", "domain"):
@@ -1240,7 +1259,7 @@ def _bounded_current_tool_evidence(messages: Optional[List[Dict[str, Any]]], *,
1240
1259
  if original is None:
1241
1260
  continue
1242
1261
  collection, context = None, {}
1243
- if original["completeness"] == "partial" and not execution_error:
1262
+ if not execution_error:
1244
1263
  if isinstance(payload, list):
1245
1264
  collection = payload
1246
1265
  elif isinstance(payload, Mapping):
@@ -1249,7 +1268,7 @@ def _bounded_current_tool_evidence(messages: Optional[List[Dict[str, Any]]], *,
1249
1268
  collection = payload[keys[0]]
1250
1269
  context = {key: value for key, value in payload.items() if key != keys[0]}
1251
1270
  if collection:
1252
- for index, value in enumerate(collection[:8]):
1271
+ for index, value in enumerate(collection):
1253
1272
  item = record({"context": context, "record": value}, f"result-record-{index}")
1254
1273
  if item is not None:
1255
1274
  for field in ("message_id", "subject", "sender", "domain", "title"):
@@ -1259,20 +1278,10 @@ def _bounded_current_tool_evidence(messages: Optional[List[Dict[str, Any]]], *,
1259
1278
  if isinstance(field_value, str) and field_value.strip() and not any(ch in field_value for ch in "\x00\r\n"):
1260
1279
  item[field] = field_value[:320]
1261
1280
  output.append(item)
1262
- if len(collection) > 8:
1263
- output.append({"tool_name": "evidence.inventory", "call_id": cid[:320],
1264
- "record_id": "overflow", "kind": "unknown", "result_status": "truncated",
1265
- "completeness": "partial", "execution_status": "success", "schema_version": "2",
1266
- "omitted_count": str(len(collection) - 8), "content": "Additional structured observations exceeded the capture budget."})
1267
1281
  else:
1268
1282
  output.append(original)
1269
1283
  seen.add(cid)
1270
- if len(output) > 8:
1271
- omitted = sum(int(row.get("omitted_count", "1")) for row in output[7:])
1272
- output = output[:7] + [{"tool_name": "evidence.inventory", "call_id": "overflow",
1273
- "kind": "unknown", "result_status": "truncated", "completeness": "partial",
1274
- "omitted_count": str(omitted), "content": "Additional host observations exceeded the capture budget."}]
1275
- return output
1284
+ return apply_evidence_budget(output)
1276
1285
 
1277
1286
 
1278
1287
  class MemleafMemoryProvider(MemoryProvider):