@namzu/sdk 38.2.1 → 39.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (493) hide show
  1. package/CHANGELOG.md +700 -0
  2. package/dist/advisory/executor.d.ts +10 -1
  3. package/dist/advisory/executor.d.ts.map +1 -1
  4. package/dist/advisory/executor.js +5 -26
  5. package/dist/advisory/executor.js.map +1 -1
  6. package/dist/advisory/history.d.ts +9 -0
  7. package/dist/advisory/history.d.ts.map +1 -0
  8. package/dist/advisory/history.js +120 -0
  9. package/dist/advisory/history.js.map +1 -0
  10. package/dist/advisory/index.d.ts +1 -1
  11. package/dist/advisory/index.d.ts.map +1 -1
  12. package/dist/advisory/index.js.map +1 -1
  13. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  14. package/dist/agents/ReactiveAgent.js +3 -0
  15. package/dist/agents/ReactiveAgent.js.map +1 -1
  16. package/dist/agents/runAgent.d.ts +10 -0
  17. package/dist/agents/runAgent.d.ts.map +1 -1
  18. package/dist/agents/runAgent.js +3 -0
  19. package/dist/agents/runAgent.js.map +1 -1
  20. package/dist/compaction/manual.d.ts +6 -0
  21. package/dist/compaction/manual.d.ts.map +1 -1
  22. package/dist/compaction/manual.js +21 -2
  23. package/dist/compaction/manual.js.map +1 -1
  24. package/dist/compaction/summary.d.ts.map +1 -1
  25. package/dist/compaction/summary.js +4 -1
  26. package/dist/compaction/summary.js.map +1 -1
  27. package/dist/config/runtime.js +2 -2
  28. package/dist/config/runtime.js.map +1 -1
  29. package/dist/contracts/schemas.js +1 -1
  30. package/dist/contracts/schemas.js.map +1 -1
  31. package/dist/eval/harness-protection.d.ts +18 -0
  32. package/dist/eval/harness-protection.d.ts.map +1 -0
  33. package/dist/eval/harness-protection.js +58 -0
  34. package/dist/eval/harness-protection.js.map +1 -0
  35. package/dist/eval/harness-verification.d.ts +6 -1
  36. package/dist/eval/harness-verification.d.ts.map +1 -1
  37. package/dist/eval/harness-verification.js +19 -2
  38. package/dist/eval/harness-verification.js.map +1 -1
  39. package/dist/eval/index.d.ts +1 -0
  40. package/dist/eval/index.d.ts.map +1 -1
  41. package/dist/eval/index.js.map +1 -1
  42. package/dist/manager/resident/activity.d.ts +50 -0
  43. package/dist/manager/resident/activity.d.ts.map +1 -0
  44. package/dist/manager/resident/activity.js +125 -0
  45. package/dist/manager/resident/activity.js.map +1 -0
  46. package/dist/manager/resident/agenda.d.ts +13 -1
  47. package/dist/manager/resident/agenda.d.ts.map +1 -1
  48. package/dist/manager/resident/agenda.js +40 -5
  49. package/dist/manager/resident/agenda.js.map +1 -1
  50. package/dist/manager/resident/consumption.d.ts +104 -0
  51. package/dist/manager/resident/consumption.d.ts.map +1 -0
  52. package/dist/manager/resident/consumption.js +233 -0
  53. package/dist/manager/resident/consumption.js.map +1 -0
  54. package/dist/manager/resident/evidence-recall.d.ts +24 -0
  55. package/dist/manager/resident/evidence-recall.d.ts.map +1 -0
  56. package/dist/manager/resident/evidence-recall.js +295 -0
  57. package/dist/manager/resident/evidence-recall.js.map +1 -0
  58. package/dist/manager/resident/history-disk.d.ts +10 -0
  59. package/dist/manager/resident/history-disk.d.ts.map +1 -0
  60. package/dist/manager/resident/history-disk.js +50 -0
  61. package/dist/manager/resident/history-disk.js.map +1 -0
  62. package/dist/manager/resident/history.d.ts +79 -0
  63. package/dist/manager/resident/history.d.ts.map +1 -0
  64. package/dist/manager/resident/history.js +203 -0
  65. package/dist/manager/resident/history.js.map +1 -0
  66. package/dist/manager/resident/initiative.d.ts.map +1 -1
  67. package/dist/manager/resident/initiative.js +11 -3
  68. package/dist/manager/resident/initiative.js.map +1 -1
  69. package/dist/manager/resident/learning-cycle.d.ts +131 -0
  70. package/dist/manager/resident/learning-cycle.d.ts.map +1 -0
  71. package/dist/manager/resident/learning-cycle.js +306 -0
  72. package/dist/manager/resident/learning-cycle.js.map +1 -0
  73. package/dist/manager/resident/learning-observation.d.ts +80 -0
  74. package/dist/manager/resident/learning-observation.d.ts.map +1 -0
  75. package/dist/manager/resident/learning-observation.js +22 -0
  76. package/dist/manager/resident/learning-observation.js.map +1 -0
  77. package/dist/manager/resident/learning-store.d.ts +106 -0
  78. package/dist/manager/resident/learning-store.d.ts.map +1 -0
  79. package/dist/manager/resident/learning-store.js +598 -0
  80. package/dist/manager/resident/learning-store.js.map +1 -0
  81. package/dist/manager/resident/learning.d.ts +246 -3
  82. package/dist/manager/resident/learning.d.ts.map +1 -1
  83. package/dist/manager/resident/learning.js +96 -6
  84. package/dist/manager/resident/learning.js.map +1 -1
  85. package/dist/manager/resident/outbox.d.ts +4 -4
  86. package/dist/manager/resident/store.d.ts +37 -4
  87. package/dist/manager/resident/store.d.ts.map +1 -1
  88. package/dist/manager/resident/store.js +27 -3
  89. package/dist/manager/resident/store.js.map +1 -1
  90. package/dist/manager/resident/tool-evidence.d.ts +71 -0
  91. package/dist/manager/resident/tool-evidence.d.ts.map +1 -0
  92. package/dist/manager/resident/tool-evidence.js +285 -0
  93. package/dist/manager/resident/tool-evidence.js.map +1 -0
  94. package/dist/manager/run/persistence.d.ts.map +1 -1
  95. package/dist/manager/run/persistence.js +6 -0
  96. package/dist/manager/run/persistence.js.map +1 -1
  97. package/dist/plugin/loader.d.ts.map +1 -1
  98. package/dist/plugin/loader.js +5 -3
  99. package/dist/plugin/loader.js.map +1 -1
  100. package/dist/prompt/coding-agent-doctrine.d.ts +1 -1
  101. package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
  102. package/dist/prompt/coding-agent-doctrine.js +2 -0
  103. package/dist/prompt/coding-agent-doctrine.js.map +1 -1
  104. package/dist/prompt/index.d.ts +2 -0
  105. package/dist/prompt/index.d.ts.map +1 -1
  106. package/dist/prompt/index.js +1 -0
  107. package/dist/prompt/index.js.map +1 -1
  108. package/dist/prompt/resident-learning.d.ts +19 -0
  109. package/dist/prompt/resident-learning.d.ts.map +1 -0
  110. package/dist/prompt/resident-learning.js +125 -0
  111. package/dist/prompt/resident-learning.js.map +1 -0
  112. package/dist/prompt/resident-step.d.ts +8 -1
  113. package/dist/prompt/resident-step.d.ts.map +1 -1
  114. package/dist/prompt/resident-step.js +65 -4
  115. package/dist/prompt/resident-step.js.map +1 -1
  116. package/dist/provider/collect-chat-completion.d.ts +2 -1
  117. package/dist/provider/collect-chat-completion.d.ts.map +1 -1
  118. package/dist/provider/collect-chat-completion.js +7 -6
  119. package/dist/provider/collect-chat-completion.js.map +1 -1
  120. package/dist/provider/fallback.d.ts.map +1 -1
  121. package/dist/provider/fallback.js +2 -1
  122. package/dist/provider/fallback.js.map +1 -1
  123. package/dist/provider/stream-text.d.ts +16 -0
  124. package/dist/provider/stream-text.d.ts.map +1 -0
  125. package/dist/provider/stream-text.js +51 -0
  126. package/dist/provider/stream-text.js.map +1 -0
  127. package/dist/public-runtime.d.ts +13 -3
  128. package/dist/public-runtime.d.ts.map +1 -1
  129. package/dist/public-runtime.js +12 -3
  130. package/dist/public-runtime.js.map +1 -1
  131. package/dist/public-tools.d.ts +2 -0
  132. package/dist/public-tools.d.ts.map +1 -1
  133. package/dist/public-tools.js +2 -0
  134. package/dist/public-tools.js.map +1 -1
  135. package/dist/public-types.d.ts +15 -3
  136. package/dist/public-types.d.ts.map +1 -1
  137. package/dist/run/LimitChecker.js +3 -3
  138. package/dist/run/LimitChecker.js.map +1 -1
  139. package/dist/run/evidence-query.d.ts +41 -0
  140. package/dist/run/evidence-query.d.ts.map +1 -0
  141. package/dist/run/evidence-query.js +270 -0
  142. package/dist/run/evidence-query.js.map +1 -0
  143. package/dist/run/evidence-recall.d.ts +99 -0
  144. package/dist/run/evidence-recall.d.ts.map +1 -0
  145. package/dist/run/evidence-recall.js +633 -0
  146. package/dist/run/evidence-recall.js.map +1 -0
  147. package/dist/run/index.d.ts +2 -0
  148. package/dist/run/index.d.ts.map +1 -1
  149. package/dist/run/index.js +1 -0
  150. package/dist/run/index.js.map +1 -1
  151. package/dist/run/json-claim-verifier.d.ts +83 -0
  152. package/dist/run/json-claim-verifier.d.ts.map +1 -0
  153. package/dist/run/json-claim-verifier.js +200 -0
  154. package/dist/run/json-claim-verifier.js.map +1 -0
  155. package/dist/run/preparation-context-error.d.ts +10 -0
  156. package/dist/run/preparation-context-error.d.ts.map +1 -0
  157. package/dist/run/preparation-context-error.js +15 -0
  158. package/dist/run/preparation-context-error.js.map +1 -0
  159. package/dist/run-query/index.d.ts +3 -1
  160. package/dist/run-query/index.d.ts.map +1 -1
  161. package/dist/runtime/query/callback-inference.d.ts +8 -0
  162. package/dist/runtime/query/callback-inference.d.ts.map +1 -0
  163. package/dist/runtime/query/callback-inference.js +89 -0
  164. package/dist/runtime/query/callback-inference.js.map +1 -0
  165. package/dist/runtime/query/checkpoint.d.ts +4 -0
  166. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  167. package/dist/runtime/query/checkpoint.js +13 -0
  168. package/dist/runtime/query/checkpoint.js.map +1 -1
  169. package/dist/runtime/query/events.d.ts +1 -0
  170. package/dist/runtime/query/events.d.ts.map +1 -1
  171. package/dist/runtime/query/events.js +18 -0
  172. package/dist/runtime/query/events.js.map +1 -1
  173. package/dist/runtime/query/executor.d.ts +7 -1
  174. package/dist/runtime/query/executor.d.ts.map +1 -1
  175. package/dist/runtime/query/executor.js +38 -7
  176. package/dist/runtime/query/executor.js.map +1 -1
  177. package/dist/runtime/query/file-evidence-context.d.ts +5 -0
  178. package/dist/runtime/query/file-evidence-context.d.ts.map +1 -0
  179. package/dist/runtime/query/file-evidence-context.js +64 -0
  180. package/dist/runtime/query/file-evidence-context.js.map +1 -0
  181. package/dist/runtime/query/guard.d.ts.map +1 -1
  182. package/dist/runtime/query/guard.js +4 -0
  183. package/dist/runtime/query/guard.js.map +1 -1
  184. package/dist/runtime/query/index.d.ts +10 -1
  185. package/dist/runtime/query/index.d.ts.map +1 -1
  186. package/dist/runtime/query/index.js +39 -24
  187. package/dist/runtime/query/index.js.map +1 -1
  188. package/dist/runtime/query/iteration/index.d.ts +9 -31
  189. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  190. package/dist/runtime/query/iteration/index.js +231 -93
  191. package/dist/runtime/query/iteration/index.js.map +1 -1
  192. package/dist/runtime/query/iteration/phases/advisory.d.ts +2 -1
  193. package/dist/runtime/query/iteration/phases/advisory.d.ts.map +1 -1
  194. package/dist/runtime/query/iteration/phases/advisory.js +2 -1
  195. package/dist/runtime/query/iteration/phases/advisory.js.map +1 -1
  196. package/dist/runtime/query/iteration/phases/context.d.ts +2 -1
  197. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  198. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  199. package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
  200. package/dist/runtime/query/iteration/phases/tool-review.js +3 -0
  201. package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
  202. package/dist/runtime/query/iteration/provider-rejected-image.d.ts +2 -1
  203. package/dist/runtime/query/iteration/provider-rejected-image.d.ts.map +1 -1
  204. package/dist/runtime/query/iteration/provider-rejected-image.js +5 -2
  205. package/dist/runtime/query/iteration/provider-rejected-image.js.map +1 -1
  206. package/dist/runtime/query/iteration/stream-turn.d.ts +4 -1
  207. package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
  208. package/dist/runtime/query/iteration/stream-turn.js +22 -8
  209. package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
  210. package/dist/runtime/query/resume-pending.d.ts +18 -33
  211. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  212. package/dist/runtime/query/resume-pending.js +59 -42
  213. package/dist/runtime/query/resume-pending.js.map +1 -1
  214. package/dist/runtime/query/review-policy.d.ts +4 -4
  215. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  216. package/dist/runtime/query/review-policy.js +10 -9
  217. package/dist/runtime/query/review-policy.js.map +1 -1
  218. package/dist/runtime/query/sandbox-lifecycle.d.ts.map +1 -1
  219. package/dist/runtime/query/sandbox-lifecycle.js +4 -0
  220. package/dist/runtime/query/sandbox-lifecycle.js.map +1 -1
  221. package/dist/runtime/query/tool-output-budget.d.ts +10 -6
  222. package/dist/runtime/query/tool-output-budget.d.ts.map +1 -1
  223. package/dist/runtime/query/tool-output-budget.js +54 -12
  224. package/dist/runtime/query/tool-output-budget.js.map +1 -1
  225. package/dist/runtime/query/tooling.d.ts +3 -1
  226. package/dist/runtime/query/tooling.d.ts.map +1 -1
  227. package/dist/runtime/query/tooling.js +4 -0
  228. package/dist/runtime/query/tooling.js.map +1 -1
  229. package/dist/scheduler/completion-inbox.d.ts +3 -0
  230. package/dist/scheduler/completion-inbox.d.ts.map +1 -1
  231. package/dist/scheduler/completion-inbox.js +28 -0
  232. package/dist/scheduler/completion-inbox.js.map +1 -1
  233. package/dist/store/evidence/compaction-archive.d.ts +109 -0
  234. package/dist/store/evidence/compaction-archive.d.ts.map +1 -0
  235. package/dist/store/evidence/compaction-archive.js +125 -0
  236. package/dist/store/evidence/compaction-archive.js.map +1 -0
  237. package/dist/store/evidence/compaction-provenance.d.ts +7 -0
  238. package/dist/store/evidence/compaction-provenance.d.ts.map +1 -0
  239. package/dist/store/evidence/compaction-provenance.js +49 -0
  240. package/dist/store/evidence/compaction-provenance.js.map +1 -0
  241. package/dist/store/evidence/compaction-text.d.ts +16 -0
  242. package/dist/store/evidence/compaction-text.d.ts.map +1 -0
  243. package/dist/store/evidence/compaction-text.js +51 -0
  244. package/dist/store/evidence/compaction-text.js.map +1 -0
  245. package/dist/store/evidence/disk.d.ts +11 -0
  246. package/dist/store/evidence/disk.d.ts.map +1 -0
  247. package/dist/store/evidence/disk.js +367 -0
  248. package/dist/store/evidence/disk.js.map +1 -0
  249. package/dist/store/evidence/format.d.ts +25 -0
  250. package/dist/store/evidence/format.d.ts.map +1 -0
  251. package/dist/store/evidence/format.js +115 -0
  252. package/dist/store/evidence/format.js.map +1 -0
  253. package/dist/store/evidence/index-page.d.ts +208 -0
  254. package/dist/store/evidence/index-page.d.ts.map +1 -0
  255. package/dist/store/evidence/index-page.js +262 -0
  256. package/dist/store/evidence/index-page.js.map +1 -0
  257. package/dist/store/evidence/io.d.ts +24 -0
  258. package/dist/store/evidence/io.d.ts.map +1 -0
  259. package/dist/store/evidence/io.js +69 -0
  260. package/dist/store/evidence/io.js.map +1 -0
  261. package/dist/store/evidence/linked.d.ts +9 -0
  262. package/dist/store/evidence/linked.d.ts.map +1 -0
  263. package/dist/store/evidence/linked.js +314 -0
  264. package/dist/store/evidence/linked.js.map +1 -0
  265. package/dist/store/evidence/passages.d.ts +15 -0
  266. package/dist/store/evidence/passages.d.ts.map +1 -0
  267. package/dist/store/evidence/passages.js +72 -0
  268. package/dist/store/evidence/passages.js.map +1 -0
  269. package/dist/store/evidence/record-chain.d.ts +42 -0
  270. package/dist/store/evidence/record-chain.d.ts.map +1 -0
  271. package/dist/store/evidence/record-chain.js +111 -0
  272. package/dist/store/evidence/record-chain.js.map +1 -0
  273. package/dist/store/evidence/search-input.d.ts +21 -0
  274. package/dist/store/evidence/search-input.d.ts.map +1 -0
  275. package/dist/store/evidence/search-input.js +44 -0
  276. package/dist/store/evidence/search-input.js.map +1 -0
  277. package/dist/store/evidence/selection.d.ts +9 -0
  278. package/dist/store/evidence/selection.d.ts.map +1 -0
  279. package/dist/store/evidence/selection.js +17 -0
  280. package/dist/store/evidence/selection.js.map +1 -0
  281. package/dist/store/evidence/source-kind.d.ts +11 -0
  282. package/dist/store/evidence/source-kind.d.ts.map +1 -0
  283. package/dist/store/evidence/source-kind.js +26 -0
  284. package/dist/store/evidence/source-kind.js.map +1 -0
  285. package/dist/store/evidence/source-text.d.ts +51 -0
  286. package/dist/store/evidence/source-text.d.ts.map +1 -0
  287. package/dist/store/evidence/source-text.js +213 -0
  288. package/dist/store/evidence/source-text.js.map +1 -0
  289. package/dist/store/evidence/types.d.ts +162 -0
  290. package/dist/store/evidence/types.d.ts.map +1 -0
  291. package/dist/store/evidence/types.js +2 -0
  292. package/dist/store/evidence/types.js.map +1 -0
  293. package/dist/store/memory/disk.d.ts +2 -0
  294. package/dist/store/memory/disk.d.ts.map +1 -1
  295. package/dist/store/memory/disk.js +2 -1
  296. package/dist/store/memory/disk.js.map +1 -1
  297. package/dist/store/run/disk.d.ts +10 -0
  298. package/dist/store/run/disk.d.ts.map +1 -1
  299. package/dist/store/run/disk.js +87 -7
  300. package/dist/store/run/disk.js.map +1 -1
  301. package/dist/store/run/memory.d.ts +1 -0
  302. package/dist/store/run/memory.d.ts.map +1 -1
  303. package/dist/store/run/memory.js +10 -0
  304. package/dist/store/run/memory.js.map +1 -1
  305. package/dist/store/run/tool-executions.d.ts +13 -0
  306. package/dist/store/run/tool-executions.d.ts.map +1 -0
  307. package/dist/store/run/tool-executions.js +99 -0
  308. package/dist/store/run/tool-executions.js.map +1 -0
  309. package/dist/store/session/index.d.ts +2 -0
  310. package/dist/store/session/index.d.ts.map +1 -1
  311. package/dist/store/session/index.js +1 -0
  312. package/dist/store/session/index.js.map +1 -1
  313. package/dist/store/session/sqlite.d.ts +57 -0
  314. package/dist/store/session/sqlite.d.ts.map +1 -0
  315. package/dist/store/session/sqlite.js +430 -0
  316. package/dist/store/session/sqlite.js.map +1 -0
  317. package/dist/tools/builtins/job.d.ts.map +1 -1
  318. package/dist/tools/builtins/job.js +4 -5
  319. package/dist/tools/builtins/job.js.map +1 -1
  320. package/dist/tools/builtins/write-file.js +2 -2
  321. package/dist/tools/builtins/write-file.js.map +1 -1
  322. package/dist/tools/defineTool.d.ts +2 -1
  323. package/dist/tools/defineTool.d.ts.map +1 -1
  324. package/dist/tools/defineTool.js +1 -1
  325. package/dist/tools/defineTool.js.map +1 -1
  326. package/dist/tools/file-read-tracker.d.ts.map +1 -1
  327. package/dist/tools/file-read-tracker.js +12 -3
  328. package/dist/tools/file-read-tracker.js.map +1 -1
  329. package/dist/tools/resident-history.d.ts +10 -0
  330. package/dist/tools/resident-history.d.ts.map +1 -0
  331. package/dist/tools/resident-history.js +80 -0
  332. package/dist/tools/resident-history.js.map +1 -0
  333. package/dist/tools/resident-tool-evidence.d.ts +5 -0
  334. package/dist/tools/resident-tool-evidence.d.ts.map +1 -0
  335. package/dist/tools/resident-tool-evidence.js +57 -0
  336. package/dist/tools/resident-tool-evidence.js.map +1 -0
  337. package/dist/types/advisory/config.d.ts +7 -0
  338. package/dist/types/advisory/config.d.ts.map +1 -1
  339. package/dist/types/agent/reactive.d.ts +1 -0
  340. package/dist/types/agent/reactive.d.ts.map +1 -1
  341. package/dist/types/authorization/index.d.ts +9 -9
  342. package/dist/types/authorization/index.d.ts.map +1 -1
  343. package/dist/types/authorization/index.js +1 -1
  344. package/dist/types/authorization/index.js.map +1 -1
  345. package/dist/types/hitl/index.d.ts +4 -0
  346. package/dist/types/hitl/index.d.ts.map +1 -1
  347. package/dist/types/hitl/index.js.map +1 -1
  348. package/dist/types/message/index.d.ts +16 -2
  349. package/dist/types/message/index.d.ts.map +1 -1
  350. package/dist/types/message/index.js +8 -1
  351. package/dist/types/message/index.js.map +1 -1
  352. package/dist/types/provider/chat.d.ts +2 -0
  353. package/dist/types/provider/chat.d.ts.map +1 -1
  354. package/dist/types/provider/stream.d.ts +4 -0
  355. package/dist/types/provider/stream.d.ts.map +1 -1
  356. package/dist/types/run/answer-review.d.ts +34 -3
  357. package/dist/types/run/answer-review.d.ts.map +1 -1
  358. package/dist/types/run/config.d.ts +3 -0
  359. package/dist/types/run/config.d.ts.map +1 -1
  360. package/dist/types/run/entity.d.ts +7 -0
  361. package/dist/types/run/entity.d.ts.map +1 -1
  362. package/dist/types/run/events.d.ts +31 -8
  363. package/dist/types/run/events.d.ts.map +1 -1
  364. package/dist/types/run/events.js.map +1 -1
  365. package/dist/types/run/prepare-step.d.ts +53 -8
  366. package/dist/types/run/prepare-step.d.ts.map +1 -1
  367. package/dist/types/run/store.d.ts +29 -4
  368. package/dist/types/run/store.d.ts.map +1 -1
  369. package/dist/types/run/store.js +0 -28
  370. package/dist/types/run/store.js.map +1 -1
  371. package/dist/types/tool/index.d.ts +15 -1
  372. package/dist/types/tool/index.d.ts.map +1 -1
  373. package/dist/types/tool/index.js.map +1 -1
  374. package/dist/utils/await-with-abort.d.ts +8 -0
  375. package/dist/utils/await-with-abort.d.ts.map +1 -0
  376. package/dist/utils/await-with-abort.js +28 -0
  377. package/dist/utils/await-with-abort.js.map +1 -0
  378. package/dist/utils/evidence-time.d.ts +3 -0
  379. package/dist/utils/evidence-time.d.ts.map +1 -0
  380. package/dist/utils/evidence-time.js +10 -0
  381. package/dist/utils/evidence-time.js.map +1 -0
  382. package/dist/utils/evidence-tokens.d.ts +13 -0
  383. package/dist/utils/evidence-tokens.d.ts.map +1 -0
  384. package/dist/utils/evidence-tokens.js +29 -0
  385. package/dist/utils/evidence-tokens.js.map +1 -0
  386. package/package.json +1 -1
  387. package/src/advisory/executor.ts +15 -30
  388. package/src/advisory/history.ts +126 -0
  389. package/src/advisory/index.ts +5 -1
  390. package/src/agents/ReactiveAgent.ts +3 -0
  391. package/src/agents/runAgent.ts +14 -1
  392. package/src/compaction/manual.ts +30 -2
  393. package/src/compaction/summary.ts +4 -1
  394. package/src/config/runtime.ts +2 -2
  395. package/src/contracts/schemas.ts +1 -1
  396. package/src/eval/harness-protection.ts +79 -0
  397. package/src/eval/harness-verification.ts +31 -1
  398. package/src/eval/index.ts +1 -0
  399. package/src/manager/resident/activity.ts +187 -0
  400. package/src/manager/resident/agenda.ts +59 -5
  401. package/src/manager/resident/consumption.ts +311 -0
  402. package/src/manager/resident/evidence-recall.ts +363 -0
  403. package/src/manager/resident/history-disk.ts +63 -0
  404. package/src/manager/resident/history.ts +312 -0
  405. package/src/manager/resident/initiative.ts +13 -3
  406. package/src/manager/resident/learning-cycle.ts +499 -0
  407. package/src/manager/resident/learning-observation.ts +39 -0
  408. package/src/manager/resident/learning-store.ts +813 -0
  409. package/src/manager/resident/learning.ts +132 -8
  410. package/src/manager/resident/store.ts +31 -3
  411. package/src/manager/resident/tool-evidence.ts +412 -0
  412. package/src/manager/run/persistence.ts +6 -0
  413. package/src/plugin/loader.ts +8 -3
  414. package/src/prompt/coding-agent-doctrine.ts +2 -0
  415. package/src/prompt/index.ts +2 -0
  416. package/src/prompt/resident-learning.ts +143 -0
  417. package/src/prompt/resident-step.ts +83 -3
  418. package/src/provider/collect-chat-completion.ts +7 -6
  419. package/src/provider/fallback.ts +2 -1
  420. package/src/provider/stream-text.ts +56 -0
  421. package/src/public-runtime.ts +30 -0
  422. package/src/public-tools.ts +3 -0
  423. package/src/public-types.ts +98 -0
  424. package/src/run/LimitChecker.ts +3 -3
  425. package/src/run/evidence-query.ts +333 -0
  426. package/src/run/evidence-recall.ts +854 -0
  427. package/src/run/index.ts +11 -0
  428. package/src/run/json-claim-verifier.ts +298 -0
  429. package/src/run/preparation-context-error.ts +16 -0
  430. package/src/run-query/index.ts +1 -1
  431. package/src/runtime/query/callback-inference.ts +94 -0
  432. package/src/runtime/query/checkpoint.ts +14 -0
  433. package/src/runtime/query/events.ts +22 -0
  434. package/src/runtime/query/executor.ts +51 -9
  435. package/src/runtime/query/file-evidence-context.ts +65 -0
  436. package/src/runtime/query/guard.ts +2 -0
  437. package/src/runtime/query/index.ts +49 -25
  438. package/src/runtime/query/iteration/index.ts +269 -86
  439. package/src/runtime/query/iteration/phases/advisory.ts +3 -0
  440. package/src/runtime/query/iteration/phases/context.ts +2 -0
  441. package/src/runtime/query/iteration/phases/tool-review.ts +3 -0
  442. package/src/runtime/query/iteration/provider-rejected-image.ts +6 -1
  443. package/src/runtime/query/iteration/stream-turn.ts +35 -8
  444. package/src/runtime/query/resume-pending.ts +65 -40
  445. package/src/runtime/query/review-policy.ts +13 -9
  446. package/src/runtime/query/sandbox-lifecycle.ts +3 -0
  447. package/src/runtime/query/tool-output-budget.ts +65 -13
  448. package/src/runtime/query/tooling.ts +7 -1
  449. package/src/scheduler/completion-inbox.ts +26 -0
  450. package/src/store/evidence/compaction-archive.ts +139 -0
  451. package/src/store/evidence/compaction-provenance.ts +52 -0
  452. package/src/store/evidence/compaction-text.ts +61 -0
  453. package/src/store/evidence/disk.ts +461 -0
  454. package/src/store/evidence/format.ts +126 -0
  455. package/src/store/evidence/index-page.ts +292 -0
  456. package/src/store/evidence/io.ts +95 -0
  457. package/src/store/evidence/linked.ts +365 -0
  458. package/src/store/evidence/passages.ts +88 -0
  459. package/src/store/evidence/record-chain.ts +108 -0
  460. package/src/store/evidence/search-input.ts +62 -0
  461. package/src/store/evidence/selection.ts +25 -0
  462. package/src/store/evidence/source-kind.ts +36 -0
  463. package/src/store/evidence/source-text.ts +285 -0
  464. package/src/store/evidence/types.ts +177 -0
  465. package/src/store/memory/disk.ts +4 -1
  466. package/src/store/run/disk.ts +110 -8
  467. package/src/store/run/memory.ts +10 -0
  468. package/src/store/run/tool-executions.ts +112 -0
  469. package/src/store/session/index.ts +2 -0
  470. package/src/store/session/sqlite.ts +584 -0
  471. package/src/tools/builtins/job.ts +4 -5
  472. package/src/tools/builtins/write-file.ts +2 -2
  473. package/src/tools/defineTool.ts +4 -2
  474. package/src/tools/file-read-tracker.ts +8 -2
  475. package/src/tools/resident-history.ts +86 -0
  476. package/src/tools/resident-tool-evidence.ts +68 -0
  477. package/src/types/advisory/config.ts +7 -0
  478. package/src/types/agent/reactive.ts +1 -0
  479. package/src/types/authorization/index.ts +2 -2
  480. package/src/types/hitl/index.ts +4 -0
  481. package/src/types/message/index.ts +20 -0
  482. package/src/types/provider/chat.ts +2 -0
  483. package/src/types/provider/stream.ts +4 -0
  484. package/src/types/run/answer-review.ts +34 -3
  485. package/src/types/run/config.ts +3 -0
  486. package/src/types/run/entity.ts +7 -0
  487. package/src/types/run/events.ts +31 -8
  488. package/src/types/run/prepare-step.ts +59 -8
  489. package/src/types/run/store.ts +35 -4
  490. package/src/types/tool/index.ts +19 -1
  491. package/src/utils/await-with-abort.ts +26 -0
  492. package/src/utils/evidence-time.ts +9 -0
  493. package/src/utils/evidence-tokens.ts +32 -0
@@ -1,3 +1,4 @@
1
+ import type { AdvisoryTurnContext } from '../../../../advisory/executor.js'
1
2
  import { serializeState } from '../../../../compaction/serializer.js'
2
3
  import { NAMZU } from '../../../../constants/telemetry/index.js'
3
4
  import type { AdvisoryRequest, TriggerEvaluationState } from '../../../../types/advisory/index.js'
@@ -47,6 +48,7 @@ export async function runAdvisoryPhase(
47
48
  ctx: IterationContext,
48
49
  iterationNum: number,
49
50
  response: ChatCompletionResponse,
51
+ turn?: AdvisoryTurnContext,
50
52
  ): Promise<void> {
51
53
  const advisoryCtx = ctx.advisoryCtx
52
54
  if (!advisoryCtx) return
@@ -105,6 +107,7 @@ export async function runAdvisoryPhase(
105
107
  try {
106
108
  const executionResult = await advisoryCtx.executor.consult(advisor, request, {
107
109
  messages: ctx.runMgr.messages,
110
+ ...(turn ? { turn } : {}),
108
111
  workingStateSummary,
109
112
  toolCatalog: ctx.tools.toLLMTools(ctx.allowedTools),
110
113
  iteration: iterationNum,
@@ -26,6 +26,7 @@ import type {
26
26
  AgentRunConfig,
27
27
  BeforeStep,
28
28
  PrepareStepChain,
29
+ PrepareStepContext,
29
30
  RunEvent,
30
31
  StepResult,
31
32
  StopCondition,
@@ -229,6 +230,7 @@ export interface IterationContext {
229
230
 
230
231
  /** Host hook that shapes each step before the model call. */
231
232
  readonly prepareStep?: PrepareStepChain
233
+ readonly captureRunEvidence?: PrepareStepContext['captureRunEvidence']
232
234
  readonly beforeStep?: BeforeStep
233
235
  }
234
236
 
@@ -220,6 +220,9 @@ export async function* runToolReview(
220
220
  toolCall.authorization = {
221
221
  decision: gateResult.decision,
222
222
  ...(gateResult.reason ? { reason: gateResult.reason } : {}),
223
+ ...(gateResult.decision === 'review' && gateResult.matchedRule
224
+ ? { explicitReview: true as const }
225
+ : {}),
223
226
  }
224
227
  }
225
228
 
@@ -1,4 +1,5 @@
1
1
  import { isProviderRequestError } from '../../../provider/errors.js'
2
+ import type { Message } from '../../../types/message/index.js'
2
3
  import type {
3
4
  ChatCompletionParams,
4
5
  LLMProvider,
@@ -42,7 +43,8 @@ function isAcceptedChunk(chunk: StreamChunk): boolean {
42
43
  chunk.delta.citation ||
43
44
  chunk.finishReason !== undefined ||
44
45
  chunk.usage !== undefined ||
45
- chunk.replayState !== undefined,
46
+ chunk.replayState !== undefined ||
47
+ chunk.textParts !== undefined,
46
48
  )
47
49
  }
48
50
 
@@ -62,8 +64,10 @@ export async function* streamWithProviderRejectedImageRecovery(
62
64
  provider: LLMProvider,
63
65
  params: ChatCompletionParams,
64
66
  onAccepted: (identity: RequestImageIdentity) => Promise<void>,
67
+ onRequest?: (messages: readonly Message[]) => void,
65
68
  ): AsyncIterable<StreamChunk> {
66
69
  const candidate = findSingleRequestImage(params.messages)
70
+ onRequest?.(params.messages)
67
71
  if (candidate === null) {
68
72
  yield* provider.chatStream(params)
69
73
  return
@@ -90,6 +94,7 @@ export async function* streamWithProviderRejectedImageRecovery(
90
94
  }
91
95
  let accepted = false
92
96
  let retryReportedError = false
97
+ onRequest?.(retryParams.messages)
93
98
  for await (const chunk of provider.chatStream(retryParams)) {
94
99
  if (chunk.error !== undefined) retryReportedError = true
95
100
  if (!accepted && isAcceptedChunk(chunk)) {
@@ -4,6 +4,7 @@ import {
4
4
  assertNativeStructuredOutputSupported,
5
5
  } from '../../../provider/capabilities.js'
6
6
  import { isProviderRequestError } from '../../../provider/errors.js'
7
+ import { StreamTextAccumulator } from '../../../provider/stream-text.js'
7
8
  import { GENAI, NAMZU, chatSpanName, parentContext } from '../../../telemetry/attributes.js'
8
9
  import {
9
10
  recordModelDuration,
@@ -14,7 +15,12 @@ import { getTracer } from '../../../telemetry/runtime-accessors.js'
14
15
  import { mergeTokenUsage } from '../../../types/common/index.js'
15
16
  import { NamzuError } from '../../../types/errors/index.js'
16
17
  import type { ToolUseId } from '../../../types/ids/index.js'
17
- import type { Citation, ReasoningBlock } from '../../../types/message/index.js'
18
+ import type {
19
+ AssistantTextPart,
20
+ Citation,
21
+ Message,
22
+ ReasoningBlock,
23
+ } from '../../../types/message/index.js'
18
24
  import { ProviderError } from '../../../types/provider/errors.js'
19
25
  import type {
20
26
  ChatCompletionResponse,
@@ -54,6 +60,8 @@ function synthesizeMessageStopReason(
54
60
  export interface StreamingTurnResult {
55
61
  response: ChatCompletionResponse
56
62
  messageId: import('../../../types/ids/index.js').MessageId
63
+ /** Captured only for host review, at the provider-chain dispatch boundary. */
64
+ requestMessages?: readonly Message[]
57
65
  }
58
66
 
59
67
  /**
@@ -100,6 +108,7 @@ async function settleCancelledTurn(args: {
100
108
  messageId: import('../../../types/ids/index.js').MessageId
101
109
  usage: ChatCompletionResponse['usage']
102
110
  text: string
111
+ textParts?: readonly AssistantTextPart[]
103
112
  model: string
104
113
  startedAt: number
105
114
  span: Span
@@ -124,6 +133,7 @@ async function settleCancelledTurn(args: {
124
133
  stopReason: 'cancelled',
125
134
  usage: args.usage,
126
135
  content: args.text || undefined,
136
+ ...(args.textParts ? { textParts: args.textParts } : {}),
127
137
  })
128
138
  } catch {
129
139
  // Best effort. The cancellation is the news.
@@ -156,6 +166,7 @@ export async function* streamProviderTurn(
156
166
  imageRecovery?: {
157
167
  readonly onAccepted: (identity: RequestImageIdentity) => Promise<void>
158
168
  },
169
+ captureRequest = false,
159
170
  ): AsyncGenerator<RunEvent, StreamingTurnResult> {
160
171
  assertNativeStructuredOutputSupported(provider, params)
161
172
  assertHostedWebSearchSupported(provider, params)
@@ -181,7 +192,7 @@ export async function* streamProviderTurn(
181
192
 
182
193
  let id = ''
183
194
  const model = ''
184
- let textBuf = ''
195
+ const text = new StreamTextAccumulator()
185
196
  let finishReason: ChatCompletionResponse['finishReason'] = 'stop'
186
197
  let usage: ChatCompletionResponse['usage'] = {
187
198
  promptTokens: 0,
@@ -239,9 +250,21 @@ export async function* streamProviderTurn(
239
250
  ...params,
240
251
  stream: true,
241
252
  } satisfies import('../../../types/provider/index.js').ChatCompletionParams
253
+ let requestMessages: readonly Message[] | undefined
254
+ const observeRequest = captureRequest
255
+ ? (messages: readonly Message[]) => {
256
+ requestMessages = structuredClone(messages)
257
+ }
258
+ : undefined
259
+ if (!imageRecovery) observeRequest?.(streamParams.messages)
242
260
  const stream = (
243
261
  imageRecovery
244
- ? streamWithProviderRejectedImageRecovery(provider, streamParams, imageRecovery.onAccepted)
262
+ ? streamWithProviderRejectedImageRecovery(
263
+ provider,
264
+ streamParams,
265
+ imageRecovery.onAccepted,
266
+ observeRequest,
267
+ )
245
268
  : provider.chatStream(streamParams)
246
269
  ) as AsyncIterable<StreamChunk>
247
270
 
@@ -328,6 +351,7 @@ export async function* streamProviderTurn(
328
351
  }
329
352
  if (!id && chunk.id) id = chunk.id
330
353
  if (chunk.replayState !== undefined) replayState = chunk.replayState
354
+ text.push(chunk)
331
355
 
332
356
  // The first delta of the turn, of ANY kind — text, reasoning or a
333
357
  // tool call. namzu streams, so perceived latency is dominated by
@@ -401,13 +425,13 @@ export async function* streamProviderTurn(
401
425
  }
402
426
 
403
427
  if (chunk.delta.content) {
404
- textBuf += chunk.delta.content
405
428
  await emitEvent({
406
429
  type: 'text_delta',
407
430
  runId,
408
431
  iteration,
409
432
  messageId,
410
433
  text: chunk.delta.content,
434
+ ...(chunk.delta.textPart ? { textPart: chunk.delta.textPart } : {}),
411
435
  })
412
436
  yield* drainPending()
413
437
  }
@@ -518,7 +542,8 @@ export async function* streamProviderTurn(
518
542
  iteration,
519
543
  messageId,
520
544
  usage,
521
- text: textBuf,
545
+ text: text.text,
546
+ textParts: text.textParts,
522
547
  model: params.model,
523
548
  startedAt: callStartedAt,
524
549
  span: chatSpan,
@@ -661,7 +686,8 @@ export async function* streamProviderTurn(
661
686
  messageId,
662
687
  stopReason,
663
688
  usage,
664
- content: textBuf || undefined,
689
+ content: text.text || undefined,
690
+ ...(text.textParts ? { textParts: text.textParts } : {}),
665
691
  })
666
692
  yield* drainPending()
667
693
 
@@ -714,7 +740,8 @@ export async function* streamProviderTurn(
714
740
  model: model || params.model,
715
741
  message: {
716
742
  role: 'assistant',
717
- content: textBuf.length > 0 ? textBuf : null,
743
+ content: text.text || null,
744
+ ...(text.textParts ? { textParts: text.textParts } : {}),
718
745
  toolCalls: toolCalls.length > 0 ? toolCalls : undefined,
719
746
  ...(reasoningBlocks.length > 0 ? { reasoning: reasoningBlocks } : {}),
720
747
  ...(replayState !== undefined ? { replayState } : {}),
@@ -746,5 +773,5 @@ export async function* streamProviderTurn(
746
773
  chatSpan.setStatus({ code: SpanStatusCode.OK })
747
774
  chatSpan.end()
748
775
 
749
- return { response, messageId }
776
+ return { response, messageId, ...(requestMessages ? { requestMessages } : {}) }
750
777
  }
@@ -1,6 +1,7 @@
1
1
  import { isDeepStrictEqual } from 'node:util'
2
2
 
3
3
  import type { RunPersistence } from '../../manager/run/persistence.js'
4
+ import { ToolExecutionCollector } from '../../store/run/tool-executions.js'
4
5
  import type {
5
6
  CheckpointId,
6
7
  HITLResumeDecision,
@@ -9,6 +10,7 @@ import type {
9
10
  } from '../../types/hitl/index.js'
10
11
  import type { AssistantMessage, Message, ToolCall } from '../../types/message/index.js'
11
12
  import type { ChatCompletionResponse } from '../../types/provider/index.js'
13
+ import type { ToolExecutionSnapshot } from '../../types/run/store.js'
12
14
  import type { Logger } from '../../utils/logger.js'
13
15
  import type { PriorToolResults, ToolCallDenials, ToolExecutor } from './executor.js'
14
16
  import { PendingAnswers } from './question-park.js'
@@ -192,24 +194,11 @@ function planQuestionResume(
192
194
  }
193
195
 
194
196
  /**
195
- * Decide whether a checkpoint left behind a batch that was PART-WAY
196
- * through executing when the process died.
197
- *
198
- * The ordinary repair for an unanswered assistant turn — strip it and let
199
- * the model re-decide — is right when nothing ran: the calls were still
200
- * awaiting a decision, so re-deciding costs only a round trip. It is
201
- * exactly wrong when some of them already ran, because re-deciding means
202
- * re-executing, and a tool that charged a card does not become idempotent
203
- * on the second attempt.
204
- *
205
- * The discriminator is the transcript: a tool-review park records the
206
- * checkpoint BEFORE any execution, so it has no completed calls and takes
207
- * the cheap path unchanged. One or more completions means execution had
208
- * begun, which is a resume, not a fresh decision.
209
- *
210
- * The calls that did NOT complete are executed here for the first time,
211
- * through the ordinary executor — so every guard, permission check and
212
- * probe still applies to them.
197
+ * Preserve a restored batch's recorded and explicitly unknown outcomes.
198
+ * `completed` comes from recoverCompletedCalls: missing entries are eligible
199
+ * only after a complete scan establishes that they have no recorded start.
200
+ * A started tool with no trustworthy completion is carried as an unknown
201
+ * outcome, not executed again. An untouched batch uses ordinary history repair.
213
202
  */
214
203
  export function planCrashResume(
215
204
  checkpoint: IterationCheckpoint,
@@ -219,13 +208,19 @@ export function planCrashResume(
219
208
  const assistant = lastUnansweredBatch(checkpoint.messages)?.assistant
220
209
  const calls = assistant?.toolCalls
221
210
  if (!assistant || !calls || calls.length === 0) return null
211
+ // Only the current tail can be resumed in place. An older incomplete
212
+ // batch belongs to history repair; re-appending it would move its action
213
+ // past a newer operator message and silently reorder the conversation.
214
+ const ownerIndex = checkpoint.messages.lastIndexOf(assistant)
215
+ if (checkpoint.messages.slice(ownerIndex + 1).some((message) => message.role !== 'tool'))
216
+ return null
222
217
 
223
218
  const done = calls.filter((tc) => completed.has(tc.id))
224
219
  if (done.length === 0) return null
225
220
 
226
221
  log.warn('Checkpoint holds a tool batch that was part-way through executing', {
227
222
  'namzu.checkpoint.id': checkpoint.id,
228
- 'namzu.runtime.completed': done.length,
223
+ 'namzu.runtime.recovered': done.length,
229
224
  'namzu.runtime.total': calls.length,
230
225
  'namzu.runtime.remaining': calls
231
226
  .filter((tc) => !completed.has(tc.id))
@@ -336,37 +331,67 @@ function modifiedCallIds(decision: HITLResumeDecision): ReadonlySet<string> {
336
331
  }
337
332
 
338
333
  /**
339
- * Results the run already produced for calls in `toolCalls`.
340
- *
341
- * Read from the transcript, which records a `tool_completed` per tool as
342
- * it finishes — durable long before the batch settles. Scoped to the calls
343
- * being resumed so an id from an earlier turn can never answer this one.
344
- *
345
- * A failure to read is not fatal: the worst case is the behaviour that
346
- * existed before this recovery, and refusing to resume because a log could
347
- * not be read would be a strictly worse trade.
334
+ * Recover completed results and close interrupted calls with unknown outcomes.
335
+ * Only a complete execution scan can authorize the absence of a start record.
336
+ * Unreadable, partial or contradictory evidence never becomes permission to
337
+ * replay; an explicitly answered durable question may re-enter its own tool.
348
338
  */
349
339
  export async function recoverCompletedCalls(
350
340
  runMgr: RunPersistence,
351
341
  toolCalls: readonly ToolCall[],
352
342
  log: Logger,
343
+ options: { answers?: PendingAnswers; signal?: AbortSignal } = {},
353
344
  ): Promise<Map<string, { result: string; isError: boolean }>> {
354
345
  const recovered = new Map<string, { result: string; isError: boolean }>()
346
+ let snapshot: ToolExecutionSnapshot | undefined
355
347
  try {
356
- const completed = await runMgr.getRunStore().readCompletedTools()
357
- for (const call of toolCalls) {
358
- const record = completed.get(call.id)
359
- if (record) recovered.set(call.id, { result: record.result, isError: record.isError })
348
+ const store = runMgr.getRunStore()
349
+ if (store.readToolExecutions) {
350
+ snapshot = await store.readToolExecutions(
351
+ toolCalls.map((call) => call.id),
352
+ options.signal,
353
+ )
354
+ } else {
355
+ const collector = new ToolExecutionCollector(
356
+ runMgr.id,
357
+ toolCalls.map((call) => call.id),
358
+ )
359
+ for (const event of await store.readEvents({ integrity: 'strict' })) {
360
+ options.signal?.throwIfAborted()
361
+ collector.accept(event)
362
+ }
363
+ snapshot = collector.finish()
360
364
  }
361
365
  } catch (error) {
366
+ if (options.signal?.aborted) throw error
362
367
  log.warn('Could not read the transcript to recover completed tool calls', {
363
368
  'exception.message': error instanceof Error ? error.message : String(error),
364
369
  })
365
- return new Map()
370
+ }
371
+ for (const call of toolCalls) {
372
+ const record = snapshot?.records.get(call.id)
373
+ if (
374
+ snapshot?.complete &&
375
+ (!record || (record.toolName === call.function.name && record.toolUseId === call.id))
376
+ ) {
377
+ if (!record) continue // Complete evidence proves this call has no recorded start.
378
+ if (record.status === 'completed') {
379
+ recovered.set(call.id, { result: record.result, isError: record.isError })
380
+ continue
381
+ }
382
+ }
383
+ // The validated checkpoint and explicit answer own this re-entry even
384
+ // when the prior event store is unavailable. No sibling inherits it.
385
+ if ([...(options.answers?.entries() ?? [])].some(([id]) => isPauseForCall(id, call.id)))
386
+ continue
387
+ recovered.set(call.id, {
388
+ result: `Tool execution was interrupted and no trustworthy completion is available for \`${call.function.name}\`. Its outcome is unknown. This resume did not execute it again. Verify external state before deciding whether another call is needed; do not replay a state-changing action to recover its output.`,
389
+ isError: true,
390
+ })
366
391
  }
367
392
 
368
393
  if (recovered.size > 0) {
369
- log.info('Recovered tool results from the transcript instead of re-executing', {
394
+ log.info('Restored recorded or unknown tool outcomes without re-executing', {
370
395
  'namzu.runtime.recovered': recovered.size,
371
396
  'namzu.runtime.of_calls': toolCalls.length,
372
397
  })
@@ -452,13 +477,13 @@ function synthesizeResponse(assistant: AssistantMessage): ChatCompletionResponse
452
477
  }
453
478
 
454
479
  /**
455
- * Tool calls in the history that no `tool_result` answers.
456
- *
457
- * The set worth asking the transcript about: an already-answered call's
458
- * result is in the history and needs no recovery.
480
+ * All calls owned by the last incomplete batch. Applying a resume plan
481
+ * reconstructs that entire batch, including any partial results already in
482
+ * the checkpoint. Those answered siblings must also be recovered, or their
483
+ * missing prior-result entries would let the executor repeat them.
459
484
  */
460
- export function unansweredToolCalls(messages: readonly Message[]): ToolCall[] {
461
- return lastUnansweredBatch(messages)?.unanswered ?? []
485
+ export function interruptedToolCalls(messages: readonly Message[]): ToolCall[] {
486
+ return lastUnansweredBatch(messages)?.assistant.toolCalls ?? []
462
487
  }
463
488
 
464
489
  function lastUnansweredBatch(
@@ -1,9 +1,9 @@
1
1
  /**
2
- * How a run resolves the calls no rule decided.
2
+ * How a run resolves calls routed to review.
3
3
  *
4
4
  * An authorization rule says what a tool may do. A review policy says what
5
- * happens to everything the rules did not cover: the batch the gate routed
6
- * to REVIEW. The two axes are separate on purpose — a rule is a durable
5
+ * happens to calls the rules did not cover or explicitly routed to REVIEW.
6
+ * The two axes are separate on purpose — a rule is a durable
7
7
  * statement an operator reviewed, and a mode is a property of ONE run, the
8
8
  * difference between "we never force-push" and "this run is unattended".
9
9
  *
@@ -132,12 +132,14 @@ export function isReviewExempt(
132
132
 
133
133
  export type ReviewExemption = (name: string, input: unknown) => boolean
134
134
 
135
- /** A batch needs review when any call mutates state: flagged destructive, or not exempt. */
135
+ /** Review explicit requests and calls that are destructive or not exempt. */
136
136
  export function batchNeedsReview(
137
137
  toolCalls: readonly ToolCallSummary[],
138
138
  exempt: ReviewExemption,
139
139
  ): boolean {
140
- return toolCalls.some((tc) => tc.isDestructive || !exempt(tc.name, tc.input))
140
+ return toolCalls.some(
141
+ (tc) => tc.authorization?.explicitReview || tc.isDestructive || !exempt(tc.name, tc.input),
142
+ )
141
143
  }
142
144
 
143
145
  /** The batch a person is asked about. */
@@ -194,14 +196,16 @@ export function createReviewHandler(options: ReviewPolicyOptions = {}): ResumeHa
194
196
  if (
195
197
  mode === 'accept-edits' &&
196
198
  request.toolCalls.every(
197
- (tc) => !tc.isDestructive && (ACCEPT_EDITS_TOOLS.has(tc.name) || exempt(tc.name, tc.input)),
199
+ (tc) =>
200
+ !tc.authorization?.explicitReview &&
201
+ !tc.isDestructive &&
202
+ (ACCEPT_EDITS_TOOLS.has(tc.name) || exempt(tc.name, tc.input)),
198
203
  )
199
204
  ) {
200
205
  return { action: 'approve_tools' }
201
206
  }
202
- // Reads were approved above. Anything that reached here would change
203
- // something, and plan mode's answer is the same every time: not now,
204
- // tell the user what you would do.
207
+ // Ordinary reads were approved above. Remaining calls either change
208
+ // state or carry explicit review; plan mode does not grant that authority.
205
209
  if (mode === 'plan') return { action: 'reject_tools', feedback: PLAN_MODE_REFUSAL }
206
210
  if (mode === 'strict') return { action: 'reject_tools', feedback: STRICT_MODE_REFUSAL }
207
211
  if (mode === 'auto' || !prompt || remembered.all) {
@@ -138,6 +138,9 @@ export async function acquireSandbox(options: {
138
138
  runSignal.addEventListener('abort', onAbort, { once: true })
139
139
  })
140
140
  const timedOut = new Promise<{ readonly kind: 'timed_out'; readonly error: Error }>((resolve) => {
141
+ // An unlimited run still observes cancellation; Infinity must never reach
142
+ // setTimeout, where Node would coerce it to a one-millisecond deadline.
143
+ if (options.timeoutMs === Number.POSITIVE_INFINITY) return
141
144
  timer = setTimeout(() => {
142
145
  timeoutError = acquisitionTimeoutError(options.timeoutMs)
143
146
  operationController.abort(timeoutError)
@@ -1,6 +1,7 @@
1
1
  import { createHash } from 'node:crypto'
2
2
  import { mkdirSync, writeFileSync } from 'node:fs'
3
3
  import { join } from 'node:path'
4
+ import { digest, spillManifest } from '../../store/evidence/format.js'
4
5
 
5
6
  /**
6
7
  * Model-visible size cap for a single tool result.
@@ -104,6 +105,8 @@ export interface ToolOutputBudgetResult {
104
105
  readonly truncated: boolean
105
106
  /** Where the full output was written, when it was. */
106
107
  readonly spillPath?: string
108
+ /** Digest of the bounded chunk manifest, recorded at retention time. */
109
+ readonly spillIntegrity?: string
107
110
  }
108
111
 
109
112
  /**
@@ -155,6 +158,10 @@ export interface ApplyToolOutputBudgetOptions {
155
158
  readonly toolUseId: string
156
159
  readonly output: string
157
160
  readonly maxChars: number
161
+ /** Optional condensed presentation; used only after authenticated retention of output. */
162
+ readonly preview?: string
163
+ /** Smaller preview only after the full output and its integrity manifest are saved. */
164
+ readonly retainedPreviewChars?: number
158
165
  /** An omission notice that shares the text budget, never extends it. */
159
166
  readonly notice?: string
160
167
  /**
@@ -168,12 +175,10 @@ export interface ApplyToolOutputBudgetOptions {
168
175
  /**
169
176
  * Bound a tool result to the model-visible budget.
170
177
  *
171
- * Spilling beats truncating on every axis that matters: nothing is lost,
172
- * tokens are paid only if the agent decides the rest is worth re-reading,
173
- * and retrieval uses `read`/`grep` — tools it already has — rather than a
174
- * new affordance. A hosted agent runtime does the same thing above 100k
175
- * characters. Middle-elision is the fallback for a run with no directory
176
- * to write to.
178
+ * Retention keeps the original while bounding its model-visible preview.
179
+ * The host owns the recovery route and its permissions; a spill path is not
180
+ * proof that workspace read/grep tools can access it. Middle-elision is the
181
+ * fallback for a run with no directory to write to.
177
182
  *
178
183
  * The preview keeps head AND tail because the two ends carry different
179
184
  * information: the head has the schema/opening of a document, the tail has
@@ -194,26 +199,65 @@ export function applyToolOutputBudget(opts: ApplyToolOutputBudgetOptions): ToolO
194
199
  : Math.max(0, limit - notice.length - (notice && output ? 2 : 0))
195
200
  const withNotice = (text: string) => [text, notice].filter(Boolean).join('\n\n')
196
201
 
197
- if (textBudget === undefined || originalLength <= textBudget) {
202
+ const preview =
203
+ opts.preview !== undefined && opts.preview.length < originalLength ? opts.preview : undefined
204
+ if (preview === undefined && (textBudget === undefined || originalLength <= textBudget)) {
198
205
  return { output: withNotice(output), originalLength, truncated: false }
199
206
  }
200
207
 
201
- const spillPath = opts.spillDir
208
+ const retained = opts.spillDir
202
209
  ? spill(opts.spillDir, opts.toolUseId, output, opts.onError)
203
210
  : undefined
204
211
 
212
+ const spillPath = retained?.path
205
213
  const recovery = spillPath
206
214
  ? [
207
215
  `${SPILL_MARKER} ${spillPath}`,
208
- 'Read a specific window with `read` (offset/limit) or search it with `grep`. Do NOT read it whole — that is what exceeded the budget.',
216
+ 'This path identifies retained output, not the original input. Use the host-authorized retained-output recovery tools when available; the path does not grant filesystem access. A fresh observation of the original input cannot recover its earlier contents. If recovery is unavailable, report the missing detail; do not replay the original action.',
209
217
  ].join('\n')
210
218
  : 'The full output was not retained. Use a saved artifact or a read-only observation; do not repeat a state-changing action to recover its output.'
219
+ const requestedPreview = opts.retainedPreviewChars
220
+ // Separate the spill threshold from the cost of carrying its preview on every
221
+ // later request. Failed retention must not silently discard additional text.
222
+ // A small configured preview still needs room for the durable recovery path.
223
+ const previewLimit =
224
+ retained?.integrity &&
225
+ requestedPreview !== undefined &&
226
+ Number.isFinite(requestedPreview) &&
227
+ requestedPreview > 0
228
+ ? Math.min(limit as number, Math.floor(requestedPreview))
229
+ : (limit as number)
230
+ const previewTextBudget = Math.max(0, previewLimit - notice.length - (notice && output ? 2 : 0))
231
+ const minimumRecoveryChars = `\n[... omitted ...]\n${SPILL_MARKER} ${spillPath}\n`.length
232
+ const effectiveTextBudget =
233
+ previewTextBudget > minimumRecoveryChars ? previewTextBudget : textBudget
234
+ // Condensation is a display choice, not permission to discard evidence. The
235
+ // original can be under the ordinary cap while still losing distinct rows.
236
+ // Failed retention falls back to the original, bounded by the usual cap.
237
+ if (preview !== undefined && retained?.integrity) {
238
+ const condensed = [preview, recovery].filter(Boolean).join('\n\n')
239
+ if (effectiveTextBudget === undefined || condensed.length <= effectiveTextBudget) {
240
+ return {
241
+ output: withNotice(condensed),
242
+ originalLength,
243
+ truncated: true,
244
+ spillPath,
245
+ spillIntegrity: retained.integrity,
246
+ }
247
+ }
248
+ }
249
+ if (textBudget === undefined || originalLength <= textBudget) {
250
+ return { output: withNotice(output), originalLength, truncated: false }
251
+ }
211
252
 
212
253
  return {
213
- output: withNotice(boundedPreview(output, textBudget, opts.toolName, recovery)),
254
+ output: withNotice(
255
+ boundedPreview(output, effectiveTextBudget ?? textBudget, opts.toolName, recovery),
256
+ ),
214
257
  originalLength,
215
258
  truncated: true,
216
259
  ...(spillPath ? { spillPath } : {}),
260
+ ...(retained?.integrity ? { spillIntegrity: retained.integrity } : {}),
217
261
  }
218
262
  }
219
263
 
@@ -222,7 +266,7 @@ function spill(
222
266
  toolUseId: string,
223
267
  content: string,
224
268
  onError?: (message: string) => void,
225
- ): string | undefined {
269
+ ): { path: string; integrity?: string } | undefined {
226
270
  try {
227
271
  // `0o700` on the directory and `0o600` on the file: a spilled output is
228
272
  // routinely the largest and most sensitive thing a run produces — whole
@@ -250,8 +294,16 @@ function spill(
250
294
  // so randomising the filename would add nothing this does not already
251
295
  // have. Do not "improve" it back to a random name and a plain `w` —
252
296
  // that trades a guarantee for a guess.
253
- writeFileSync(path, content, { encoding: 'utf-8', flag: 'wx', mode: 0o600 })
254
- return path
297
+ const bytes = Buffer.from(content, 'utf8')
298
+ writeFileSync(path, bytes, { flag: 'wx', mode: 0o600 })
299
+ try {
300
+ const manifest = spillManifest(bytes)
301
+ writeFileSync(`${path}.manifest.json`, manifest, { flag: 'wx', mode: 0o600 })
302
+ return { path, integrity: digest(manifest) }
303
+ } catch {
304
+ onError?.('The full output was retained, but its integrity manifest could not be written.')
305
+ return { path }
306
+ }
255
307
  } catch (err) {
256
308
  // A spill failure must never fail the tool call — the model still
257
309
  // gets the preview, just without a path to recover the rest.
@@ -47,8 +47,10 @@ export interface ToolingBootstrapConfig {
47
47
  maxToolCalls?: number
48
48
  readToolCallBudgetEvents?: () => Promise<readonly RunEvent[]>
49
49
  maxToolOutputChars?: number
50
+ retainedToolPreviewChars?: number
50
51
  maxToolContentBytes?: number
51
- toolOutputDir?: string
52
+ captureRunEvidence?: import('../../types/tool/index.js').ToolContext['captureRunEvidence']
53
+ toolOutputDir?: string | (() => string | undefined)
52
54
  repairToolCall?: RepairToolCall
53
55
  /** Operator authorization shared with the direct-call review path. */
54
56
  authorizationGate?: AuthorizationGate
@@ -79,6 +81,7 @@ export class ToolingBootstrap {
79
81
  abortSignal: config.abortSignal,
80
82
  allowedTools: config.allowedTools,
81
83
  invocationState: config.invocationState,
84
+ captureRunEvidence: config.captureRunEvidence,
82
85
  pluginManager: config.pluginManager,
83
86
  ...(config.backgroundJobs ? { backgroundJobs: config.backgroundJobs } : {}),
84
87
  ...(config.backgroundJobOwner ? { backgroundJobOwner: config.backgroundJobOwner } : {}),
@@ -98,6 +101,9 @@ export class ToolingBootstrap {
98
101
  ...(config.maxToolOutputChars !== undefined
99
102
  ? { maxToolOutputChars: config.maxToolOutputChars }
100
103
  : {}),
104
+ ...(config.retainedToolPreviewChars !== undefined
105
+ ? { retainedToolPreviewChars: config.retainedToolPreviewChars }
106
+ : {}),
101
107
  ...(config.maxToolContentBytes !== undefined
102
108
  ? { maxToolContentBytes: config.maxToolContentBytes }
103
109
  : {}),