@namzu/sdk 38.2.1 → 39.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (493) hide show
  1. package/CHANGELOG.md +700 -0
  2. package/dist/advisory/executor.d.ts +10 -1
  3. package/dist/advisory/executor.d.ts.map +1 -1
  4. package/dist/advisory/executor.js +5 -26
  5. package/dist/advisory/executor.js.map +1 -1
  6. package/dist/advisory/history.d.ts +9 -0
  7. package/dist/advisory/history.d.ts.map +1 -0
  8. package/dist/advisory/history.js +120 -0
  9. package/dist/advisory/history.js.map +1 -0
  10. package/dist/advisory/index.d.ts +1 -1
  11. package/dist/advisory/index.d.ts.map +1 -1
  12. package/dist/advisory/index.js.map +1 -1
  13. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  14. package/dist/agents/ReactiveAgent.js +3 -0
  15. package/dist/agents/ReactiveAgent.js.map +1 -1
  16. package/dist/agents/runAgent.d.ts +10 -0
  17. package/dist/agents/runAgent.d.ts.map +1 -1
  18. package/dist/agents/runAgent.js +3 -0
  19. package/dist/agents/runAgent.js.map +1 -1
  20. package/dist/compaction/manual.d.ts +6 -0
  21. package/dist/compaction/manual.d.ts.map +1 -1
  22. package/dist/compaction/manual.js +21 -2
  23. package/dist/compaction/manual.js.map +1 -1
  24. package/dist/compaction/summary.d.ts.map +1 -1
  25. package/dist/compaction/summary.js +4 -1
  26. package/dist/compaction/summary.js.map +1 -1
  27. package/dist/config/runtime.js +2 -2
  28. package/dist/config/runtime.js.map +1 -1
  29. package/dist/contracts/schemas.js +1 -1
  30. package/dist/contracts/schemas.js.map +1 -1
  31. package/dist/eval/harness-protection.d.ts +18 -0
  32. package/dist/eval/harness-protection.d.ts.map +1 -0
  33. package/dist/eval/harness-protection.js +58 -0
  34. package/dist/eval/harness-protection.js.map +1 -0
  35. package/dist/eval/harness-verification.d.ts +6 -1
  36. package/dist/eval/harness-verification.d.ts.map +1 -1
  37. package/dist/eval/harness-verification.js +19 -2
  38. package/dist/eval/harness-verification.js.map +1 -1
  39. package/dist/eval/index.d.ts +1 -0
  40. package/dist/eval/index.d.ts.map +1 -1
  41. package/dist/eval/index.js.map +1 -1
  42. package/dist/manager/resident/activity.d.ts +50 -0
  43. package/dist/manager/resident/activity.d.ts.map +1 -0
  44. package/dist/manager/resident/activity.js +125 -0
  45. package/dist/manager/resident/activity.js.map +1 -0
  46. package/dist/manager/resident/agenda.d.ts +13 -1
  47. package/dist/manager/resident/agenda.d.ts.map +1 -1
  48. package/dist/manager/resident/agenda.js +40 -5
  49. package/dist/manager/resident/agenda.js.map +1 -1
  50. package/dist/manager/resident/consumption.d.ts +104 -0
  51. package/dist/manager/resident/consumption.d.ts.map +1 -0
  52. package/dist/manager/resident/consumption.js +233 -0
  53. package/dist/manager/resident/consumption.js.map +1 -0
  54. package/dist/manager/resident/evidence-recall.d.ts +24 -0
  55. package/dist/manager/resident/evidence-recall.d.ts.map +1 -0
  56. package/dist/manager/resident/evidence-recall.js +295 -0
  57. package/dist/manager/resident/evidence-recall.js.map +1 -0
  58. package/dist/manager/resident/history-disk.d.ts +10 -0
  59. package/dist/manager/resident/history-disk.d.ts.map +1 -0
  60. package/dist/manager/resident/history-disk.js +50 -0
  61. package/dist/manager/resident/history-disk.js.map +1 -0
  62. package/dist/manager/resident/history.d.ts +79 -0
  63. package/dist/manager/resident/history.d.ts.map +1 -0
  64. package/dist/manager/resident/history.js +203 -0
  65. package/dist/manager/resident/history.js.map +1 -0
  66. package/dist/manager/resident/initiative.d.ts.map +1 -1
  67. package/dist/manager/resident/initiative.js +11 -3
  68. package/dist/manager/resident/initiative.js.map +1 -1
  69. package/dist/manager/resident/learning-cycle.d.ts +131 -0
  70. package/dist/manager/resident/learning-cycle.d.ts.map +1 -0
  71. package/dist/manager/resident/learning-cycle.js +306 -0
  72. package/dist/manager/resident/learning-cycle.js.map +1 -0
  73. package/dist/manager/resident/learning-observation.d.ts +80 -0
  74. package/dist/manager/resident/learning-observation.d.ts.map +1 -0
  75. package/dist/manager/resident/learning-observation.js +22 -0
  76. package/dist/manager/resident/learning-observation.js.map +1 -0
  77. package/dist/manager/resident/learning-store.d.ts +106 -0
  78. package/dist/manager/resident/learning-store.d.ts.map +1 -0
  79. package/dist/manager/resident/learning-store.js +598 -0
  80. package/dist/manager/resident/learning-store.js.map +1 -0
  81. package/dist/manager/resident/learning.d.ts +246 -3
  82. package/dist/manager/resident/learning.d.ts.map +1 -1
  83. package/dist/manager/resident/learning.js +96 -6
  84. package/dist/manager/resident/learning.js.map +1 -1
  85. package/dist/manager/resident/outbox.d.ts +4 -4
  86. package/dist/manager/resident/store.d.ts +37 -4
  87. package/dist/manager/resident/store.d.ts.map +1 -1
  88. package/dist/manager/resident/store.js +27 -3
  89. package/dist/manager/resident/store.js.map +1 -1
  90. package/dist/manager/resident/tool-evidence.d.ts +71 -0
  91. package/dist/manager/resident/tool-evidence.d.ts.map +1 -0
  92. package/dist/manager/resident/tool-evidence.js +285 -0
  93. package/dist/manager/resident/tool-evidence.js.map +1 -0
  94. package/dist/manager/run/persistence.d.ts.map +1 -1
  95. package/dist/manager/run/persistence.js +6 -0
  96. package/dist/manager/run/persistence.js.map +1 -1
  97. package/dist/plugin/loader.d.ts.map +1 -1
  98. package/dist/plugin/loader.js +5 -3
  99. package/dist/plugin/loader.js.map +1 -1
  100. package/dist/prompt/coding-agent-doctrine.d.ts +1 -1
  101. package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
  102. package/dist/prompt/coding-agent-doctrine.js +2 -0
  103. package/dist/prompt/coding-agent-doctrine.js.map +1 -1
  104. package/dist/prompt/index.d.ts +2 -0
  105. package/dist/prompt/index.d.ts.map +1 -1
  106. package/dist/prompt/index.js +1 -0
  107. package/dist/prompt/index.js.map +1 -1
  108. package/dist/prompt/resident-learning.d.ts +19 -0
  109. package/dist/prompt/resident-learning.d.ts.map +1 -0
  110. package/dist/prompt/resident-learning.js +125 -0
  111. package/dist/prompt/resident-learning.js.map +1 -0
  112. package/dist/prompt/resident-step.d.ts +8 -1
  113. package/dist/prompt/resident-step.d.ts.map +1 -1
  114. package/dist/prompt/resident-step.js +65 -4
  115. package/dist/prompt/resident-step.js.map +1 -1
  116. package/dist/provider/collect-chat-completion.d.ts +2 -1
  117. package/dist/provider/collect-chat-completion.d.ts.map +1 -1
  118. package/dist/provider/collect-chat-completion.js +7 -6
  119. package/dist/provider/collect-chat-completion.js.map +1 -1
  120. package/dist/provider/fallback.d.ts.map +1 -1
  121. package/dist/provider/fallback.js +2 -1
  122. package/dist/provider/fallback.js.map +1 -1
  123. package/dist/provider/stream-text.d.ts +16 -0
  124. package/dist/provider/stream-text.d.ts.map +1 -0
  125. package/dist/provider/stream-text.js +51 -0
  126. package/dist/provider/stream-text.js.map +1 -0
  127. package/dist/public-runtime.d.ts +13 -3
  128. package/dist/public-runtime.d.ts.map +1 -1
  129. package/dist/public-runtime.js +12 -3
  130. package/dist/public-runtime.js.map +1 -1
  131. package/dist/public-tools.d.ts +2 -0
  132. package/dist/public-tools.d.ts.map +1 -1
  133. package/dist/public-tools.js +2 -0
  134. package/dist/public-tools.js.map +1 -1
  135. package/dist/public-types.d.ts +15 -3
  136. package/dist/public-types.d.ts.map +1 -1
  137. package/dist/run/LimitChecker.js +3 -3
  138. package/dist/run/LimitChecker.js.map +1 -1
  139. package/dist/run/evidence-query.d.ts +41 -0
  140. package/dist/run/evidence-query.d.ts.map +1 -0
  141. package/dist/run/evidence-query.js +270 -0
  142. package/dist/run/evidence-query.js.map +1 -0
  143. package/dist/run/evidence-recall.d.ts +99 -0
  144. package/dist/run/evidence-recall.d.ts.map +1 -0
  145. package/dist/run/evidence-recall.js +633 -0
  146. package/dist/run/evidence-recall.js.map +1 -0
  147. package/dist/run/index.d.ts +2 -0
  148. package/dist/run/index.d.ts.map +1 -1
  149. package/dist/run/index.js +1 -0
  150. package/dist/run/index.js.map +1 -1
  151. package/dist/run/json-claim-verifier.d.ts +83 -0
  152. package/dist/run/json-claim-verifier.d.ts.map +1 -0
  153. package/dist/run/json-claim-verifier.js +200 -0
  154. package/dist/run/json-claim-verifier.js.map +1 -0
  155. package/dist/run/preparation-context-error.d.ts +10 -0
  156. package/dist/run/preparation-context-error.d.ts.map +1 -0
  157. package/dist/run/preparation-context-error.js +15 -0
  158. package/dist/run/preparation-context-error.js.map +1 -0
  159. package/dist/run-query/index.d.ts +3 -1
  160. package/dist/run-query/index.d.ts.map +1 -1
  161. package/dist/runtime/query/callback-inference.d.ts +8 -0
  162. package/dist/runtime/query/callback-inference.d.ts.map +1 -0
  163. package/dist/runtime/query/callback-inference.js +89 -0
  164. package/dist/runtime/query/callback-inference.js.map +1 -0
  165. package/dist/runtime/query/checkpoint.d.ts +4 -0
  166. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  167. package/dist/runtime/query/checkpoint.js +13 -0
  168. package/dist/runtime/query/checkpoint.js.map +1 -1
  169. package/dist/runtime/query/events.d.ts +1 -0
  170. package/dist/runtime/query/events.d.ts.map +1 -1
  171. package/dist/runtime/query/events.js +18 -0
  172. package/dist/runtime/query/events.js.map +1 -1
  173. package/dist/runtime/query/executor.d.ts +7 -1
  174. package/dist/runtime/query/executor.d.ts.map +1 -1
  175. package/dist/runtime/query/executor.js +38 -7
  176. package/dist/runtime/query/executor.js.map +1 -1
  177. package/dist/runtime/query/file-evidence-context.d.ts +5 -0
  178. package/dist/runtime/query/file-evidence-context.d.ts.map +1 -0
  179. package/dist/runtime/query/file-evidence-context.js +64 -0
  180. package/dist/runtime/query/file-evidence-context.js.map +1 -0
  181. package/dist/runtime/query/guard.d.ts.map +1 -1
  182. package/dist/runtime/query/guard.js +4 -0
  183. package/dist/runtime/query/guard.js.map +1 -1
  184. package/dist/runtime/query/index.d.ts +10 -1
  185. package/dist/runtime/query/index.d.ts.map +1 -1
  186. package/dist/runtime/query/index.js +39 -24
  187. package/dist/runtime/query/index.js.map +1 -1
  188. package/dist/runtime/query/iteration/index.d.ts +9 -31
  189. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  190. package/dist/runtime/query/iteration/index.js +231 -93
  191. package/dist/runtime/query/iteration/index.js.map +1 -1
  192. package/dist/runtime/query/iteration/phases/advisory.d.ts +2 -1
  193. package/dist/runtime/query/iteration/phases/advisory.d.ts.map +1 -1
  194. package/dist/runtime/query/iteration/phases/advisory.js +2 -1
  195. package/dist/runtime/query/iteration/phases/advisory.js.map +1 -1
  196. package/dist/runtime/query/iteration/phases/context.d.ts +2 -1
  197. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  198. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  199. package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
  200. package/dist/runtime/query/iteration/phases/tool-review.js +3 -0
  201. package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
  202. package/dist/runtime/query/iteration/provider-rejected-image.d.ts +2 -1
  203. package/dist/runtime/query/iteration/provider-rejected-image.d.ts.map +1 -1
  204. package/dist/runtime/query/iteration/provider-rejected-image.js +5 -2
  205. package/dist/runtime/query/iteration/provider-rejected-image.js.map +1 -1
  206. package/dist/runtime/query/iteration/stream-turn.d.ts +4 -1
  207. package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
  208. package/dist/runtime/query/iteration/stream-turn.js +22 -8
  209. package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
  210. package/dist/runtime/query/resume-pending.d.ts +18 -33
  211. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  212. package/dist/runtime/query/resume-pending.js +59 -42
  213. package/dist/runtime/query/resume-pending.js.map +1 -1
  214. package/dist/runtime/query/review-policy.d.ts +4 -4
  215. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  216. package/dist/runtime/query/review-policy.js +10 -9
  217. package/dist/runtime/query/review-policy.js.map +1 -1
  218. package/dist/runtime/query/sandbox-lifecycle.d.ts.map +1 -1
  219. package/dist/runtime/query/sandbox-lifecycle.js +4 -0
  220. package/dist/runtime/query/sandbox-lifecycle.js.map +1 -1
  221. package/dist/runtime/query/tool-output-budget.d.ts +10 -6
  222. package/dist/runtime/query/tool-output-budget.d.ts.map +1 -1
  223. package/dist/runtime/query/tool-output-budget.js +54 -12
  224. package/dist/runtime/query/tool-output-budget.js.map +1 -1
  225. package/dist/runtime/query/tooling.d.ts +3 -1
  226. package/dist/runtime/query/tooling.d.ts.map +1 -1
  227. package/dist/runtime/query/tooling.js +4 -0
  228. package/dist/runtime/query/tooling.js.map +1 -1
  229. package/dist/scheduler/completion-inbox.d.ts +3 -0
  230. package/dist/scheduler/completion-inbox.d.ts.map +1 -1
  231. package/dist/scheduler/completion-inbox.js +28 -0
  232. package/dist/scheduler/completion-inbox.js.map +1 -1
  233. package/dist/store/evidence/compaction-archive.d.ts +109 -0
  234. package/dist/store/evidence/compaction-archive.d.ts.map +1 -0
  235. package/dist/store/evidence/compaction-archive.js +125 -0
  236. package/dist/store/evidence/compaction-archive.js.map +1 -0
  237. package/dist/store/evidence/compaction-provenance.d.ts +7 -0
  238. package/dist/store/evidence/compaction-provenance.d.ts.map +1 -0
  239. package/dist/store/evidence/compaction-provenance.js +49 -0
  240. package/dist/store/evidence/compaction-provenance.js.map +1 -0
  241. package/dist/store/evidence/compaction-text.d.ts +16 -0
  242. package/dist/store/evidence/compaction-text.d.ts.map +1 -0
  243. package/dist/store/evidence/compaction-text.js +51 -0
  244. package/dist/store/evidence/compaction-text.js.map +1 -0
  245. package/dist/store/evidence/disk.d.ts +11 -0
  246. package/dist/store/evidence/disk.d.ts.map +1 -0
  247. package/dist/store/evidence/disk.js +367 -0
  248. package/dist/store/evidence/disk.js.map +1 -0
  249. package/dist/store/evidence/format.d.ts +25 -0
  250. package/dist/store/evidence/format.d.ts.map +1 -0
  251. package/dist/store/evidence/format.js +115 -0
  252. package/dist/store/evidence/format.js.map +1 -0
  253. package/dist/store/evidence/index-page.d.ts +208 -0
  254. package/dist/store/evidence/index-page.d.ts.map +1 -0
  255. package/dist/store/evidence/index-page.js +262 -0
  256. package/dist/store/evidence/index-page.js.map +1 -0
  257. package/dist/store/evidence/io.d.ts +24 -0
  258. package/dist/store/evidence/io.d.ts.map +1 -0
  259. package/dist/store/evidence/io.js +69 -0
  260. package/dist/store/evidence/io.js.map +1 -0
  261. package/dist/store/evidence/linked.d.ts +9 -0
  262. package/dist/store/evidence/linked.d.ts.map +1 -0
  263. package/dist/store/evidence/linked.js +314 -0
  264. package/dist/store/evidence/linked.js.map +1 -0
  265. package/dist/store/evidence/passages.d.ts +15 -0
  266. package/dist/store/evidence/passages.d.ts.map +1 -0
  267. package/dist/store/evidence/passages.js +72 -0
  268. package/dist/store/evidence/passages.js.map +1 -0
  269. package/dist/store/evidence/record-chain.d.ts +42 -0
  270. package/dist/store/evidence/record-chain.d.ts.map +1 -0
  271. package/dist/store/evidence/record-chain.js +111 -0
  272. package/dist/store/evidence/record-chain.js.map +1 -0
  273. package/dist/store/evidence/search-input.d.ts +21 -0
  274. package/dist/store/evidence/search-input.d.ts.map +1 -0
  275. package/dist/store/evidence/search-input.js +44 -0
  276. package/dist/store/evidence/search-input.js.map +1 -0
  277. package/dist/store/evidence/selection.d.ts +9 -0
  278. package/dist/store/evidence/selection.d.ts.map +1 -0
  279. package/dist/store/evidence/selection.js +17 -0
  280. package/dist/store/evidence/selection.js.map +1 -0
  281. package/dist/store/evidence/source-kind.d.ts +11 -0
  282. package/dist/store/evidence/source-kind.d.ts.map +1 -0
  283. package/dist/store/evidence/source-kind.js +26 -0
  284. package/dist/store/evidence/source-kind.js.map +1 -0
  285. package/dist/store/evidence/source-text.d.ts +51 -0
  286. package/dist/store/evidence/source-text.d.ts.map +1 -0
  287. package/dist/store/evidence/source-text.js +213 -0
  288. package/dist/store/evidence/source-text.js.map +1 -0
  289. package/dist/store/evidence/types.d.ts +162 -0
  290. package/dist/store/evidence/types.d.ts.map +1 -0
  291. package/dist/store/evidence/types.js +2 -0
  292. package/dist/store/evidence/types.js.map +1 -0
  293. package/dist/store/memory/disk.d.ts +2 -0
  294. package/dist/store/memory/disk.d.ts.map +1 -1
  295. package/dist/store/memory/disk.js +2 -1
  296. package/dist/store/memory/disk.js.map +1 -1
  297. package/dist/store/run/disk.d.ts +10 -0
  298. package/dist/store/run/disk.d.ts.map +1 -1
  299. package/dist/store/run/disk.js +87 -7
  300. package/dist/store/run/disk.js.map +1 -1
  301. package/dist/store/run/memory.d.ts +1 -0
  302. package/dist/store/run/memory.d.ts.map +1 -1
  303. package/dist/store/run/memory.js +10 -0
  304. package/dist/store/run/memory.js.map +1 -1
  305. package/dist/store/run/tool-executions.d.ts +13 -0
  306. package/dist/store/run/tool-executions.d.ts.map +1 -0
  307. package/dist/store/run/tool-executions.js +99 -0
  308. package/dist/store/run/tool-executions.js.map +1 -0
  309. package/dist/store/session/index.d.ts +2 -0
  310. package/dist/store/session/index.d.ts.map +1 -1
  311. package/dist/store/session/index.js +1 -0
  312. package/dist/store/session/index.js.map +1 -1
  313. package/dist/store/session/sqlite.d.ts +57 -0
  314. package/dist/store/session/sqlite.d.ts.map +1 -0
  315. package/dist/store/session/sqlite.js +430 -0
  316. package/dist/store/session/sqlite.js.map +1 -0
  317. package/dist/tools/builtins/job.d.ts.map +1 -1
  318. package/dist/tools/builtins/job.js +4 -5
  319. package/dist/tools/builtins/job.js.map +1 -1
  320. package/dist/tools/builtins/write-file.js +2 -2
  321. package/dist/tools/builtins/write-file.js.map +1 -1
  322. package/dist/tools/defineTool.d.ts +2 -1
  323. package/dist/tools/defineTool.d.ts.map +1 -1
  324. package/dist/tools/defineTool.js +1 -1
  325. package/dist/tools/defineTool.js.map +1 -1
  326. package/dist/tools/file-read-tracker.d.ts.map +1 -1
  327. package/dist/tools/file-read-tracker.js +12 -3
  328. package/dist/tools/file-read-tracker.js.map +1 -1
  329. package/dist/tools/resident-history.d.ts +10 -0
  330. package/dist/tools/resident-history.d.ts.map +1 -0
  331. package/dist/tools/resident-history.js +80 -0
  332. package/dist/tools/resident-history.js.map +1 -0
  333. package/dist/tools/resident-tool-evidence.d.ts +5 -0
  334. package/dist/tools/resident-tool-evidence.d.ts.map +1 -0
  335. package/dist/tools/resident-tool-evidence.js +57 -0
  336. package/dist/tools/resident-tool-evidence.js.map +1 -0
  337. package/dist/types/advisory/config.d.ts +7 -0
  338. package/dist/types/advisory/config.d.ts.map +1 -1
  339. package/dist/types/agent/reactive.d.ts +1 -0
  340. package/dist/types/agent/reactive.d.ts.map +1 -1
  341. package/dist/types/authorization/index.d.ts +9 -9
  342. package/dist/types/authorization/index.d.ts.map +1 -1
  343. package/dist/types/authorization/index.js +1 -1
  344. package/dist/types/authorization/index.js.map +1 -1
  345. package/dist/types/hitl/index.d.ts +4 -0
  346. package/dist/types/hitl/index.d.ts.map +1 -1
  347. package/dist/types/hitl/index.js.map +1 -1
  348. package/dist/types/message/index.d.ts +16 -2
  349. package/dist/types/message/index.d.ts.map +1 -1
  350. package/dist/types/message/index.js +8 -1
  351. package/dist/types/message/index.js.map +1 -1
  352. package/dist/types/provider/chat.d.ts +2 -0
  353. package/dist/types/provider/chat.d.ts.map +1 -1
  354. package/dist/types/provider/stream.d.ts +4 -0
  355. package/dist/types/provider/stream.d.ts.map +1 -1
  356. package/dist/types/run/answer-review.d.ts +34 -3
  357. package/dist/types/run/answer-review.d.ts.map +1 -1
  358. package/dist/types/run/config.d.ts +3 -0
  359. package/dist/types/run/config.d.ts.map +1 -1
  360. package/dist/types/run/entity.d.ts +7 -0
  361. package/dist/types/run/entity.d.ts.map +1 -1
  362. package/dist/types/run/events.d.ts +31 -8
  363. package/dist/types/run/events.d.ts.map +1 -1
  364. package/dist/types/run/events.js.map +1 -1
  365. package/dist/types/run/prepare-step.d.ts +53 -8
  366. package/dist/types/run/prepare-step.d.ts.map +1 -1
  367. package/dist/types/run/store.d.ts +29 -4
  368. package/dist/types/run/store.d.ts.map +1 -1
  369. package/dist/types/run/store.js +0 -28
  370. package/dist/types/run/store.js.map +1 -1
  371. package/dist/types/tool/index.d.ts +15 -1
  372. package/dist/types/tool/index.d.ts.map +1 -1
  373. package/dist/types/tool/index.js.map +1 -1
  374. package/dist/utils/await-with-abort.d.ts +8 -0
  375. package/dist/utils/await-with-abort.d.ts.map +1 -0
  376. package/dist/utils/await-with-abort.js +28 -0
  377. package/dist/utils/await-with-abort.js.map +1 -0
  378. package/dist/utils/evidence-time.d.ts +3 -0
  379. package/dist/utils/evidence-time.d.ts.map +1 -0
  380. package/dist/utils/evidence-time.js +10 -0
  381. package/dist/utils/evidence-time.js.map +1 -0
  382. package/dist/utils/evidence-tokens.d.ts +13 -0
  383. package/dist/utils/evidence-tokens.d.ts.map +1 -0
  384. package/dist/utils/evidence-tokens.js +29 -0
  385. package/dist/utils/evidence-tokens.js.map +1 -0
  386. package/package.json +1 -1
  387. package/src/advisory/executor.ts +15 -30
  388. package/src/advisory/history.ts +126 -0
  389. package/src/advisory/index.ts +5 -1
  390. package/src/agents/ReactiveAgent.ts +3 -0
  391. package/src/agents/runAgent.ts +14 -1
  392. package/src/compaction/manual.ts +30 -2
  393. package/src/compaction/summary.ts +4 -1
  394. package/src/config/runtime.ts +2 -2
  395. package/src/contracts/schemas.ts +1 -1
  396. package/src/eval/harness-protection.ts +79 -0
  397. package/src/eval/harness-verification.ts +31 -1
  398. package/src/eval/index.ts +1 -0
  399. package/src/manager/resident/activity.ts +187 -0
  400. package/src/manager/resident/agenda.ts +59 -5
  401. package/src/manager/resident/consumption.ts +311 -0
  402. package/src/manager/resident/evidence-recall.ts +363 -0
  403. package/src/manager/resident/history-disk.ts +63 -0
  404. package/src/manager/resident/history.ts +312 -0
  405. package/src/manager/resident/initiative.ts +13 -3
  406. package/src/manager/resident/learning-cycle.ts +499 -0
  407. package/src/manager/resident/learning-observation.ts +39 -0
  408. package/src/manager/resident/learning-store.ts +813 -0
  409. package/src/manager/resident/learning.ts +132 -8
  410. package/src/manager/resident/store.ts +31 -3
  411. package/src/manager/resident/tool-evidence.ts +412 -0
  412. package/src/manager/run/persistence.ts +6 -0
  413. package/src/plugin/loader.ts +8 -3
  414. package/src/prompt/coding-agent-doctrine.ts +2 -0
  415. package/src/prompt/index.ts +2 -0
  416. package/src/prompt/resident-learning.ts +143 -0
  417. package/src/prompt/resident-step.ts +83 -3
  418. package/src/provider/collect-chat-completion.ts +7 -6
  419. package/src/provider/fallback.ts +2 -1
  420. package/src/provider/stream-text.ts +56 -0
  421. package/src/public-runtime.ts +30 -0
  422. package/src/public-tools.ts +3 -0
  423. package/src/public-types.ts +98 -0
  424. package/src/run/LimitChecker.ts +3 -3
  425. package/src/run/evidence-query.ts +333 -0
  426. package/src/run/evidence-recall.ts +854 -0
  427. package/src/run/index.ts +11 -0
  428. package/src/run/json-claim-verifier.ts +298 -0
  429. package/src/run/preparation-context-error.ts +16 -0
  430. package/src/run-query/index.ts +1 -1
  431. package/src/runtime/query/callback-inference.ts +94 -0
  432. package/src/runtime/query/checkpoint.ts +14 -0
  433. package/src/runtime/query/events.ts +22 -0
  434. package/src/runtime/query/executor.ts +51 -9
  435. package/src/runtime/query/file-evidence-context.ts +65 -0
  436. package/src/runtime/query/guard.ts +2 -0
  437. package/src/runtime/query/index.ts +49 -25
  438. package/src/runtime/query/iteration/index.ts +269 -86
  439. package/src/runtime/query/iteration/phases/advisory.ts +3 -0
  440. package/src/runtime/query/iteration/phases/context.ts +2 -0
  441. package/src/runtime/query/iteration/phases/tool-review.ts +3 -0
  442. package/src/runtime/query/iteration/provider-rejected-image.ts +6 -1
  443. package/src/runtime/query/iteration/stream-turn.ts +35 -8
  444. package/src/runtime/query/resume-pending.ts +65 -40
  445. package/src/runtime/query/review-policy.ts +13 -9
  446. package/src/runtime/query/sandbox-lifecycle.ts +3 -0
  447. package/src/runtime/query/tool-output-budget.ts +65 -13
  448. package/src/runtime/query/tooling.ts +7 -1
  449. package/src/scheduler/completion-inbox.ts +26 -0
  450. package/src/store/evidence/compaction-archive.ts +139 -0
  451. package/src/store/evidence/compaction-provenance.ts +52 -0
  452. package/src/store/evidence/compaction-text.ts +61 -0
  453. package/src/store/evidence/disk.ts +461 -0
  454. package/src/store/evidence/format.ts +126 -0
  455. package/src/store/evidence/index-page.ts +292 -0
  456. package/src/store/evidence/io.ts +95 -0
  457. package/src/store/evidence/linked.ts +365 -0
  458. package/src/store/evidence/passages.ts +88 -0
  459. package/src/store/evidence/record-chain.ts +108 -0
  460. package/src/store/evidence/search-input.ts +62 -0
  461. package/src/store/evidence/selection.ts +25 -0
  462. package/src/store/evidence/source-kind.ts +36 -0
  463. package/src/store/evidence/source-text.ts +285 -0
  464. package/src/store/evidence/types.ts +177 -0
  465. package/src/store/memory/disk.ts +4 -1
  466. package/src/store/run/disk.ts +110 -8
  467. package/src/store/run/memory.ts +10 -0
  468. package/src/store/run/tool-executions.ts +112 -0
  469. package/src/store/session/index.ts +2 -0
  470. package/src/store/session/sqlite.ts +584 -0
  471. package/src/tools/builtins/job.ts +4 -5
  472. package/src/tools/builtins/write-file.ts +2 -2
  473. package/src/tools/defineTool.ts +4 -2
  474. package/src/tools/file-read-tracker.ts +8 -2
  475. package/src/tools/resident-history.ts +86 -0
  476. package/src/tools/resident-tool-evidence.ts +68 -0
  477. package/src/types/advisory/config.ts +7 -0
  478. package/src/types/agent/reactive.ts +1 -0
  479. package/src/types/authorization/index.ts +2 -2
  480. package/src/types/hitl/index.ts +4 -0
  481. package/src/types/message/index.ts +20 -0
  482. package/src/types/provider/chat.ts +2 -0
  483. package/src/types/provider/stream.ts +4 -0
  484. package/src/types/run/answer-review.ts +34 -3
  485. package/src/types/run/config.ts +3 -0
  486. package/src/types/run/entity.ts +7 -0
  487. package/src/types/run/events.ts +31 -8
  488. package/src/types/run/prepare-step.ts +59 -8
  489. package/src/types/run/store.ts +35 -4
  490. package/src/types/tool/index.ts +19 -1
  491. package/src/utils/await-with-abort.ts +26 -0
  492. package/src/utils/evidence-time.ts +9 -0
  493. package/src/utils/evidence-tokens.ts +32 -0
@@ -14,6 +14,7 @@ import { renderSkillsSection } from '../../../persona/assembler.js'
14
14
  import { resolveProviderCapabilities } from '../../../provider/capabilities.js'
15
15
  import { collectChatCompletion } from '../../../provider/collect-chat-completion.js'
16
16
  import { renderToolSchema } from '../../../registry/tool/schema.js'
17
+ import { PreparationContextError } from '../../../run/preparation-context-error.js'
17
18
  import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js'
18
19
  import {
19
20
  GENAI,
@@ -37,7 +38,7 @@ import {
37
38
  import type { ToolChoice } from '../../../types/provider/chat.js'
38
39
  import { classifyProviderError } from '../../../types/provider/errors.js'
39
40
  import type { ChatCompletionResponse } from '../../../types/provider/index.js'
40
- import type { AnswerReview } from '../../../types/run/answer-review.js'
41
+ import type { AnswerReview, AnswerReviewContext } from '../../../types/run/answer-review.js'
41
42
  import type {
42
43
  PrepareStepContext,
43
44
  PrepareStepResult,
@@ -53,6 +54,7 @@ import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/in
53
54
  import { toErrorMessage } from '../../../utils/error.js'
54
55
  import { stableDigest } from '../../../utils/hash.js'
55
56
  import { generateMessageId } from '../../../utils/id.js'
57
+ import { createCallbackInference } from '../callback-inference.js'
56
58
  import type { ToolCallOutcome } from '../executor.js'
57
59
  import { projectObservationContext } from '../observation-context.js'
58
60
  import { applyLifecycleHookResults } from '../plugin-hooks.js'
@@ -84,6 +86,20 @@ import { refreshWorkingMemory } from './phases/working-memory.js'
84
86
  import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js'
85
87
  import { streamProviderTurn } from './stream-turn.js'
86
88
 
89
+ type ReviewRequest = Pick<AnswerReviewContext, 'requestMessages' | 'latestUserMessage'>
90
+
91
+ /** A host reviewer is not the model transport, even when its cause is an HTTP failure. */
92
+ class AnswerReviewFailure extends NamzuError {
93
+ constructor(cause: unknown) {
94
+ super({
95
+ code: 'unknown',
96
+ message: `Answer review failed: ${toErrorMessage(cause)}`,
97
+ retryable: false,
98
+ details: { phase: 'answer-review' },
99
+ cause,
100
+ })
101
+ }
102
+ }
87
103
  export type { IterationContext } from './phases/index.js'
88
104
  export type { PhaseSignal } from './phases/index.js'
89
105
  export type { ToolReviewOutcome } from './phases/index.js'
@@ -98,6 +114,11 @@ export type { ToolReviewOutcome } from './phases/index.js'
98
114
  */
99
115
  const DEFAULT_ANSWER_REVIEW_LIMIT = 3
100
116
 
117
+ // Ending a run changes the available actions, not the strength of its evidence.
118
+ // Use the same standard for warning closure and empty-completion recovery.
119
+ const CLOSING_RESPONSE_GUIDANCE =
120
+ 'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.'
121
+
101
122
  /**
102
123
  * The share of a run's REMAINING time a settle-hold may take.
103
124
  *
@@ -162,6 +183,28 @@ export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
162
183
 
163
184
  export class IterationOrchestrator {
164
185
  private ctx: IterationContext
186
+ private advisoryTurn:
187
+ | {
188
+ readonly iteration: number
189
+ readonly requestMessages: readonly Message[]
190
+ readonly response: Message
191
+ }
192
+ | undefined
193
+
194
+ /** Live only within its iteration; never joined by a guessed array offset. */
195
+ getAdvisoryTurnContext():
196
+ | import('../../../advisory/executor.js').AdvisoryTurnContext
197
+ | undefined {
198
+ const turn = this.advisoryTurn
199
+ if (!turn || turn.iteration !== this.ctx.runMgr.currentIteration) return undefined
200
+ const start = this.ctx.runMgr.messages.indexOf(turn.response)
201
+ if (start < 0) return undefined
202
+ return {
203
+ iteration: turn.iteration,
204
+ requestMessages: turn.requestMessages,
205
+ subsequentMessages: this.ctx.runMgr.messages.slice(start),
206
+ }
207
+ }
165
208
  /** Rejections so far. See {@link DEFAULT_ANSWER_REVIEW_LIMIT}. */
166
209
  private answerReviewAttempts = 0
167
210
  /**
@@ -206,6 +249,7 @@ export class IterationOrchestrator {
206
249
  },
207
250
  }
208
251
  ctx.checkpointMgr.setLatestUserMessageSource(() => this.latestUserMessage)
252
+ ctx.checkpointMgr.setAnswerReviewAttemptsSource?.(() => this.answerReviewAttempts)
209
253
  ctx.checkpointMgr.setStructuredReviewAttemptsSource?.(() => this.structuredReviewAttempts)
210
254
  ctx.checkpointMgr.setNativeStructuredAttemptsSource?.(() => this.nativeStructuredAttempts)
211
255
  if (ctx.structuredOutput?.mode === 'native') {
@@ -223,6 +267,11 @@ export class IterationOrchestrator {
223
267
  const maxReviews = ctx.structuredOutput?.maxReviews
224
268
  if (maxReviews !== undefined && (!Number.isSafeInteger(maxReviews) || maxReviews < 0))
225
269
  throw new RangeError('structuredOutput.maxReviews must be a nonnegative safe integer')
270
+ if (
271
+ ctx.maxAnswerReviews !== undefined &&
272
+ (!Number.isSafeInteger(ctx.maxAnswerReviews) || ctx.maxAnswerReviews < 0)
273
+ )
274
+ throw new RangeError('maxAnswerReviews must be a nonnegative safe integer')
226
275
  }
227
276
 
228
277
  /**
@@ -306,6 +355,7 @@ export class IterationOrchestrator {
306
355
  const tracer = getTracer()
307
356
  // Resume hydration happens after construction, before the loop starts.
308
357
  this.latestUserMessage = this.ctx.checkpointMgr.restoredLatestUserMessage
358
+ this.answerReviewAttempts = this.ctx.checkpointMgr.restoredAnswerReviewAttempts ?? 0
309
359
  this.structuredReviewAttempts = this.ctx.checkpointMgr.restoredStructuredReviewAttempts ?? 0
310
360
  this.nativeStructuredAttempts = this.ctx.checkpointMgr.restoredNativeStructuredAttempts ?? 0
311
361
  if (!this.latestUserMessage) {
@@ -353,6 +403,13 @@ export class IterationOrchestrator {
353
403
  runMgr.setStopReason('structured_output_failed')
354
404
  break
355
405
  }
406
+ if (
407
+ this.ctx.reviewAnswer &&
408
+ this.answerReviewAttempts > (this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT)
409
+ ) {
410
+ runMgr.setStopReason('answer_rejected')
411
+ break
412
+ }
356
413
  if (
357
414
  this.ctx.structuredOutput?.review &&
358
415
  this.structuredReviewAttempts >
@@ -533,14 +590,15 @@ export class IterationOrchestrator {
533
590
  // Snapshot the cumulative counters so the step can report ITS
534
591
  // own usage rather than the run total.
535
592
  stepStartedAt = Date.now()
536
- usageBefore = { ...runMgr.tokenUsage }
537
- costBefore = { ...runMgr.costInfo }
538
593
 
539
594
  // Shape this step before calling the model. `stopWhen` decides
540
595
  // whether to keep going; this decides HOW. No-op when the host
541
596
  // supplied no hook.
542
597
  const contextModelBeforePreparation = this.ctx.contextModel ?? model
543
598
  const step = await this.prepareStep(iterationNum)
599
+ // Preparation inference belongs to the run, not the main-model step.
600
+ usageBefore = { ...runMgr.tokenUsage }
601
+ costBefore = { ...runMgr.costInfo }
544
602
  stepModel = step.model ?? model
545
603
  await this.selectContextModel(stepModel)
546
604
  // Preserve post-compaction preparation/recall semantics. A changed
@@ -570,7 +628,7 @@ export class IterationOrchestrator {
570
628
  ? [
571
629
  ...runMgr.messages,
572
630
  createRuntimeContextMessage(
573
- '[SYSTEM] You are approaching your resource limits. Provide your final, comprehensive response now based on everything you have gathered so far. Do not request any more tool calls.',
631
+ `[SYSTEM] You are approaching your resource limits. ${CLOSING_RESPONSE_GUIDANCE}`,
574
632
  'limit-finalization',
575
633
  ),
576
634
  ]
@@ -589,9 +647,9 @@ export class IterationOrchestrator {
589
647
  // mutation, and per-iteration this is trivial next to the model
590
648
  // call it precedes.
591
649
  // A step's skills and its guidance ride the same ephemeral
592
- // trailing system message. Appending leaves the cached prefix
593
- // intact; rewriting the run's own prompt to carry a phase's
594
- // skills would invalidate it on every iteration.
650
+ // system message. A driver may move it before history; changing
651
+ // system guidance can therefore affect prefix caching. Observations
652
+ // that need no system authority use step.context below.
595
653
  // `renderSkillsSection` already answers null for an empty list, so
596
654
  // there is no length check here — a second guard for the same
597
655
  // case is one more thing to keep in agreement with the first.
@@ -612,10 +670,9 @@ export class IterationOrchestrator {
612
670
  ? `Approval policy changed from "${policyChange.from}" to "${policyChange.to}" (${policyChange.reason}). Tool calls from here on are reviewed under the new policy.`
613
671
  : null
614
672
  // State that changed during the run, reported once per turn.
615
- // `turn` contributions land HERE and nowhere else: in the
616
- // system prompt they would be cached for the run or read as
617
- // a standing instruction, and either way the state they
618
- // exist to report goes stale silently.
673
+ // `turn` contributions are recomputed here, not fixed when the
674
+ // run's prompt is assembled. They retain system authority and
675
+ // may affect caching just like the other system contributions.
619
676
  const turnSections =
620
677
  this.ctx.promptContributions?.render('turn', {
621
678
  iteration: iterationNum,
@@ -627,10 +684,12 @@ export class IterationOrchestrator {
627
684
  const requestHistory = stepPreamble
628
685
  ? [...baseMessages, createSystemMessage(stepPreamble)]
629
686
  : [...baseMessages]
687
+ if (step.context) requestHistory.push(this.stepContextMessage(step.context))
630
688
  const messages = projectRequestRichContent(
631
689
  this.projectObservations(requestHistory),
632
690
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
633
691
  )
692
+ this.appendWorkContext(messages, iterationNum, step)
634
693
  await this.reportUnsupportedToolResults(messages)
635
694
  yield* this.ctx.drainPending()
636
695
 
@@ -741,7 +800,12 @@ export class IterationOrchestrator {
741
800
  model: requestedMember.model ?? stepModel,
742
801
  chainIndex: requestedMember.index,
743
802
  }
744
- const { response, messageId } = yield* streamProviderTurn(
803
+ const operatorInputAtDispatch = this.latestUserMessage
804
+ const latestReviewUserMessage =
805
+ (this.ctx.reviewAnswer || this.ctx.structuredOutput?.review) && operatorInputAtDispatch
806
+ ? structuredClone(operatorInputAtDispatch)
807
+ : undefined
808
+ const { response, messageId, requestMessages } = yield* streamProviderTurn(
745
809
  this.ctx.provider,
746
810
  {
747
811
  model: stepModel,
@@ -794,8 +858,15 @@ export class IterationOrchestrator {
794
858
  {
795
859
  onAccepted: (identity) => this.acceptProviderRejectedImage(identity),
796
860
  },
861
+ Boolean(
862
+ this.ctx.reviewAnswer || this.ctx.structuredOutput?.review || this.ctx.advisoryCtx,
863
+ ),
797
864
  )
798
865
  stepResponse = response
866
+ const reviewRequest: ReviewRequest = {
867
+ ...(requestMessages ? { requestMessages } : {}),
868
+ ...(latestReviewUserMessage ? { latestUserMessage: latestReviewUserMessage } : {}),
869
+ }
799
870
 
800
871
  // Who answered THIS turn.
801
872
  //
@@ -804,18 +875,9 @@ export class IterationOrchestrator {
804
875
  // request, so the member at the cursor when the stream ends is
805
876
  // the one whose bytes are in `response`.
806
877
  //
807
- // It is taken here rather than at `recordStep` several hundred
808
- // lines below, and the honest account of that is defence in
809
- // depth, not a defect it currently prevents. Moving it down
810
- // fails no test, because nothing between the two asks this
811
- // provider for anything: compaction and working memory run
812
- // BEFORE the turn, the advisory phase runs after the step is
813
- // already recorded, and the only thing in between is tool
814
- // execution. That is a fact about today's phase order, which a
815
- // later phase inserted here would change silently — and the
816
- // symptom would be a step attributed to a member that first
817
- // served the turn after it, which is the class of wrongness
818
- // this whole field exists to end.
878
+ // Capture before host review: its auxiliary inference can move
879
+ // the fallback cursor. Main-step usage and provenance must keep
880
+ // naming the provider that produced this candidate.
819
881
  const servedBy: StepProvenance = ((): StepProvenance => {
820
882
  const member = this.ctx.servingMember?.() ?? {
821
883
  index: 0,
@@ -935,8 +997,12 @@ export class IterationOrchestrator {
935
997
  ? { replayState: response.message.replayState }
936
998
  : {}),
937
999
  },
1000
+ response.message.textParts,
938
1001
  )
939
1002
  runMgr.pushMessage(assistantMsg)
1003
+ if (this.ctx.advisoryCtx && requestMessages) {
1004
+ this.advisoryTurn = { iteration: iterationNum, requestMessages, response: assistantMsg }
1005
+ }
940
1006
 
941
1007
  if (this.ctx.workingStateManager && this.ctx.compactionConfig && assistantMsg.content) {
942
1008
  extractFromAssistantMessage(
@@ -1061,7 +1127,12 @@ export class IterationOrchestrator {
1061
1127
  this.ctx.abortController.signal,
1062
1128
  )
1063
1129
  let outcome: 'accepted' | 'retry' | 'exhausted' | 'cancelled'
1064
- if (candidate.success) outcome = await this.reviewStructuredOutput(candidate.value)
1130
+ if (candidate.success)
1131
+ outcome = await this.reviewStructuredOutput(
1132
+ candidate.value,
1133
+ reviewRequest,
1134
+ stepModel,
1135
+ )
1065
1136
  else {
1066
1137
  this.nativeStructuredAttempts++
1067
1138
  runMgr.pushMessage(
@@ -1207,7 +1278,11 @@ export class IterationOrchestrator {
1207
1278
  // judge: bounded attempts, feedback as a user message, and
1208
1279
  // a loud stop rather than a loop.
1209
1280
  if (!forceFinalize && this.ctx.reviewAnswer) {
1210
- const review = await this.reviewAnswer(response.message.content ?? '')
1281
+ const review = await this.reviewAnswer(
1282
+ response.message.content ?? '',
1283
+ reviewRequest,
1284
+ stepModel,
1285
+ )
1211
1286
  if (this.ctx.abortController.signal.aborted) {
1212
1287
  runMgr.setStopReason('cancelled')
1213
1288
  runMgr.markCancelled()
@@ -1215,6 +1290,21 @@ export class IterationOrchestrator {
1215
1290
  }
1216
1291
  if (review && !review.accept) {
1217
1292
  const attempt = ++this.answerReviewAttempts
1293
+ runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'))
1294
+ // Commit the consumed allowance with its feedback before another
1295
+ // request, including exhaustion. Compaction cannot reset this quota.
1296
+ const checkpoint = await this.ctx.checkpointMgr.create(runMgr, iterationNum)
1297
+ await this.ctx.emitEvent({
1298
+ type: 'checkpoint_created',
1299
+ runId: runMgr.id,
1300
+ checkpointId: checkpoint.id,
1301
+ iteration: iterationNum,
1302
+ })
1303
+ if (this.ctx.abortController.signal.aborted) {
1304
+ runMgr.setStopReason('cancelled')
1305
+ runMgr.markCancelled()
1306
+ break
1307
+ }
1218
1308
  const limit = this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT
1219
1309
  if (attempt > limit) {
1220
1310
  this.ctx.log.warn('Answer rejected more times than the run allows', {
@@ -1230,7 +1320,6 @@ export class IterationOrchestrator {
1230
1320
  'namzu.retry.attempt': attempt,
1231
1321
  'namzu.runtime.limit': limit,
1232
1322
  })
1233
- runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'))
1234
1323
  await this.ctx.emitEvent({
1235
1324
  type: 'iteration_completed',
1236
1325
  runId: runMgr.id,
@@ -1272,7 +1361,12 @@ export class IterationOrchestrator {
1272
1361
  continue
1273
1362
  }
1274
1363
 
1275
- let closingStopReason: StopReason | undefined
1364
+ // A limit-requested summary bypasses prose review and further
1365
+ // work. Preserve that limit on settlement, even if the provider
1366
+ // reports a normal text completion and headroom still remains.
1367
+ let closingStopReason: StopReason | undefined = forceFinalize
1368
+ ? guardResult.stopReason
1369
+ : undefined
1276
1370
  if (!hasContent && !forceFinalize) {
1277
1371
  this.ctx.log.warn('Empty completion detected — requesting final summary', {
1278
1372
  [NAMZU.ITERATION]: iterationNum,
@@ -1353,6 +1447,8 @@ export class IterationOrchestrator {
1353
1447
  const structuredOutcome = await this.captureStructuredOutput(
1354
1448
  reviewOutcome.results,
1355
1449
  response,
1450
+ reviewRequest,
1451
+ stepModel,
1356
1452
  )
1357
1453
  if (
1358
1454
  structuredOutcome === 'retry' ||
@@ -1379,10 +1475,6 @@ export class IterationOrchestrator {
1379
1475
  break
1380
1476
  }
1381
1477
  if (structuredOutcome === 'accepted') {
1382
- this.ctx.log.info('Structured output produced — ending run', {
1383
- [NAMZU.RUN_ID]: runMgr.id,
1384
- [NAMZU.ITERATION]: iterationNum,
1385
- })
1386
1478
  await this.ctx.emitEvent({
1387
1479
  type: 'iteration_completed',
1388
1480
  runId: runMgr.id,
@@ -1395,6 +1487,21 @@ export class IterationOrchestrator {
1395
1487
  runMgr.markCancelled()
1396
1488
  break
1397
1489
  }
1490
+ if (!forceFinalize) {
1491
+ const inbound = this.deliverInbound()
1492
+ // Tool-result steering may already have been delivered by
1493
+ // runToolReview. Its candidate still answers the older input.
1494
+ if (inbound > 0 || this.latestUserMessage !== operatorInputAtDispatch) continue
1495
+ }
1496
+ if (this.ctx.abortController.signal.aborted) {
1497
+ runMgr.setStopReason('cancelled')
1498
+ runMgr.markCancelled()
1499
+ break
1500
+ }
1501
+ this.ctx.log.info('Structured output produced — ending run', {
1502
+ [NAMZU.RUN_ID]: runMgr.id,
1503
+ [NAMZU.ITERATION]: iterationNum,
1504
+ })
1398
1505
  this.publishStructuredOutput()
1399
1506
  runMgr.setStopReason('end_turn')
1400
1507
  break
@@ -1512,7 +1619,7 @@ export class IterationOrchestrator {
1512
1619
  // in the history the next request is built from.
1513
1620
  this.deliverInbound()
1514
1621
 
1515
- await runAdvisoryPhase(this.ctx, iterationNum, response)
1622
+ await runAdvisoryPhase(this.ctx, iterationNum, response, this.getAdvisoryTurnContext())
1516
1623
 
1517
1624
  if (this.ctx.pluginManager) {
1518
1625
  const hookResults = await this.ctx.pluginManager.executeHooks(
@@ -1633,6 +1740,7 @@ export class IterationOrchestrator {
1633
1740
  // would burn the budget to arrive at the same error.
1634
1741
  if (
1635
1742
  !overflowRelieved &&
1743
+ !(err instanceof AnswerReviewFailure) &&
1636
1744
  classifyProviderError(err, this.ctx.provider.id).code === 'context_length_exceeded'
1637
1745
  ) {
1638
1746
  overflowRelieved = true
@@ -1660,6 +1768,7 @@ export class IterationOrchestrator {
1660
1768
  iterSpan.recordException(err instanceof Error ? err : new Error(String(err)))
1661
1769
  throw err
1662
1770
  } finally {
1771
+ this.advisoryTurn = undefined
1663
1772
  // The only place the iteration span ends. It used to be ended at each of
1664
1773
  // seventeen exits, which is a rule every future edit has to
1665
1774
  // remember; a generator abandoned by its consumer never reached
@@ -1809,25 +1918,37 @@ export class IterationOrchestrator {
1809
1918
  )
1810
1919
  }
1811
1920
 
1812
- /**
1813
- * Ask the host how to shape this step.
1814
- *
1815
- * Fails OPEN on a throw — same reasoning as `stopWhen` and deliberately
1816
- * opposite to a guardrail: a broken step-shaping hook should not kill an
1817
- * otherwise healthy run, and unlike a safety check, nothing unsafe gets
1818
- * through when it is skipped.
1819
- */
1820
- /**
1821
- * A host's chance to refuse the next model call.
1822
- *
1823
- * Fails CLOSED, which is the opposite of `prepareStep` below and the
1824
- * reason they are separate hooks rather than one with two return
1825
- * shapes. A broken step-SHAPER skipped costs a run its per-step tuning;
1826
- * a broken step-REFUSER skipped is a refusal that did not happen, which
1827
- * is precisely what the hook exists to prevent. The thrown error's
1828
- * message becomes the reason, so an operator is not left with a run
1829
- * that stopped and no account of it.
1830
- */
1921
+ private stepContextMessage(content: string) {
1922
+ return createRuntimeContextMessage(
1923
+ `Current step context (runtime-generated; not a new user request):\n${content}`,
1924
+ 'step-context',
1925
+ )
1926
+ }
1927
+
1928
+ /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
1929
+ private appendWorkContext(
1930
+ messages: Message[],
1931
+ stepNumber: number,
1932
+ prepared: PrepareStepResult,
1933
+ ): void {
1934
+ const contributions = [
1935
+ this.ctx.completionInbox?.describeOwnedWork(),
1936
+ this.ctx.toolExecutor.describeFileEvidence(messages),
1937
+ ].filter((content): content is string => Boolean(content))
1938
+ if (contributions.length === 0) return
1939
+ let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0
1940
+ // Leave room for the actual task; admit whole contributions, never dangling partial references.
1941
+ if (room < 1_500) return
1942
+ for (const content of contributions) {
1943
+ if (!content || content.length > 8_000) continue
1944
+ const message = this.stepContextMessage(content)
1945
+ const tokens = estimateMessageTokens(message)
1946
+ if (tokens > Math.min(2_000, room - 1_000)) continue
1947
+ messages.push(message)
1948
+ room -= tokens
1949
+ }
1950
+ }
1951
+
1831
1952
  private stepContext(stepNumber: number, prepared: PrepareStepResult): PrepareStepContext {
1832
1953
  const model = prepared.model ?? this.ctx.runConfig.model
1833
1954
  const window = resolveContextWindow(
@@ -1841,7 +1962,9 @@ export class IterationOrchestrator {
1841
1962
  )
1842
1963
  const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null
1843
1964
  const preamble = [prepared.system, skills].filter(Boolean).join('\n\n')
1844
- const preparedTokens = preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0
1965
+ const preparedTokens =
1966
+ (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
1967
+ (prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0)
1845
1968
  const responseReserve = Math.min(
1846
1969
  prepared.maxResponseTokens ??
1847
1970
  this.ctx.runConfig.maxResponseTokens ??
@@ -1852,6 +1975,7 @@ export class IterationOrchestrator {
1852
1975
  runId: this.ctx.runMgr.id,
1853
1976
  stepNumber,
1854
1977
  messages: this.ctx.runMgr.messages,
1978
+ ...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
1855
1979
  ...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
1856
1980
  signal: this.ctx.abortController.signal,
1857
1981
  contextBudget: {
@@ -1868,6 +1992,7 @@ export class IterationOrchestrator {
1868
1992
  }
1869
1993
  }
1870
1994
 
1995
+ /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
1871
1996
  private async beforeStep(stepNumber: number): Promise<StepVeto | undefined> {
1872
1997
  const configured = this.ctx.beforeStep
1873
1998
  if (!configured) return undefined
@@ -1878,11 +2003,13 @@ export class IterationOrchestrator {
1878
2003
  }
1879
2004
  }
1880
2005
 
2006
+ /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
1881
2007
  private async prepareStep(stepNumber: number): Promise<{
1882
2008
  allowedTools?: string[]
1883
2009
  toolChoice?: ToolChoice
1884
2010
  model?: string
1885
2011
  system?: string
2012
+ context?: string
1886
2013
  skills?: readonly Skill[]
1887
2014
  temperature?: number
1888
2015
  maxResponseTokens?: number
@@ -1897,8 +2024,16 @@ export class IterationOrchestrator {
1897
2024
  // rather than an accident of install history.
1898
2025
  let result: PrepareStepResult = {}
1899
2026
  for (const stage of stages) {
2027
+ const inference = createCallbackInference(
2028
+ this.ctx,
2029
+ result.model ?? this.ctx.runConfig.model,
2030
+ 'preparation',
2031
+ )
1900
2032
  try {
1901
- const decided = await stage(this.stepContext(stepNumber, result))
2033
+ const decided = await stage({
2034
+ ...this.stepContext(stepNumber, result),
2035
+ generateText: inference.generateText,
2036
+ })
1902
2037
  if (decided) result = { ...result, ...decided }
1903
2038
  await this.selectContextModel(result.model ?? this.ctx.runConfig.model)
1904
2039
  } catch (err) {
@@ -1909,6 +2044,23 @@ export class IterationOrchestrator {
1909
2044
  'namzu.runtime.step_number': stepNumber,
1910
2045
  'exception.message': toErrorMessage(err),
1911
2046
  })
2047
+ // An SDK stage may report availability and validated fallback evidence
2048
+ // without exposing its error. Preserve prior decisions and the context budget;
2049
+ // ordinary exceptions still contribute nothing to the model request.
2050
+ if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
2051
+ const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0
2052
+ if (
2053
+ typeof err.context === 'string' &&
2054
+ err.context.length > 0 &&
2055
+ err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room))
2056
+ )
2057
+ result = {
2058
+ ...result,
2059
+ context: [result.context, err.context].filter(Boolean).join('\n\n'),
2060
+ }
2061
+ }
2062
+ } finally {
2063
+ inference.close()
1912
2064
  }
1913
2065
  }
1914
2066
 
@@ -1917,6 +2069,7 @@ export class IterationOrchestrator {
1917
2069
  toolChoice?: ToolChoice
1918
2070
  model?: string
1919
2071
  system?: string
2072
+ context?: string
1920
2073
  skills?: readonly Skill[]
1921
2074
  temperature?: number
1922
2075
  maxResponseTokens?: number
@@ -1953,6 +2106,7 @@ export class IterationOrchestrator {
1953
2106
  if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
1954
2107
  if (result.model !== undefined) prepared.model = result.model
1955
2108
  if (result.system !== undefined) prepared.system = result.system
2109
+ if (result.context !== undefined) prepared.context = result.context
1956
2110
  if (result.skills !== undefined) prepared.skills = result.skills
1957
2111
  if (result.temperature !== undefined) prepared.temperature = result.temperature
1958
2112
  if (result.maxResponseTokens !== undefined) {
@@ -2237,6 +2391,8 @@ export class IterationOrchestrator {
2237
2391
  private async captureStructuredOutput(
2238
2392
  results: readonly ToolCallOutcome[],
2239
2393
  response: ChatCompletionResponse,
2394
+ reviewRequest: ReviewRequest,
2395
+ model: string,
2240
2396
  ): Promise<'absent' | 'accepted' | 'retry' | 'exhausted' | 'cancelled'> {
2241
2397
  if (!this.needsStructuredOutput() || this.ctx.structuredOutput?.mode === 'native')
2242
2398
  return 'absent'
@@ -2264,7 +2420,7 @@ export class IterationOrchestrator {
2264
2420
  )
2265
2421
  parsed = hit.output
2266
2422
  }
2267
- return this.reviewStructuredOutput(parsed)
2423
+ return this.reviewStructuredOutput(parsed, reviewRequest, model)
2268
2424
  }
2269
2425
 
2270
2426
  private publishStructuredOutput(): void {
@@ -2274,6 +2430,8 @@ export class IterationOrchestrator {
2274
2430
 
2275
2431
  private async reviewStructuredOutput(
2276
2432
  parsed: unknown,
2433
+ reviewRequest: ReviewRequest,
2434
+ model: string,
2277
2435
  ): Promise<'accepted' | 'retry' | 'exhausted' | 'cancelled'> {
2278
2436
  if (this.ctx.abortController.signal.aborted) return 'cancelled'
2279
2437
  const reviewer = this.ctx.structuredOutput?.review
@@ -2286,22 +2444,27 @@ export class IterationOrchestrator {
2286
2444
  signal.addEventListener('abort', onAbort, { once: true })
2287
2445
  })
2288
2446
  let verdict: AnswerReview
2447
+ const inference = createCallbackInference(this.ctx, model, 'review')
2289
2448
  try {
2290
2449
  verdict = await Promise.race([
2291
- Promise.resolve().then(() =>
2292
- reviewer(structuredClone(parsed), {
2450
+ Promise.resolve().then(() => {
2451
+ signal.throwIfAborted()
2452
+ return reviewer(structuredClone(parsed), {
2293
2453
  runId: this.ctx.runMgr.id,
2294
2454
  iteration: this.ctx.runMgr.currentIteration,
2295
2455
  signal,
2296
2456
  messages: this.ctx.runMgr.messages,
2297
- }),
2298
- ),
2457
+ ...reviewRequest,
2458
+ generateText: inference.generateText,
2459
+ })
2460
+ }),
2299
2461
  aborted,
2300
2462
  ])
2301
2463
  } catch (error) {
2302
2464
  if (signal.aborted) return 'cancelled'
2303
2465
  throw error
2304
2466
  } finally {
2467
+ inference.close()
2305
2468
  signal.removeEventListener('abort', onAbort)
2306
2469
  }
2307
2470
  if (signal.aborted) return 'cancelled'
@@ -2335,33 +2498,48 @@ export class IterationOrchestrator {
2335
2498
  return 'accepted'
2336
2499
  }
2337
2500
 
2338
- /**
2339
- * Ask the host whether this answer is good enough.
2340
- *
2341
- * A hook that throws **accepts**, which is the opposite of what the
2342
- * safety gates do, and deliberately so. Those are asked "is this
2343
- * dangerous", where the cost of failing closed is one refused
2344
- * operation. This is asked "is this good enough", where failing closed
2345
- * means handing the answer back forever — so a broken judge would turn
2346
- * every run into a loop that ends on a budget error naming nothing. One
2347
- * unreviewed answer is the cheaper failure, and the throw is logged at
2348
- * `error` so it is not mistaken for approval.
2349
- */
2350
- private async reviewAnswer(answer: string): Promise<AnswerReview | undefined> {
2351
- if (!this.ctx.reviewAnswer) return undefined
2501
+ /** A reviewer failure aborts settlement; only an explicit rejection requests correction. */
2502
+ private async reviewAnswer(
2503
+ answer: string,
2504
+ reviewRequest: ReviewRequest,
2505
+ model: string,
2506
+ ): Promise<AnswerReview | undefined> {
2507
+ const reviewer = this.ctx.reviewAnswer
2508
+ const signal = this.ctx.abortController.signal
2509
+ if (!reviewer || signal.aborted) return undefined
2510
+ let onAbort = () => {}
2511
+ const aborted = new Promise<never>((_resolve, reject) => {
2512
+ onAbort = () => reject(signal.reason ?? new Error('Answer review cancelled'))
2513
+ signal.addEventListener('abort', onAbort, { once: true })
2514
+ })
2515
+ const inference = createCallbackInference(this.ctx, model, 'review')
2352
2516
  try {
2353
- return await this.ctx.reviewAnswer(answer, {
2354
- runId: this.ctx.runMgr.id,
2355
- iteration: this.ctx.runMgr.currentIteration,
2356
- signal: this.ctx.abortController.signal,
2357
- messages: this.ctx.runMgr.messages,
2358
- })
2359
- } catch (err) {
2360
- this.ctx.log.error('Answer review threw — accepting the answer unreviewed', {
2361
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2362
- 'exception.message': toErrorMessage(err),
2363
- })
2364
- return { accept: true }
2517
+ const verdict = await Promise.race([
2518
+ Promise.resolve().then(() => {
2519
+ signal.throwIfAborted()
2520
+ return reviewer(answer, {
2521
+ runId: this.ctx.runMgr.id,
2522
+ iteration: this.ctx.runMgr.currentIteration,
2523
+ signal,
2524
+ messages: this.ctx.runMgr.messages,
2525
+ ...reviewRequest,
2526
+ generateText: inference.generateText,
2527
+ })
2528
+ }),
2529
+ aborted,
2530
+ ])
2531
+ if (signal.aborted) return undefined
2532
+ if (!verdict || typeof verdict.accept !== 'boolean')
2533
+ throw new Error('Answer reviewer returned an invalid verdict')
2534
+ if (!verdict.accept && (typeof verdict.feedback !== 'string' || !verdict.feedback.trim()))
2535
+ throw new Error('Answer reviewer rejection requires feedback')
2536
+ return verdict
2537
+ } catch (error) {
2538
+ if (signal.aborted) return undefined
2539
+ throw new AnswerReviewFailure(error)
2540
+ } finally {
2541
+ inference.close()
2542
+ signal.removeEventListener('abort', onAbort)
2365
2543
  }
2366
2544
  }
2367
2545
 
@@ -2445,7 +2623,7 @@ export class IterationOrchestrator {
2445
2623
  const finalHistory = [
2446
2624
  ...this.ctx.runMgr.messages,
2447
2625
  createRuntimeContextMessage(
2448
- `[SYSTEM] Run is ending due to ${reason}. You MUST provide a final response now summarizing all your findings and work so far. Do not use any tools.`,
2626
+ `[SYSTEM] Run is ending due to ${reason}. ${CLOSING_RESPONSE_GUIDANCE}`,
2449
2627
  'limit-finalization',
2450
2628
  ),
2451
2629
  ]
@@ -2453,6 +2631,7 @@ export class IterationOrchestrator {
2453
2631
  this.projectObservations(finalHistory),
2454
2632
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
2455
2633
  )
2634
+ this.appendWorkContext(finalMessages, this.steps.length + 1, { model })
2456
2635
  await this.reportUnsupportedToolResults(finalMessages)
2457
2636
 
2458
2637
  // Same cache discipline as the forced-final iteration: keep the
@@ -2517,6 +2696,7 @@ export class IterationOrchestrator {
2517
2696
  ? { replayState: response.message.replayState }
2518
2697
  : {}),
2519
2698
  },
2699
+ response.message.textParts,
2520
2700
  )
2521
2701
  this.ctx.runMgr.pushMessage(assistantMsg)
2522
2702
 
@@ -2535,6 +2715,7 @@ export class IterationOrchestrator {
2535
2715
  stopReason: 'forced_finalize',
2536
2716
  usage: response.usage,
2537
2717
  content: response.message.content ?? undefined,
2718
+ ...(response.message.textParts ? { textParts: response.message.textParts } : {}),
2538
2719
  })
2539
2720
  } catch (err) {
2540
2721
  this.ctx.log.error('Failed to get final response', {
@@ -2556,6 +2737,8 @@ export class IterationOrchestrator {
2556
2737
  * threw and is left saying so rather than dressed up as something specific.
2557
2738
  */
2558
2739
  function describeStepFailure(err: unknown, providerId: string): StepFailure {
2740
+ if (err instanceof AnswerReviewFailure)
2741
+ return { message: err.message, code: 'unknown', retryable: false }
2559
2742
  const classified = classifyProviderError(err, providerId)
2560
2743
  return {
2561
2744
  message: toErrorMessage(err),