@namzu/sdk 38.2.1 → 40.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (602) hide show
  1. package/CHANGELOG.md +851 -0
  2. package/dist/advisory/executor.d.ts +10 -1
  3. package/dist/advisory/executor.d.ts.map +1 -1
  4. package/dist/advisory/executor.js +5 -26
  5. package/dist/advisory/executor.js.map +1 -1
  6. package/dist/advisory/history.d.ts +9 -0
  7. package/dist/advisory/history.d.ts.map +1 -0
  8. package/dist/advisory/history.js +120 -0
  9. package/dist/advisory/history.js.map +1 -0
  10. package/dist/advisory/index.d.ts +1 -1
  11. package/dist/advisory/index.d.ts.map +1 -1
  12. package/dist/advisory/index.js.map +1 -1
  13. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  14. package/dist/agents/ReactiveAgent.js +3 -0
  15. package/dist/agents/ReactiveAgent.js.map +1 -1
  16. package/dist/agents/runAgent.d.ts +10 -0
  17. package/dist/agents/runAgent.d.ts.map +1 -1
  18. package/dist/agents/runAgent.js +3 -0
  19. package/dist/agents/runAgent.js.map +1 -1
  20. package/dist/compaction/manual.d.ts +6 -0
  21. package/dist/compaction/manual.d.ts.map +1 -1
  22. package/dist/compaction/manual.js +21 -2
  23. package/dist/compaction/manual.js.map +1 -1
  24. package/dist/compaction/summary.d.ts.map +1 -1
  25. package/dist/compaction/summary.js +4 -1
  26. package/dist/compaction/summary.js.map +1 -1
  27. package/dist/config/runtime.js +2 -2
  28. package/dist/config/runtime.js.map +1 -1
  29. package/dist/connector/mcp/adapter.d.ts.map +1 -1
  30. package/dist/connector/mcp/adapter.js +20 -6
  31. package/dist/connector/mcp/adapter.js.map +1 -1
  32. package/dist/contracts/schemas.js +1 -1
  33. package/dist/contracts/schemas.js.map +1 -1
  34. package/dist/eval/harness-protection.d.ts +18 -0
  35. package/dist/eval/harness-protection.d.ts.map +1 -0
  36. package/dist/eval/harness-protection.js +58 -0
  37. package/dist/eval/harness-protection.js.map +1 -0
  38. package/dist/eval/harness-verification.d.ts +6 -1
  39. package/dist/eval/harness-verification.d.ts.map +1 -1
  40. package/dist/eval/harness-verification.js +19 -2
  41. package/dist/eval/harness-verification.js.map +1 -1
  42. package/dist/eval/index.d.ts +1 -0
  43. package/dist/eval/index.d.ts.map +1 -1
  44. package/dist/eval/index.js.map +1 -1
  45. package/dist/manager/resident/activity.d.ts +50 -0
  46. package/dist/manager/resident/activity.d.ts.map +1 -0
  47. package/dist/manager/resident/activity.js +125 -0
  48. package/dist/manager/resident/activity.js.map +1 -0
  49. package/dist/manager/resident/agenda.d.ts +13 -1
  50. package/dist/manager/resident/agenda.d.ts.map +1 -1
  51. package/dist/manager/resident/agenda.js +40 -5
  52. package/dist/manager/resident/agenda.js.map +1 -1
  53. package/dist/manager/resident/consumption.d.ts +104 -0
  54. package/dist/manager/resident/consumption.d.ts.map +1 -0
  55. package/dist/manager/resident/consumption.js +233 -0
  56. package/dist/manager/resident/consumption.js.map +1 -0
  57. package/dist/manager/resident/evidence-recall.d.ts +24 -0
  58. package/dist/manager/resident/evidence-recall.d.ts.map +1 -0
  59. package/dist/manager/resident/evidence-recall.js +295 -0
  60. package/dist/manager/resident/evidence-recall.js.map +1 -0
  61. package/dist/manager/resident/history-disk.d.ts +10 -0
  62. package/dist/manager/resident/history-disk.d.ts.map +1 -0
  63. package/dist/manager/resident/history-disk.js +50 -0
  64. package/dist/manager/resident/history-disk.js.map +1 -0
  65. package/dist/manager/resident/history.d.ts +79 -0
  66. package/dist/manager/resident/history.d.ts.map +1 -0
  67. package/dist/manager/resident/history.js +203 -0
  68. package/dist/manager/resident/history.js.map +1 -0
  69. package/dist/manager/resident/initiative.d.ts.map +1 -1
  70. package/dist/manager/resident/initiative.js +11 -3
  71. package/dist/manager/resident/initiative.js.map +1 -1
  72. package/dist/manager/resident/learning-cycle.d.ts +131 -0
  73. package/dist/manager/resident/learning-cycle.d.ts.map +1 -0
  74. package/dist/manager/resident/learning-cycle.js +306 -0
  75. package/dist/manager/resident/learning-cycle.js.map +1 -0
  76. package/dist/manager/resident/learning-observation.d.ts +80 -0
  77. package/dist/manager/resident/learning-observation.d.ts.map +1 -0
  78. package/dist/manager/resident/learning-observation.js +22 -0
  79. package/dist/manager/resident/learning-observation.js.map +1 -0
  80. package/dist/manager/resident/learning-store.d.ts +106 -0
  81. package/dist/manager/resident/learning-store.d.ts.map +1 -0
  82. package/dist/manager/resident/learning-store.js +598 -0
  83. package/dist/manager/resident/learning-store.js.map +1 -0
  84. package/dist/manager/resident/learning.d.ts +246 -3
  85. package/dist/manager/resident/learning.d.ts.map +1 -1
  86. package/dist/manager/resident/learning.js +96 -6
  87. package/dist/manager/resident/learning.js.map +1 -1
  88. package/dist/manager/resident/outbox.d.ts +4 -4
  89. package/dist/manager/resident/store.d.ts +37 -4
  90. package/dist/manager/resident/store.d.ts.map +1 -1
  91. package/dist/manager/resident/store.js +27 -3
  92. package/dist/manager/resident/store.js.map +1 -1
  93. package/dist/manager/resident/tool-evidence.d.ts +71 -0
  94. package/dist/manager/resident/tool-evidence.d.ts.map +1 -0
  95. package/dist/manager/resident/tool-evidence.js +285 -0
  96. package/dist/manager/resident/tool-evidence.js.map +1 -0
  97. package/dist/manager/run/persistence.d.ts +8 -0
  98. package/dist/manager/run/persistence.d.ts.map +1 -1
  99. package/dist/manager/run/persistence.js +18 -0
  100. package/dist/manager/run/persistence.js.map +1 -1
  101. package/dist/plugin/loader.d.ts.map +1 -1
  102. package/dist/plugin/loader.js +5 -3
  103. package/dist/plugin/loader.js.map +1 -1
  104. package/dist/prompt/coding-agent-doctrine.d.ts +1 -1
  105. package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
  106. package/dist/prompt/coding-agent-doctrine.js +2 -0
  107. package/dist/prompt/coding-agent-doctrine.js.map +1 -1
  108. package/dist/prompt/index.d.ts +2 -0
  109. package/dist/prompt/index.d.ts.map +1 -1
  110. package/dist/prompt/index.js +1 -0
  111. package/dist/prompt/index.js.map +1 -1
  112. package/dist/prompt/resident-learning.d.ts +19 -0
  113. package/dist/prompt/resident-learning.d.ts.map +1 -0
  114. package/dist/prompt/resident-learning.js +125 -0
  115. package/dist/prompt/resident-learning.js.map +1 -0
  116. package/dist/prompt/resident-step.d.ts +8 -1
  117. package/dist/prompt/resident-step.d.ts.map +1 -1
  118. package/dist/prompt/resident-step.js +65 -4
  119. package/dist/prompt/resident-step.js.map +1 -1
  120. package/dist/provider/collect-chat-completion.d.ts +2 -1
  121. package/dist/provider/collect-chat-completion.d.ts.map +1 -1
  122. package/dist/provider/collect-chat-completion.js +7 -6
  123. package/dist/provider/collect-chat-completion.js.map +1 -1
  124. package/dist/provider/fallback.d.ts.map +1 -1
  125. package/dist/provider/fallback.js +2 -1
  126. package/dist/provider/fallback.js.map +1 -1
  127. package/dist/provider/stream-text.d.ts +16 -0
  128. package/dist/provider/stream-text.d.ts.map +1 -0
  129. package/dist/provider/stream-text.js +51 -0
  130. package/dist/provider/stream-text.js.map +1 -0
  131. package/dist/public-runtime.d.ts +16 -4
  132. package/dist/public-runtime.d.ts.map +1 -1
  133. package/dist/public-runtime.js +17 -4
  134. package/dist/public-runtime.js.map +1 -1
  135. package/dist/public-tools.d.ts +13 -0
  136. package/dist/public-tools.d.ts.map +1 -1
  137. package/dist/public-tools.js +16 -0
  138. package/dist/public-tools.js.map +1 -1
  139. package/dist/public-types.d.ts +15 -3
  140. package/dist/public-types.d.ts.map +1 -1
  141. package/dist/registry/tool/execute.d.ts.map +1 -1
  142. package/dist/registry/tool/execute.js +2 -3
  143. package/dist/registry/tool/execute.js.map +1 -1
  144. package/dist/registry/tool/portable.d.ts +65 -0
  145. package/dist/registry/tool/portable.d.ts.map +1 -0
  146. package/dist/registry/tool/portable.js +244 -0
  147. package/dist/registry/tool/portable.js.map +1 -0
  148. package/dist/registry/tool/schema.d.ts +32 -5
  149. package/dist/registry/tool/schema.d.ts.map +1 -1
  150. package/dist/registry/tool/schema.js +35 -9
  151. package/dist/registry/tool/schema.js.map +1 -1
  152. package/dist/registry/toolset/catalog.js +8 -8
  153. package/dist/registry/toolset/catalog.js.map +1 -1
  154. package/dist/run/LimitChecker.js +3 -3
  155. package/dist/run/LimitChecker.js.map +1 -1
  156. package/dist/run/evidence-query.d.ts +41 -0
  157. package/dist/run/evidence-query.d.ts.map +1 -0
  158. package/dist/run/evidence-query.js +270 -0
  159. package/dist/run/evidence-query.js.map +1 -0
  160. package/dist/run/evidence-recall.d.ts +99 -0
  161. package/dist/run/evidence-recall.d.ts.map +1 -0
  162. package/dist/run/evidence-recall.js +633 -0
  163. package/dist/run/evidence-recall.js.map +1 -0
  164. package/dist/run/index.d.ts +2 -0
  165. package/dist/run/index.d.ts.map +1 -1
  166. package/dist/run/index.js +1 -0
  167. package/dist/run/index.js.map +1 -1
  168. package/dist/run/json-claim-verifier.d.ts +83 -0
  169. package/dist/run/json-claim-verifier.d.ts.map +1 -0
  170. package/dist/run/json-claim-verifier.js +200 -0
  171. package/dist/run/json-claim-verifier.js.map +1 -0
  172. package/dist/run/preparation-context-error.d.ts +10 -0
  173. package/dist/run/preparation-context-error.d.ts.map +1 -0
  174. package/dist/run/preparation-context-error.js +15 -0
  175. package/dist/run/preparation-context-error.js.map +1 -0
  176. package/dist/run-query/index.d.ts +3 -1
  177. package/dist/run-query/index.d.ts.map +1 -1
  178. package/dist/runtime/jobs/awaited-jobs.d.ts +215 -0
  179. package/dist/runtime/jobs/awaited-jobs.d.ts.map +1 -0
  180. package/dist/runtime/jobs/awaited-jobs.js +259 -0
  181. package/dist/runtime/jobs/awaited-jobs.js.map +1 -0
  182. package/dist/runtime/jobs/registry.d.ts +33 -2
  183. package/dist/runtime/jobs/registry.d.ts.map +1 -1
  184. package/dist/runtime/jobs/registry.js +37 -0
  185. package/dist/runtime/jobs/registry.js.map +1 -1
  186. package/dist/runtime/query/callback-inference.d.ts +8 -0
  187. package/dist/runtime/query/callback-inference.d.ts.map +1 -0
  188. package/dist/runtime/query/callback-inference.js +89 -0
  189. package/dist/runtime/query/callback-inference.js.map +1 -0
  190. package/dist/runtime/query/checkpoint.d.ts +4 -0
  191. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  192. package/dist/runtime/query/checkpoint.js +13 -0
  193. package/dist/runtime/query/checkpoint.js.map +1 -1
  194. package/dist/runtime/query/events.d.ts +1 -0
  195. package/dist/runtime/query/events.d.ts.map +1 -1
  196. package/dist/runtime/query/events.js +18 -0
  197. package/dist/runtime/query/events.js.map +1 -1
  198. package/dist/runtime/query/executor.d.ts +35 -1
  199. package/dist/runtime/query/executor.d.ts.map +1 -1
  200. package/dist/runtime/query/executor.js +77 -8
  201. package/dist/runtime/query/executor.js.map +1 -1
  202. package/dist/runtime/query/file-evidence-context.d.ts +5 -0
  203. package/dist/runtime/query/file-evidence-context.d.ts.map +1 -0
  204. package/dist/runtime/query/file-evidence-context.js +180 -0
  205. package/dist/runtime/query/file-evidence-context.js.map +1 -0
  206. package/dist/runtime/query/file-evidence-replay.d.ts +260 -0
  207. package/dist/runtime/query/file-evidence-replay.d.ts.map +1 -0
  208. package/dist/runtime/query/file-evidence-replay.js +647 -0
  209. package/dist/runtime/query/file-evidence-replay.js.map +1 -0
  210. package/dist/runtime/query/file-evidence-seed.d.ts +50 -0
  211. package/dist/runtime/query/file-evidence-seed.d.ts.map +1 -0
  212. package/dist/runtime/query/file-evidence-seed.js +100 -0
  213. package/dist/runtime/query/file-evidence-seed.js.map +1 -0
  214. package/dist/runtime/query/guard.d.ts.map +1 -1
  215. package/dist/runtime/query/guard.js +4 -0
  216. package/dist/runtime/query/guard.js.map +1 -1
  217. package/dist/runtime/query/index.d.ts +10 -1
  218. package/dist/runtime/query/index.d.ts.map +1 -1
  219. package/dist/runtime/query/index.js +133 -26
  220. package/dist/runtime/query/index.js.map +1 -1
  221. package/dist/runtime/query/iteration/index.d.ts +92 -36
  222. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  223. package/dist/runtime/query/iteration/index.js +420 -117
  224. package/dist/runtime/query/iteration/index.js.map +1 -1
  225. package/dist/runtime/query/iteration/phases/advisory.d.ts +2 -1
  226. package/dist/runtime/query/iteration/phases/advisory.d.ts.map +1 -1
  227. package/dist/runtime/query/iteration/phases/advisory.js +2 -1
  228. package/dist/runtime/query/iteration/phases/advisory.js.map +1 -1
  229. package/dist/runtime/query/iteration/phases/context.d.ts +12 -1
  230. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  231. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  232. package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
  233. package/dist/runtime/query/iteration/phases/tool-review.js +8 -1
  234. package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
  235. package/dist/runtime/query/iteration/provider-rejected-image.d.ts +2 -1
  236. package/dist/runtime/query/iteration/provider-rejected-image.d.ts.map +1 -1
  237. package/dist/runtime/query/iteration/provider-rejected-image.js +5 -2
  238. package/dist/runtime/query/iteration/provider-rejected-image.js.map +1 -1
  239. package/dist/runtime/query/iteration/stream-turn.d.ts +4 -1
  240. package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
  241. package/dist/runtime/query/iteration/stream-turn.js +22 -8
  242. package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
  243. package/dist/runtime/query/plugin-hooks.d.ts +14 -0
  244. package/dist/runtime/query/plugin-hooks.d.ts.map +1 -1
  245. package/dist/runtime/query/plugin-hooks.js +18 -0
  246. package/dist/runtime/query/plugin-hooks.js.map +1 -1
  247. package/dist/runtime/query/repeat-call.d.ts +17 -4
  248. package/dist/runtime/query/repeat-call.d.ts.map +1 -1
  249. package/dist/runtime/query/repeat-call.js +26 -19
  250. package/dist/runtime/query/repeat-call.js.map +1 -1
  251. package/dist/runtime/query/resume-pending.d.ts +18 -33
  252. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  253. package/dist/runtime/query/resume-pending.js +59 -42
  254. package/dist/runtime/query/resume-pending.js.map +1 -1
  255. package/dist/runtime/query/review-policy.d.ts +4 -4
  256. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  257. package/dist/runtime/query/review-policy.js +10 -9
  258. package/dist/runtime/query/review-policy.js.map +1 -1
  259. package/dist/runtime/query/sandbox-lifecycle.d.ts.map +1 -1
  260. package/dist/runtime/query/sandbox-lifecycle.js +4 -0
  261. package/dist/runtime/query/sandbox-lifecycle.js.map +1 -1
  262. package/dist/runtime/query/steering.d.ts +11 -1
  263. package/dist/runtime/query/steering.d.ts.map +1 -1
  264. package/dist/runtime/query/steering.js +12 -1
  265. package/dist/runtime/query/steering.js.map +1 -1
  266. package/dist/runtime/query/tool-output-budget.d.ts +10 -6
  267. package/dist/runtime/query/tool-output-budget.d.ts.map +1 -1
  268. package/dist/runtime/query/tool-output-budget.js +54 -12
  269. package/dist/runtime/query/tool-output-budget.js.map +1 -1
  270. package/dist/runtime/query/tooling.d.ts +5 -1
  271. package/dist/runtime/query/tooling.d.ts.map +1 -1
  272. package/dist/runtime/query/tooling.js +5 -0
  273. package/dist/runtime/query/tooling.js.map +1 -1
  274. package/dist/scheduler/completion-inbox.d.ts +49 -0
  275. package/dist/scheduler/completion-inbox.d.ts.map +1 -1
  276. package/dist/scheduler/completion-inbox.js +122 -2
  277. package/dist/scheduler/completion-inbox.js.map +1 -1
  278. package/dist/store/evidence/compaction-archive.d.ts +109 -0
  279. package/dist/store/evidence/compaction-archive.d.ts.map +1 -0
  280. package/dist/store/evidence/compaction-archive.js +125 -0
  281. package/dist/store/evidence/compaction-archive.js.map +1 -0
  282. package/dist/store/evidence/compaction-provenance.d.ts +7 -0
  283. package/dist/store/evidence/compaction-provenance.d.ts.map +1 -0
  284. package/dist/store/evidence/compaction-provenance.js +49 -0
  285. package/dist/store/evidence/compaction-provenance.js.map +1 -0
  286. package/dist/store/evidence/compaction-text.d.ts +16 -0
  287. package/dist/store/evidence/compaction-text.d.ts.map +1 -0
  288. package/dist/store/evidence/compaction-text.js +51 -0
  289. package/dist/store/evidence/compaction-text.js.map +1 -0
  290. package/dist/store/evidence/disk.d.ts +11 -0
  291. package/dist/store/evidence/disk.d.ts.map +1 -0
  292. package/dist/store/evidence/disk.js +367 -0
  293. package/dist/store/evidence/disk.js.map +1 -0
  294. package/dist/store/evidence/format.d.ts +25 -0
  295. package/dist/store/evidence/format.d.ts.map +1 -0
  296. package/dist/store/evidence/format.js +115 -0
  297. package/dist/store/evidence/format.js.map +1 -0
  298. package/dist/store/evidence/index-page.d.ts +208 -0
  299. package/dist/store/evidence/index-page.d.ts.map +1 -0
  300. package/dist/store/evidence/index-page.js +262 -0
  301. package/dist/store/evidence/index-page.js.map +1 -0
  302. package/dist/store/evidence/io.d.ts +24 -0
  303. package/dist/store/evidence/io.d.ts.map +1 -0
  304. package/dist/store/evidence/io.js +69 -0
  305. package/dist/store/evidence/io.js.map +1 -0
  306. package/dist/store/evidence/linked.d.ts +9 -0
  307. package/dist/store/evidence/linked.d.ts.map +1 -0
  308. package/dist/store/evidence/linked.js +314 -0
  309. package/dist/store/evidence/linked.js.map +1 -0
  310. package/dist/store/evidence/passages.d.ts +15 -0
  311. package/dist/store/evidence/passages.d.ts.map +1 -0
  312. package/dist/store/evidence/passages.js +72 -0
  313. package/dist/store/evidence/passages.js.map +1 -0
  314. package/dist/store/evidence/record-chain.d.ts +42 -0
  315. package/dist/store/evidence/record-chain.d.ts.map +1 -0
  316. package/dist/store/evidence/record-chain.js +111 -0
  317. package/dist/store/evidence/record-chain.js.map +1 -0
  318. package/dist/store/evidence/search-input.d.ts +21 -0
  319. package/dist/store/evidence/search-input.d.ts.map +1 -0
  320. package/dist/store/evidence/search-input.js +44 -0
  321. package/dist/store/evidence/search-input.js.map +1 -0
  322. package/dist/store/evidence/selection.d.ts +9 -0
  323. package/dist/store/evidence/selection.d.ts.map +1 -0
  324. package/dist/store/evidence/selection.js +17 -0
  325. package/dist/store/evidence/selection.js.map +1 -0
  326. package/dist/store/evidence/source-kind.d.ts +11 -0
  327. package/dist/store/evidence/source-kind.d.ts.map +1 -0
  328. package/dist/store/evidence/source-kind.js +26 -0
  329. package/dist/store/evidence/source-kind.js.map +1 -0
  330. package/dist/store/evidence/source-text.d.ts +51 -0
  331. package/dist/store/evidence/source-text.d.ts.map +1 -0
  332. package/dist/store/evidence/source-text.js +213 -0
  333. package/dist/store/evidence/source-text.js.map +1 -0
  334. package/dist/store/evidence/types.d.ts +162 -0
  335. package/dist/store/evidence/types.d.ts.map +1 -0
  336. package/dist/store/evidence/types.js +2 -0
  337. package/dist/store/evidence/types.js.map +1 -0
  338. package/dist/store/memory/disk.d.ts +2 -0
  339. package/dist/store/memory/disk.d.ts.map +1 -1
  340. package/dist/store/memory/disk.js +2 -1
  341. package/dist/store/memory/disk.js.map +1 -1
  342. package/dist/store/run/disk.d.ts +10 -0
  343. package/dist/store/run/disk.d.ts.map +1 -1
  344. package/dist/store/run/disk.js +87 -7
  345. package/dist/store/run/disk.js.map +1 -1
  346. package/dist/store/run/memory.d.ts +1 -0
  347. package/dist/store/run/memory.d.ts.map +1 -1
  348. package/dist/store/run/memory.js +10 -0
  349. package/dist/store/run/memory.js.map +1 -1
  350. package/dist/store/run/tool-executions.d.ts +13 -0
  351. package/dist/store/run/tool-executions.d.ts.map +1 -0
  352. package/dist/store/run/tool-executions.js +99 -0
  353. package/dist/store/run/tool-executions.js.map +1 -0
  354. package/dist/store/session/index.d.ts +2 -0
  355. package/dist/store/session/index.d.ts.map +1 -1
  356. package/dist/store/session/index.js +1 -0
  357. package/dist/store/session/index.js.map +1 -1
  358. package/dist/store/session/sqlite.d.ts +57 -0
  359. package/dist/store/session/sqlite.d.ts.map +1 -0
  360. package/dist/store/session/sqlite.js +430 -0
  361. package/dist/store/session/sqlite.js.map +1 -0
  362. package/dist/tools/builtins/bash.d.ts.map +1 -1
  363. package/dist/tools/builtins/bash.js +4 -10
  364. package/dist/tools/builtins/bash.js.map +1 -1
  365. package/dist/tools/builtins/edit-apply.d.ts +126 -0
  366. package/dist/tools/builtins/edit-apply.d.ts.map +1 -0
  367. package/dist/tools/builtins/edit-apply.js +360 -0
  368. package/dist/tools/builtins/edit-apply.js.map +1 -0
  369. package/dist/tools/builtins/edit.d.ts +143 -1
  370. package/dist/tools/builtins/edit.d.ts.map +1 -1
  371. package/dist/tools/builtins/edit.js +37 -219
  372. package/dist/tools/builtins/edit.js.map +1 -1
  373. package/dist/tools/builtins/index.d.ts +1 -0
  374. package/dist/tools/builtins/index.d.ts.map +1 -1
  375. package/dist/tools/builtins/index.js +9 -3
  376. package/dist/tools/builtins/index.js.map +1 -1
  377. package/dist/tools/builtins/job.d.ts.map +1 -1
  378. package/dist/tools/builtins/job.js +5 -6
  379. package/dist/tools/builtins/job.js.map +1 -1
  380. package/dist/tools/builtins/read-file.d.ts +2 -2
  381. package/dist/tools/builtins/read-file.d.ts.map +1 -1
  382. package/dist/tools/builtins/read-file.js +50 -65
  383. package/dist/tools/builtins/read-file.js.map +1 -1
  384. package/dist/tools/builtins/read-render.d.ts +56 -0
  385. package/dist/tools/builtins/read-render.d.ts.map +1 -0
  386. package/dist/tools/builtins/read-render.js +73 -0
  387. package/dist/tools/builtins/read-render.js.map +1 -0
  388. package/dist/tools/builtins/wait-for-job-bounds.d.ts +67 -0
  389. package/dist/tools/builtins/wait-for-job-bounds.d.ts.map +1 -0
  390. package/dist/tools/builtins/wait-for-job-bounds.js +108 -0
  391. package/dist/tools/builtins/wait-for-job-bounds.js.map +1 -0
  392. package/dist/tools/builtins/wait-for-job.d.ts +6 -0
  393. package/dist/tools/builtins/wait-for-job.d.ts.map +1 -0
  394. package/dist/tools/builtins/wait-for-job.js +162 -0
  395. package/dist/tools/builtins/wait-for-job.js.map +1 -0
  396. package/dist/tools/builtins/write-file.js +7 -2
  397. package/dist/tools/builtins/write-file.js.map +1 -1
  398. package/dist/tools/coordinator/index.d.ts.map +1 -1
  399. package/dist/tools/coordinator/index.js +1 -7
  400. package/dist/tools/coordinator/index.js.map +1 -1
  401. package/dist/tools/defineTool.d.ts +2 -1
  402. package/dist/tools/defineTool.d.ts.map +1 -1
  403. package/dist/tools/defineTool.js +1 -1
  404. package/dist/tools/defineTool.js.map +1 -1
  405. package/dist/tools/file-read-tracker.d.ts.map +1 -1
  406. package/dist/tools/file-read-tracker.js +90 -3
  407. package/dist/tools/file-read-tracker.js.map +1 -1
  408. package/dist/tools/resident-history.d.ts +10 -0
  409. package/dist/tools/resident-history.d.ts.map +1 -0
  410. package/dist/tools/resident-history.js +80 -0
  411. package/dist/tools/resident-history.js.map +1 -0
  412. package/dist/tools/resident-tool-evidence.d.ts +5 -0
  413. package/dist/tools/resident-tool-evidence.d.ts.map +1 -0
  414. package/dist/tools/resident-tool-evidence.js +57 -0
  415. package/dist/tools/resident-tool-evidence.js.map +1 -0
  416. package/dist/types/advisory/config.d.ts +7 -0
  417. package/dist/types/advisory/config.d.ts.map +1 -1
  418. package/dist/types/agent/reactive.d.ts +1 -0
  419. package/dist/types/agent/reactive.d.ts.map +1 -1
  420. package/dist/types/authorization/index.d.ts +9 -9
  421. package/dist/types/authorization/index.d.ts.map +1 -1
  422. package/dist/types/authorization/index.js +1 -1
  423. package/dist/types/authorization/index.js.map +1 -1
  424. package/dist/types/hitl/index.d.ts +4 -0
  425. package/dist/types/hitl/index.d.ts.map +1 -1
  426. package/dist/types/hitl/index.js.map +1 -1
  427. package/dist/types/message/index.d.ts +16 -2
  428. package/dist/types/message/index.d.ts.map +1 -1
  429. package/dist/types/message/index.js +10 -1
  430. package/dist/types/message/index.js.map +1 -1
  431. package/dist/types/provider/chat.d.ts +2 -0
  432. package/dist/types/provider/chat.d.ts.map +1 -1
  433. package/dist/types/provider/stream.d.ts +4 -0
  434. package/dist/types/provider/stream.d.ts.map +1 -1
  435. package/dist/types/run/answer-review.d.ts +34 -3
  436. package/dist/types/run/answer-review.d.ts.map +1 -1
  437. package/dist/types/run/config.d.ts +3 -0
  438. package/dist/types/run/config.d.ts.map +1 -1
  439. package/dist/types/run/entity.d.ts +20 -0
  440. package/dist/types/run/entity.d.ts.map +1 -1
  441. package/dist/types/run/events.d.ts +31 -8
  442. package/dist/types/run/events.d.ts.map +1 -1
  443. package/dist/types/run/events.js.map +1 -1
  444. package/dist/types/run/prepare-step.d.ts +53 -8
  445. package/dist/types/run/prepare-step.d.ts.map +1 -1
  446. package/dist/types/run/store.d.ts +29 -4
  447. package/dist/types/run/store.d.ts.map +1 -1
  448. package/dist/types/run/store.js +0 -28
  449. package/dist/types/run/store.js.map +1 -1
  450. package/dist/types/sandbox/index.d.ts +15 -14
  451. package/dist/types/sandbox/index.d.ts.map +1 -1
  452. package/dist/types/sandbox/index.js.map +1 -1
  453. package/dist/types/tool/index.d.ts +124 -1
  454. package/dist/types/tool/index.d.ts.map +1 -1
  455. package/dist/types/tool/index.js.map +1 -1
  456. package/dist/utils/await-with-abort.d.ts +8 -0
  457. package/dist/utils/await-with-abort.d.ts.map +1 -0
  458. package/dist/utils/await-with-abort.js +28 -0
  459. package/dist/utils/await-with-abort.js.map +1 -0
  460. package/dist/utils/env.d.ts +19 -0
  461. package/dist/utils/env.d.ts.map +1 -0
  462. package/dist/utils/env.js +25 -0
  463. package/dist/utils/env.js.map +1 -0
  464. package/dist/utils/evidence-time.d.ts +3 -0
  465. package/dist/utils/evidence-time.d.ts.map +1 -0
  466. package/dist/utils/evidence-time.js +10 -0
  467. package/dist/utils/evidence-time.js.map +1 -0
  468. package/dist/utils/evidence-tokens.d.ts +13 -0
  469. package/dist/utils/evidence-tokens.d.ts.map +1 -0
  470. package/dist/utils/evidence-tokens.js +29 -0
  471. package/dist/utils/evidence-tokens.js.map +1 -0
  472. package/package.json +1 -1
  473. package/src/advisory/executor.ts +15 -30
  474. package/src/advisory/history.ts +126 -0
  475. package/src/advisory/index.ts +5 -1
  476. package/src/agents/ReactiveAgent.ts +3 -0
  477. package/src/agents/runAgent.ts +14 -1
  478. package/src/compaction/manual.ts +30 -2
  479. package/src/compaction/summary.ts +4 -1
  480. package/src/config/runtime.ts +2 -2
  481. package/src/connector/mcp/adapter.ts +20 -6
  482. package/src/contracts/schemas.ts +1 -1
  483. package/src/eval/harness-protection.ts +79 -0
  484. package/src/eval/harness-verification.ts +31 -1
  485. package/src/eval/index.ts +1 -0
  486. package/src/manager/resident/activity.ts +187 -0
  487. package/src/manager/resident/agenda.ts +59 -5
  488. package/src/manager/resident/consumption.ts +311 -0
  489. package/src/manager/resident/evidence-recall.ts +363 -0
  490. package/src/manager/resident/history-disk.ts +63 -0
  491. package/src/manager/resident/history.ts +312 -0
  492. package/src/manager/resident/initiative.ts +13 -3
  493. package/src/manager/resident/learning-cycle.ts +499 -0
  494. package/src/manager/resident/learning-observation.ts +39 -0
  495. package/src/manager/resident/learning-store.ts +813 -0
  496. package/src/manager/resident/learning.ts +132 -8
  497. package/src/manager/resident/store.ts +31 -3
  498. package/src/manager/resident/tool-evidence.ts +412 -0
  499. package/src/manager/run/persistence.ts +18 -0
  500. package/src/plugin/loader.ts +8 -3
  501. package/src/prompt/coding-agent-doctrine.ts +2 -0
  502. package/src/prompt/index.ts +2 -0
  503. package/src/prompt/resident-learning.ts +143 -0
  504. package/src/prompt/resident-step.ts +83 -3
  505. package/src/provider/collect-chat-completion.ts +7 -6
  506. package/src/provider/fallback.ts +2 -1
  507. package/src/provider/stream-text.ts +56 -0
  508. package/src/public-runtime.ts +39 -1
  509. package/src/public-tools.ts +21 -0
  510. package/src/public-types.ts +98 -0
  511. package/src/registry/tool/execute.ts +2 -4
  512. package/src/registry/tool/portable.ts +264 -0
  513. package/src/registry/tool/schema.ts +38 -8
  514. package/src/registry/toolset/catalog.ts +8 -9
  515. package/src/run/LimitChecker.ts +3 -3
  516. package/src/run/evidence-query.ts +333 -0
  517. package/src/run/evidence-recall.ts +854 -0
  518. package/src/run/index.ts +11 -0
  519. package/src/run/json-claim-verifier.ts +298 -0
  520. package/src/run/preparation-context-error.ts +16 -0
  521. package/src/run-query/index.ts +1 -1
  522. package/src/runtime/jobs/awaited-jobs.ts +271 -0
  523. package/src/runtime/jobs/registry.ts +50 -0
  524. package/src/runtime/query/callback-inference.ts +94 -0
  525. package/src/runtime/query/checkpoint.ts +14 -0
  526. package/src/runtime/query/events.ts +22 -0
  527. package/src/runtime/query/executor.ts +100 -10
  528. package/src/runtime/query/file-evidence-context.ts +209 -0
  529. package/src/runtime/query/file-evidence-replay.ts +776 -0
  530. package/src/runtime/query/file-evidence-seed.ts +126 -0
  531. package/src/runtime/query/guard.ts +2 -0
  532. package/src/runtime/query/index.ts +153 -27
  533. package/src/runtime/query/iteration/index.ts +467 -110
  534. package/src/runtime/query/iteration/phases/advisory.ts +3 -0
  535. package/src/runtime/query/iteration/phases/context.ts +12 -0
  536. package/src/runtime/query/iteration/phases/tool-review.ts +7 -0
  537. package/src/runtime/query/iteration/provider-rejected-image.ts +6 -1
  538. package/src/runtime/query/iteration/stream-turn.ts +35 -8
  539. package/src/runtime/query/plugin-hooks.ts +20 -0
  540. package/src/runtime/query/repeat-call.ts +28 -18
  541. package/src/runtime/query/resume-pending.ts +65 -40
  542. package/src/runtime/query/review-policy.ts +13 -9
  543. package/src/runtime/query/sandbox-lifecycle.ts +3 -0
  544. package/src/runtime/query/steering.ts +11 -0
  545. package/src/runtime/query/tool-output-budget.ts +65 -13
  546. package/src/runtime/query/tooling.ts +10 -1
  547. package/src/scheduler/completion-inbox.ts +124 -2
  548. package/src/store/evidence/compaction-archive.ts +139 -0
  549. package/src/store/evidence/compaction-provenance.ts +52 -0
  550. package/src/store/evidence/compaction-text.ts +61 -0
  551. package/src/store/evidence/disk.ts +461 -0
  552. package/src/store/evidence/format.ts +126 -0
  553. package/src/store/evidence/index-page.ts +292 -0
  554. package/src/store/evidence/io.ts +95 -0
  555. package/src/store/evidence/linked.ts +365 -0
  556. package/src/store/evidence/passages.ts +88 -0
  557. package/src/store/evidence/record-chain.ts +108 -0
  558. package/src/store/evidence/search-input.ts +62 -0
  559. package/src/store/evidence/selection.ts +25 -0
  560. package/src/store/evidence/source-kind.ts +36 -0
  561. package/src/store/evidence/source-text.ts +285 -0
  562. package/src/store/evidence/types.ts +177 -0
  563. package/src/store/memory/disk.ts +4 -1
  564. package/src/store/run/disk.ts +110 -8
  565. package/src/store/run/memory.ts +10 -0
  566. package/src/store/run/tool-executions.ts +112 -0
  567. package/src/store/session/index.ts +2 -0
  568. package/src/store/session/sqlite.ts +584 -0
  569. package/src/tools/builtins/bash.ts +4 -10
  570. package/src/tools/builtins/edit-apply.ts +456 -0
  571. package/src/tools/builtins/edit.ts +39 -270
  572. package/src/tools/builtins/index.ts +9 -3
  573. package/src/tools/builtins/job.ts +5 -6
  574. package/src/tools/builtins/read-file.ts +56 -77
  575. package/src/tools/builtins/read-render.ts +104 -0
  576. package/src/tools/builtins/wait-for-job-bounds.ts +179 -0
  577. package/src/tools/builtins/wait-for-job.ts +184 -0
  578. package/src/tools/builtins/write-file.ts +7 -2
  579. package/src/tools/coordinator/index.ts +1 -7
  580. package/src/tools/defineTool.ts +4 -2
  581. package/src/tools/file-read-tracker.ts +86 -2
  582. package/src/tools/resident-history.ts +86 -0
  583. package/src/tools/resident-tool-evidence.ts +68 -0
  584. package/src/types/advisory/config.ts +7 -0
  585. package/src/types/agent/reactive.ts +1 -0
  586. package/src/types/authorization/index.ts +2 -2
  587. package/src/types/hitl/index.ts +4 -0
  588. package/src/types/message/index.ts +22 -0
  589. package/src/types/provider/chat.ts +2 -0
  590. package/src/types/provider/stream.ts +4 -0
  591. package/src/types/run/answer-review.ts +34 -3
  592. package/src/types/run/config.ts +3 -0
  593. package/src/types/run/entity.ts +21 -0
  594. package/src/types/run/events.ts +31 -8
  595. package/src/types/run/prepare-step.ts +59 -8
  596. package/src/types/run/store.ts +35 -4
  597. package/src/types/sandbox/index.ts +15 -14
  598. package/src/types/tool/index.ts +123 -1
  599. package/src/utils/await-with-abort.ts +26 -0
  600. package/src/utils/env.ts +23 -0
  601. package/src/utils/evidence-time.ts +9 -0
  602. package/src/utils/evidence-tokens.ts +32 -0
@@ -14,6 +14,7 @@ import { renderSkillsSection } from '../../../persona/assembler.js'
14
14
  import { resolveProviderCapabilities } from '../../../provider/capabilities.js'
15
15
  import { collectChatCompletion } from '../../../provider/collect-chat-completion.js'
16
16
  import { renderToolSchema } from '../../../registry/tool/schema.js'
17
+ import { PreparationContextError } from '../../../run/preparation-context-error.js'
17
18
  import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js'
18
19
  import {
19
20
  GENAI,
@@ -37,7 +38,7 @@ import {
37
38
  import type { ToolChoice } from '../../../types/provider/chat.js'
38
39
  import { classifyProviderError } from '../../../types/provider/errors.js'
39
40
  import type { ChatCompletionResponse } from '../../../types/provider/index.js'
40
- import type { AnswerReview } from '../../../types/run/answer-review.js'
41
+ import type { AnswerReview, AnswerReviewContext } from '../../../types/run/answer-review.js'
41
42
  import type {
42
43
  PrepareStepContext,
43
44
  PrepareStepResult,
@@ -50,9 +51,11 @@ import type {
50
51
  } from '../../../types/run/index.js'
51
52
  import type { Skill } from '../../../types/skills/index.js'
52
53
  import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/index.js'
54
+ import { readPositiveIntEnv } from '../../../utils/env.js'
53
55
  import { toErrorMessage } from '../../../utils/error.js'
54
56
  import { stableDigest } from '../../../utils/hash.js'
55
57
  import { generateMessageId } from '../../../utils/id.js'
58
+ import { createCallbackInference } from '../callback-inference.js'
56
59
  import type { ToolCallOutcome } from '../executor.js'
57
60
  import { projectObservationContext } from '../observation-context.js'
58
61
  import { applyLifecycleHookResults } from '../plugin-hooks.js'
@@ -67,7 +70,7 @@ import {
67
70
  markProviderRejectedImage,
68
71
  projectRequestRichContent,
69
72
  } from '../request-rich-content.js'
70
- import { formatSteeringNote, isOperatorUserMessage } from '../steering.js'
73
+ import { formatJobNote, formatSteeringNote, isOperatorUserMessage } from '../steering.js'
71
74
  import { parseNativeCandidate } from './native-output.js'
72
75
  import { runAdvisoryPhase } from './phases/advisory.js'
73
76
  import { runIterationCheckpoint } from './phases/checkpoint.js'
@@ -84,6 +87,20 @@ import { refreshWorkingMemory } from './phases/working-memory.js'
84
87
  import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js'
85
88
  import { streamProviderTurn } from './stream-turn.js'
86
89
 
90
+ type ReviewRequest = Pick<AnswerReviewContext, 'requestMessages' | 'latestUserMessage'>
91
+
92
+ /** A host reviewer is not the model transport, even when its cause is an HTTP failure. */
93
+ class AnswerReviewFailure extends NamzuError {
94
+ constructor(cause: unknown) {
95
+ super({
96
+ code: 'unknown',
97
+ message: `Answer review failed: ${toErrorMessage(cause)}`,
98
+ retryable: false,
99
+ details: { phase: 'answer-review' },
100
+ cause,
101
+ })
102
+ }
103
+ }
87
104
  export type { IterationContext } from './phases/index.js'
88
105
  export type { PhaseSignal } from './phases/index.js'
89
106
  export type { ToolReviewOutcome } from './phases/index.js'
@@ -98,6 +115,11 @@ export type { ToolReviewOutcome } from './phases/index.js'
98
115
  */
99
116
  const DEFAULT_ANSWER_REVIEW_LIMIT = 3
100
117
 
118
+ // Ending a run changes the available actions, not the strength of its evidence.
119
+ // Use the same standard for warning closure and empty-completion recovery.
120
+ const CLOSING_RESPONSE_GUIDANCE =
121
+ 'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.'
122
+
101
123
  /**
102
124
  * The share of a run's REMAINING time a settle-hold may take.
103
125
  *
@@ -160,8 +182,68 @@ export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
160
182
  )
161
183
  }
162
184
 
185
+ /**
186
+ * The ceiling on the job half of that grace, in milliseconds.
187
+ *
188
+ * `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
189
+ * only opens where it matters most: a run with no `timeoutMs` — the CLI's
190
+ * shipping default, `No run deadline by default` — has infinite time before
191
+ * it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
192
+ * delegated task that is sound, because the hour is the longest the task
193
+ * itself may live: the hold cannot outlast the work. A background job has no
194
+ * such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
195
+ * the same arithmetic parks an interactive session for an hour on a job that
196
+ * was never going to exit.
197
+ *
198
+ * So the job leg gets its own bound, and it is sized to what the wait buys
199
+ * rather than to how long a job may live: a turn in which to use the exit.
200
+ * A model that already waited its `wait_for_job` bound out and saw nothing is
201
+ * not usually two minutes from an exit, and the run ending is not the news
202
+ * being lost — with no run in flight the session announces the exit itself
203
+ * (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
204
+ * cheaper of the two places to hear it.
205
+ */
206
+ const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000
207
+
208
+ /**
209
+ * The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
210
+ *
211
+ * `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
212
+ * longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
213
+ * `wait_for_job`'s own bound — and it is the same parse, so a value that is
214
+ * not a positive whole number of milliseconds leaves the default standing
215
+ * rather than holding a run for `NaN`. Called here rather than at module
216
+ * load, because a host that sets it after import is not ignored.
217
+ */
218
+ export function awaitedJobGraceMs(remainingBeforeFinalizeMs: number): number {
219
+ const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS)
220
+ return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling)
221
+ }
222
+
163
223
  export class IterationOrchestrator {
164
224
  private ctx: IterationContext
225
+ private advisoryTurn:
226
+ | {
227
+ readonly iteration: number
228
+ readonly requestMessages: readonly Message[]
229
+ readonly response: Message
230
+ }
231
+ | undefined
232
+
233
+ /** Live only within its iteration; never joined by a guessed array offset. */
234
+ getAdvisoryTurnContext():
235
+ | import('../../../advisory/executor.js').AdvisoryTurnContext
236
+ | undefined {
237
+ const turn = this.advisoryTurn
238
+ if (!turn || turn.iteration !== this.ctx.runMgr.currentIteration) return undefined
239
+ const start = this.ctx.runMgr.messages.indexOf(turn.response)
240
+ if (start < 0) return undefined
241
+ return {
242
+ iteration: turn.iteration,
243
+ requestMessages: turn.requestMessages,
244
+ subsequentMessages: this.ctx.runMgr.messages.slice(start),
245
+ }
246
+ }
165
247
  /** Rejections so far. See {@link DEFAULT_ANSWER_REVIEW_LIMIT}. */
166
248
  private answerReviewAttempts = 0
167
249
  /**
@@ -206,6 +288,7 @@ export class IterationOrchestrator {
206
288
  },
207
289
  }
208
290
  ctx.checkpointMgr.setLatestUserMessageSource(() => this.latestUserMessage)
291
+ ctx.checkpointMgr.setAnswerReviewAttemptsSource?.(() => this.answerReviewAttempts)
209
292
  ctx.checkpointMgr.setStructuredReviewAttemptsSource?.(() => this.structuredReviewAttempts)
210
293
  ctx.checkpointMgr.setNativeStructuredAttemptsSource?.(() => this.nativeStructuredAttempts)
211
294
  if (ctx.structuredOutput?.mode === 'native') {
@@ -223,6 +306,11 @@ export class IterationOrchestrator {
223
306
  const maxReviews = ctx.structuredOutput?.maxReviews
224
307
  if (maxReviews !== undefined && (!Number.isSafeInteger(maxReviews) || maxReviews < 0))
225
308
  throw new RangeError('structuredOutput.maxReviews must be a nonnegative safe integer')
309
+ if (
310
+ ctx.maxAnswerReviews !== undefined &&
311
+ (!Number.isSafeInteger(ctx.maxAnswerReviews) || ctx.maxAnswerReviews < 0)
312
+ )
313
+ throw new RangeError('maxAnswerReviews must be a nonnegative safe integer')
226
314
  }
227
315
 
228
316
  /**
@@ -306,6 +394,7 @@ export class IterationOrchestrator {
306
394
  const tracer = getTracer()
307
395
  // Resume hydration happens after construction, before the loop starts.
308
396
  this.latestUserMessage = this.ctx.checkpointMgr.restoredLatestUserMessage
397
+ this.answerReviewAttempts = this.ctx.checkpointMgr.restoredAnswerReviewAttempts ?? 0
309
398
  this.structuredReviewAttempts = this.ctx.checkpointMgr.restoredStructuredReviewAttempts ?? 0
310
399
  this.nativeStructuredAttempts = this.ctx.checkpointMgr.restoredNativeStructuredAttempts ?? 0
311
400
  if (!this.latestUserMessage) {
@@ -353,6 +442,13 @@ export class IterationOrchestrator {
353
442
  runMgr.setStopReason('structured_output_failed')
354
443
  break
355
444
  }
445
+ if (
446
+ this.ctx.reviewAnswer &&
447
+ this.answerReviewAttempts > (this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT)
448
+ ) {
449
+ runMgr.setStopReason('answer_rejected')
450
+ break
451
+ }
356
452
  if (
357
453
  this.ctx.structuredOutput?.review &&
358
454
  this.structuredReviewAttempts >
@@ -533,14 +629,15 @@ export class IterationOrchestrator {
533
629
  // Snapshot the cumulative counters so the step can report ITS
534
630
  // own usage rather than the run total.
535
631
  stepStartedAt = Date.now()
536
- usageBefore = { ...runMgr.tokenUsage }
537
- costBefore = { ...runMgr.costInfo }
538
632
 
539
633
  // Shape this step before calling the model. `stopWhen` decides
540
634
  // whether to keep going; this decides HOW. No-op when the host
541
635
  // supplied no hook.
542
636
  const contextModelBeforePreparation = this.ctx.contextModel ?? model
543
637
  const step = await this.prepareStep(iterationNum)
638
+ // Preparation inference belongs to the run, not the main-model step.
639
+ usageBefore = { ...runMgr.tokenUsage }
640
+ costBefore = { ...runMgr.costInfo }
544
641
  stepModel = step.model ?? model
545
642
  await this.selectContextModel(stepModel)
546
643
  // Preserve post-compaction preparation/recall semantics. A changed
@@ -570,7 +667,7 @@ export class IterationOrchestrator {
570
667
  ? [
571
668
  ...runMgr.messages,
572
669
  createRuntimeContextMessage(
573
- '[SYSTEM] You are approaching your resource limits. Provide your final, comprehensive response now based on everything you have gathered so far. Do not request any more tool calls.',
670
+ `[SYSTEM] You are approaching your resource limits. ${CLOSING_RESPONSE_GUIDANCE}`,
574
671
  'limit-finalization',
575
672
  ),
576
673
  ]
@@ -589,9 +686,9 @@ export class IterationOrchestrator {
589
686
  // mutation, and per-iteration this is trivial next to the model
590
687
  // call it precedes.
591
688
  // A step's skills and its guidance ride the same ephemeral
592
- // trailing system message. Appending leaves the cached prefix
593
- // intact; rewriting the run's own prompt to carry a phase's
594
- // skills would invalidate it on every iteration.
689
+ // system message. A driver may move it before history; changing
690
+ // system guidance can therefore affect prefix caching. Observations
691
+ // that need no system authority use step.context below.
595
692
  // `renderSkillsSection` already answers null for an empty list, so
596
693
  // there is no length check here — a second guard for the same
597
694
  // case is one more thing to keep in agreement with the first.
@@ -612,10 +709,9 @@ export class IterationOrchestrator {
612
709
  ? `Approval policy changed from "${policyChange.from}" to "${policyChange.to}" (${policyChange.reason}). Tool calls from here on are reviewed under the new policy.`
613
710
  : null
614
711
  // State that changed during the run, reported once per turn.
615
- // `turn` contributions land HERE and nowhere else: in the
616
- // system prompt they would be cached for the run or read as
617
- // a standing instruction, and either way the state they
618
- // exist to report goes stale silently.
712
+ // `turn` contributions are recomputed here, not fixed when the
713
+ // run's prompt is assembled. They retain system authority and
714
+ // may affect caching just like the other system contributions.
619
715
  const turnSections =
620
716
  this.ctx.promptContributions?.render('turn', {
621
717
  iteration: iterationNum,
@@ -627,10 +723,12 @@ export class IterationOrchestrator {
627
723
  const requestHistory = stepPreamble
628
724
  ? [...baseMessages, createSystemMessage(stepPreamble)]
629
725
  : [...baseMessages]
726
+ if (step.context) requestHistory.push(this.stepContextMessage(step.context))
630
727
  const messages = projectRequestRichContent(
631
728
  this.projectObservations(requestHistory),
632
729
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
633
730
  )
731
+ this.appendWorkContext(messages, iterationNum, step)
634
732
  await this.reportUnsupportedToolResults(messages)
635
733
  yield* this.ctx.drainPending()
636
734
 
@@ -741,7 +839,12 @@ export class IterationOrchestrator {
741
839
  model: requestedMember.model ?? stepModel,
742
840
  chainIndex: requestedMember.index,
743
841
  }
744
- const { response, messageId } = yield* streamProviderTurn(
842
+ const operatorInputAtDispatch = this.latestUserMessage
843
+ const latestReviewUserMessage =
844
+ (this.ctx.reviewAnswer || this.ctx.structuredOutput?.review) && operatorInputAtDispatch
845
+ ? structuredClone(operatorInputAtDispatch)
846
+ : undefined
847
+ const { response, messageId, requestMessages } = yield* streamProviderTurn(
745
848
  this.ctx.provider,
746
849
  {
747
850
  model: stepModel,
@@ -794,8 +897,15 @@ export class IterationOrchestrator {
794
897
  {
795
898
  onAccepted: (identity) => this.acceptProviderRejectedImage(identity),
796
899
  },
900
+ Boolean(
901
+ this.ctx.reviewAnswer || this.ctx.structuredOutput?.review || this.ctx.advisoryCtx,
902
+ ),
797
903
  )
798
904
  stepResponse = response
905
+ const reviewRequest: ReviewRequest = {
906
+ ...(requestMessages ? { requestMessages } : {}),
907
+ ...(latestReviewUserMessage ? { latestUserMessage: latestReviewUserMessage } : {}),
908
+ }
799
909
 
800
910
  // Who answered THIS turn.
801
911
  //
@@ -804,18 +914,9 @@ export class IterationOrchestrator {
804
914
  // request, so the member at the cursor when the stream ends is
805
915
  // the one whose bytes are in `response`.
806
916
  //
807
- // It is taken here rather than at `recordStep` several hundred
808
- // lines below, and the honest account of that is defence in
809
- // depth, not a defect it currently prevents. Moving it down
810
- // fails no test, because nothing between the two asks this
811
- // provider for anything: compaction and working memory run
812
- // BEFORE the turn, the advisory phase runs after the step is
813
- // already recorded, and the only thing in between is tool
814
- // execution. That is a fact about today's phase order, which a
815
- // later phase inserted here would change silently — and the
816
- // symptom would be a step attributed to a member that first
817
- // served the turn after it, which is the class of wrongness
818
- // this whole field exists to end.
917
+ // Capture before host review: its auxiliary inference can move
918
+ // the fallback cursor. Main-step usage and provenance must keep
919
+ // naming the provider that produced this candidate.
819
920
  const servedBy: StepProvenance = ((): StepProvenance => {
820
921
  const member = this.ctx.servingMember?.() ?? {
821
922
  index: 0,
@@ -935,8 +1036,12 @@ export class IterationOrchestrator {
935
1036
  ? { replayState: response.message.replayState }
936
1037
  : {}),
937
1038
  },
1039
+ response.message.textParts,
938
1040
  )
939
1041
  runMgr.pushMessage(assistantMsg)
1042
+ if (this.ctx.advisoryCtx && requestMessages) {
1043
+ this.advisoryTurn = { iteration: iterationNum, requestMessages, response: assistantMsg }
1044
+ }
940
1045
 
941
1046
  if (this.ctx.workingStateManager && this.ctx.compactionConfig && assistantMsg.content) {
942
1047
  extractFromAssistantMessage(
@@ -1061,7 +1166,12 @@ export class IterationOrchestrator {
1061
1166
  this.ctx.abortController.signal,
1062
1167
  )
1063
1168
  let outcome: 'accepted' | 'retry' | 'exhausted' | 'cancelled'
1064
- if (candidate.success) outcome = await this.reviewStructuredOutput(candidate.value)
1169
+ if (candidate.success)
1170
+ outcome = await this.reviewStructuredOutput(
1171
+ candidate.value,
1172
+ reviewRequest,
1173
+ stepModel,
1174
+ )
1065
1175
  else {
1066
1176
  this.nativeStructuredAttempts++
1067
1177
  runMgr.pushMessage(
@@ -1207,7 +1317,11 @@ export class IterationOrchestrator {
1207
1317
  // judge: bounded attempts, feedback as a user message, and
1208
1318
  // a loud stop rather than a loop.
1209
1319
  if (!forceFinalize && this.ctx.reviewAnswer) {
1210
- const review = await this.reviewAnswer(response.message.content ?? '')
1320
+ const review = await this.reviewAnswer(
1321
+ response.message.content ?? '',
1322
+ reviewRequest,
1323
+ stepModel,
1324
+ )
1211
1325
  if (this.ctx.abortController.signal.aborted) {
1212
1326
  runMgr.setStopReason('cancelled')
1213
1327
  runMgr.markCancelled()
@@ -1215,6 +1329,21 @@ export class IterationOrchestrator {
1215
1329
  }
1216
1330
  if (review && !review.accept) {
1217
1331
  const attempt = ++this.answerReviewAttempts
1332
+ runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'))
1333
+ // Commit the consumed allowance with its feedback before another
1334
+ // request, including exhaustion. Compaction cannot reset this quota.
1335
+ const checkpoint = await this.ctx.checkpointMgr.create(runMgr, iterationNum)
1336
+ await this.ctx.emitEvent({
1337
+ type: 'checkpoint_created',
1338
+ runId: runMgr.id,
1339
+ checkpointId: checkpoint.id,
1340
+ iteration: iterationNum,
1341
+ })
1342
+ if (this.ctx.abortController.signal.aborted) {
1343
+ runMgr.setStopReason('cancelled')
1344
+ runMgr.markCancelled()
1345
+ break
1346
+ }
1218
1347
  const limit = this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT
1219
1348
  if (attempt > limit) {
1220
1349
  this.ctx.log.warn('Answer rejected more times than the run allows', {
@@ -1230,7 +1359,6 @@ export class IterationOrchestrator {
1230
1359
  'namzu.retry.attempt': attempt,
1231
1360
  'namzu.runtime.limit': limit,
1232
1361
  })
1233
- runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'))
1234
1362
  await this.ctx.emitEvent({
1235
1363
  type: 'iteration_completed',
1236
1364
  runId: runMgr.id,
@@ -1272,7 +1400,12 @@ export class IterationOrchestrator {
1272
1400
  continue
1273
1401
  }
1274
1402
 
1275
- let closingStopReason: StopReason | undefined
1403
+ // A limit-requested summary bypasses prose review and further
1404
+ // work. Preserve that limit on settlement, even if the provider
1405
+ // reports a normal text completion and headroom still remains.
1406
+ let closingStopReason: StopReason | undefined = forceFinalize
1407
+ ? guardResult.stopReason
1408
+ : undefined
1276
1409
  if (!hasContent && !forceFinalize) {
1277
1410
  this.ctx.log.warn('Empty completion detected — requesting final summary', {
1278
1411
  [NAMZU.ITERATION]: iterationNum,
@@ -1353,6 +1486,8 @@ export class IterationOrchestrator {
1353
1486
  const structuredOutcome = await this.captureStructuredOutput(
1354
1487
  reviewOutcome.results,
1355
1488
  response,
1489
+ reviewRequest,
1490
+ stepModel,
1356
1491
  )
1357
1492
  if (
1358
1493
  structuredOutcome === 'retry' ||
@@ -1379,10 +1514,6 @@ export class IterationOrchestrator {
1379
1514
  break
1380
1515
  }
1381
1516
  if (structuredOutcome === 'accepted') {
1382
- this.ctx.log.info('Structured output produced — ending run', {
1383
- [NAMZU.RUN_ID]: runMgr.id,
1384
- [NAMZU.ITERATION]: iterationNum,
1385
- })
1386
1517
  await this.ctx.emitEvent({
1387
1518
  type: 'iteration_completed',
1388
1519
  runId: runMgr.id,
@@ -1395,6 +1526,21 @@ export class IterationOrchestrator {
1395
1526
  runMgr.markCancelled()
1396
1527
  break
1397
1528
  }
1529
+ if (!forceFinalize) {
1530
+ const inbound = this.deliverInbound()
1531
+ // Tool-result steering may already have been delivered by
1532
+ // runToolReview. Its candidate still answers the older input.
1533
+ if (inbound > 0 || this.latestUserMessage !== operatorInputAtDispatch) continue
1534
+ }
1535
+ if (this.ctx.abortController.signal.aborted) {
1536
+ runMgr.setStopReason('cancelled')
1537
+ runMgr.markCancelled()
1538
+ break
1539
+ }
1540
+ this.ctx.log.info('Structured output produced — ending run', {
1541
+ [NAMZU.RUN_ID]: runMgr.id,
1542
+ [NAMZU.ITERATION]: iterationNum,
1543
+ })
1398
1544
  this.publishStructuredOutput()
1399
1545
  runMgr.setStopReason('end_turn')
1400
1546
  break
@@ -1430,8 +1576,10 @@ export class IterationOrchestrator {
1430
1576
  // returned — which is what makes a terminal submit_answer tool
1431
1577
  // usable without discarding its output.
1432
1578
  if (await this.shouldStop()) {
1433
- // Outstanding delegated work outranks the host's stop
1434
- // predicate, exactly once.
1579
+ // Outstanding work outranks the host's stop predicate —
1580
+ // a delegated task the completion inbox is expecting, or
1581
+ // a background job the model told `wait_for_job` it is
1582
+ // waiting on.
1435
1583
  //
1436
1584
  // This is a precedence rule chosen here, not something
1437
1585
  // `stopWhen` implies — a stop predicate is a programmable
@@ -1440,13 +1588,20 @@ export class IterationOrchestrator {
1440
1588
  // tool or a captured structured output. Those decide the
1441
1589
  // result, so no turn follows and a hold would buy nothing.
1442
1590
  // This one only says "stop", and stopping one turn later
1443
- // with the worker's result in hand is a better reading of
1444
- // the host's intent than stopping now and discarding it.
1591
+ // with the result in hand is a better reading of the
1592
+ // host's intent than stopping now and discarding it.
1445
1593
  //
1446
- // Bounded: after the notification is delivered the inbox
1447
- // is drained, so the predicate fires again next turn with
1448
- // nothing pending and the run stops. Exactly one extra
1449
- // turn, and `maxIterations` bounds it regardless.
1594
+ // Bounded by what is left to deliver, not by a count.
1595
+ // Each delivery consumes what it delivered — the inbox is
1596
+ // drained, and a job exit's notice is taken with the
1597
+ // record of the exits it accounts for — so the predicate
1598
+ // is asked again next turn against whatever is still
1599
+ // outstanding. One task deferred it once; two awaited
1600
+ // jobs exiting a minute apart defer it twice, each time
1601
+ // for a turn the model spends on news it has not read.
1602
+ // `maxIterations` and the run's own deadline bound all of
1603
+ // it regardless, and a leg with nothing pending never
1604
+ // opens a hold at all.
1450
1605
  if (yield* this.holdForOutstandingWork(iterationNum, true)) {
1451
1606
  // Remember WHY the next turn exists, so the turn that
1452
1607
  // ends the run can name the host's decision instead of
@@ -1512,7 +1667,7 @@ export class IterationOrchestrator {
1512
1667
  // in the history the next request is built from.
1513
1668
  this.deliverInbound()
1514
1669
 
1515
- await runAdvisoryPhase(this.ctx, iterationNum, response)
1670
+ await runAdvisoryPhase(this.ctx, iterationNum, response, this.getAdvisoryTurnContext())
1516
1671
 
1517
1672
  if (this.ctx.pluginManager) {
1518
1673
  const hookResults = await this.ctx.pluginManager.executeHooks(
@@ -1633,6 +1788,7 @@ export class IterationOrchestrator {
1633
1788
  // would burn the budget to arrive at the same error.
1634
1789
  if (
1635
1790
  !overflowRelieved &&
1791
+ !(err instanceof AnswerReviewFailure) &&
1636
1792
  classifyProviderError(err, this.ctx.provider.id).code === 'context_length_exceeded'
1637
1793
  ) {
1638
1794
  overflowRelieved = true
@@ -1660,6 +1816,7 @@ export class IterationOrchestrator {
1660
1816
  iterSpan.recordException(err instanceof Error ? err : new Error(String(err)))
1661
1817
  throw err
1662
1818
  } finally {
1819
+ this.advisoryTurn = undefined
1663
1820
  // The only place the iteration span ends. It used to be ended at each of
1664
1821
  // seventeen exits, which is a rule every future edit has to
1665
1822
  // remember; a generator abandoned by its consumer never reached
@@ -1673,30 +1830,63 @@ export class IterationOrchestrator {
1673
1830
  }
1674
1831
 
1675
1832
  /**
1676
- * Hold the run open for a worker that has not finished, and deliver it.
1833
+ * Hold the run open for work that has not finished, and deliver it.
1677
1834
  *
1678
- * Returns whether a completion or operator message entered the transcript —
1679
- * the caller continues on `true`, so the model gets a turn to respond.
1680
- * That turn is the entire justification for waiting, which
1835
+ * Returns whether a completion, a job exit or an operator message entered
1836
+ * the transcript — the caller continues on `true`, so the model gets a turn
1837
+ * to respond. That turn is the entire justification for waiting, which
1681
1838
  * is why only the exits that can still take one call this.
1682
1839
  *
1683
- * Bounded by `settleGraceMs` and by `maxIterations`, so a worker that never
1684
- * finishes cannot keep the run open.
1840
+ * Two kinds of work qualify and they are raced together, because a run has
1841
+ * one settle point and one grace period to spend at it:
1842
+ *
1843
+ * - a delegated task the `CompletionInbox` is still expecting;
1844
+ * - a background job the model told `wait_for_job` it is waiting on.
1845
+ *
1846
+ * The job half is deliberately narrow. Intent comes from the wait and from
1847
+ * nothing else — a dev server the model started and never waited on is
1848
+ * running because somebody wanted it running, and a hold for it would add
1849
+ * the grace period to the end of every turn for the rest of the session.
1850
+ *
1851
+ * Each leg is opened only when it has something pending: both
1852
+ * `waitForArrival` implementations resolve immediately when their own side
1853
+ * is idle, so racing an idle one would end the hold before it began.
1854
+ *
1855
+ * Bounded by `settleGraceMs` and by `maxIterations`, so work that never
1856
+ * finishes cannot keep the run open. On a run with a deadline the grace is
1857
+ * a share of what is LEFT of it rather than a fresh allowance, so a
1858
+ * `wait_for_job` call that already spent minutes has shortened this hold
1859
+ * by the same minutes. On a run without one — the CLI's default — there is
1860
+ * no remainder to take a share of, and the job leg's own ceiling
1861
+ * (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
1862
+ * by an hour of silence.
1685
1863
  */
1686
1864
  private async *holdForOutstandingWork(
1687
1865
  iterationNum: number,
1688
1866
  hasToolCalls: boolean,
1689
1867
  ): AsyncGenerator<RunEvent, boolean> {
1690
- if (!this.ctx.completionInbox?.hasPendingWork) return false
1868
+ const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined
1869
+ const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined
1870
+ if (!inbox && !jobs) return false
1691
1871
 
1692
1872
  // Read HERE rather than from `forceFinalize`, which was sampled at the
1693
1873
  // top of the iteration: one that has since crossed the finalize point
1694
1874
  // must not open a wait against a reserve it has already entered.
1695
- const graceMs = settleGraceMs(this.ctx.guard.remainingBeforeFinalizeMs())
1696
- this.ctx.log.info('Holding the run open for a background task', {
1875
+ const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs()
1876
+ // One deadline for the race, and it is the LONGEST ceiling any pending
1877
+ // leg justifies. A leg resolving on its own timer ends the whole race,
1878
+ // so handing the job leg its shorter ceiling while a task was also
1879
+ // outstanding would cut the task's hold down to the job's — a run
1880
+ // walking away from a worker it had time for, because a job happened
1881
+ // to be running. A job therefore never shortens a wait, and it never
1882
+ // lengthens one either: where a task is outstanding too, that is how
1883
+ // long this run was waiting anyway.
1884
+ const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs)
1885
+ this.ctx.log.info('Holding the run open for outstanding work', {
1697
1886
  [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1698
1887
  [NAMZU.ITERATION]: iterationNum,
1699
1888
  'namzu.runtime.grace_ms': graceMs,
1889
+ 'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
1700
1890
  })
1701
1891
  // User input releases this wait without cancelling any child. Both waits
1702
1892
  // share a disposable signal so the losing arrival listener cannot leak.
@@ -1707,7 +1897,8 @@ export class IterationOrchestrator {
1707
1897
  if (runSignal.aborted) cancelWait()
1708
1898
  try {
1709
1899
  await Promise.race([
1710
- this.ctx.completionInbox.waitForArrival(graceMs, waiting.signal),
1900
+ ...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
1901
+ ...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
1711
1902
  ...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
1712
1903
  ])
1713
1904
  } catch (error) {
@@ -1718,14 +1909,15 @@ export class IterationOrchestrator {
1718
1909
  }
1719
1910
  runSignal.throwIfAborted()
1720
1911
 
1721
- const arrived = this.ctx.completionInbox.drain()
1912
+ const arrived = this.ctx.completionInbox?.drain() ?? []
1722
1913
  if (arrived.length > 0) {
1723
1914
  this.ctx.runMgr.pushMessage(
1724
1915
  createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'),
1725
1916
  )
1726
1917
  }
1918
+ const exited = this.deliverAwaitedJobExits()
1727
1919
  const inbound = this.deliverInbound()
1728
- if (arrived.length === 0 && inbound === 0) return false
1920
+ if (arrived.length === 0 && !exited && inbound === 0) return false
1729
1921
  await this.ctx.emitEvent({
1730
1922
  type: 'iteration_completed',
1731
1923
  runId: this.ctx.runMgr.id,
@@ -1737,8 +1929,48 @@ export class IterationOrchestrator {
1737
1929
  }
1738
1930
 
1739
1931
  /**
1740
- * Account for delegated work on the way out: deliver what arrived, and say
1741
- * what did not.
1932
+ * Put the job exits this hold was waiting for in front of the model.
1933
+ *
1934
+ * Through `jobNotices`, which is the channel a job exit already travels on
1935
+ * — `attachNotice` rides it out on the next tool result — rather than a
1936
+ * second one built for this path. A turn that called no tools has no such
1937
+ * result, so the queued text becomes a `runtime-context` message instead,
1938
+ * exactly as `deliverInbound` does for steering that found no tool result
1939
+ * to attach to.
1940
+ *
1941
+ * That drain is also what keeps one exit from being delivered twice: the
1942
+ * channel hands its text over once, so an exit already attached to a tool
1943
+ * result earlier in the turn leaves nothing here — and the record of it
1944
+ * went with that delivery, so this returns `false` rather than buying a
1945
+ * turn to re-read what the model has read.
1946
+ *
1947
+ * `takeDelivery` is what pairs the two. Taking the exits first and then
1948
+ * finding no notice would discard them, which is the one way this path
1949
+ * can lose an exit outright; neither is taken unless both are there.
1950
+ *
1951
+ * The channel is not per-job, so the text taken here can include a notice
1952
+ * for a job nobody awaited that ended while the hold was open. Delivering
1953
+ * it is right — it is unread either way, and the alternative is stranding
1954
+ * it — but it is not a reason to WAIT, which is why what opens this hold
1955
+ * is `AwaitedJobs`, and the two are asked separately.
1956
+ */
1957
+ private deliverAwaitedJobExits(): boolean {
1958
+ const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
1959
+ if (!delivered) return false
1960
+
1961
+ this.ctx.log.info('Delivering a background job exit the run held open for', {
1962
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1963
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1964
+ })
1965
+ this.ctx.runMgr.pushMessage(
1966
+ createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
1967
+ )
1968
+ return true
1969
+ }
1970
+
1971
+ /**
1972
+ * Account for outstanding work on the way out: deliver what arrived, and
1973
+ * say what did not.
1742
1974
  *
1743
1975
  * A run that ends with a worker outstanding must not leave the impression
1744
1976
  * that the worker's result was delivered. There are exactly two honest
@@ -1762,19 +1994,33 @@ export class IterationOrchestrator {
1762
1994
  */
1763
1995
  private settleOutstandingWork(): void {
1764
1996
  this.deliverArrivedCompletions()
1997
+ this.deliverArrivedJobExits()
1765
1998
  this.recordAbandonedWork()
1766
1999
  }
1767
2000
 
1768
- /** Delegated work this run walked away from. See {@link settleOutstandingWork}. */
2001
+ /** Work this run walked away from. See {@link settleOutstandingWork}. */
1769
2002
  private recordAbandonedWork(): void {
1770
2003
  const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? []
1771
- if (abandoned.length === 0) return
2004
+ if (abandoned.length > 0) {
2005
+ this.ctx.log.warn('Run ended with delegated work still running', {
2006
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2007
+ 'namzu.runtime.tasks': abandoned,
2008
+ })
2009
+ this.ctx.runMgr.setAbandonedTaskIds(abandoned)
2010
+ }
1772
2011
 
1773
- this.ctx.log.warn('Run ended with delegated work still running', {
2012
+ // The same statement for a job the model was waiting on when the grace
2013
+ // ran out. Only awaited ones: a job nobody waited for was never work
2014
+ // this run was holding, so naming it would report an abandonment that
2015
+ // did not happen.
2016
+ const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? []
2017
+ if (abandonedJobs.length === 0) return
2018
+
2019
+ this.ctx.log.warn('Run ended with an awaited background job still running', {
1774
2020
  [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1775
- 'namzu.runtime.tasks': abandoned,
2021
+ 'namzu.runtime.jobs': abandonedJobs,
1776
2022
  })
1777
- this.ctx.runMgr.setAbandonedTaskIds(abandoned)
2023
+ this.ctx.runMgr.setAbandonedJobIds(abandonedJobs)
1778
2024
  }
1779
2025
 
1780
2026
  private deliverArrivedCompletions(): void {
@@ -1810,24 +2056,73 @@ export class IterationOrchestrator {
1810
2056
  }
1811
2057
 
1812
2058
  /**
1813
- * Ask the host how to shape this step.
2059
+ * The job half of {@link deliverArrivedCompletions}: an exit that arrived
2060
+ * too late to earn a turn is still delivered on the way out.
1814
2061
  *
1815
- * Fails OPEN on a throw — same reasoning as `stopWhen` and deliberately
1816
- * opposite to a guardrail: a broken step-shaping hook should not kill an
1817
- * otherwise healthy run, and unlike a safety check, nothing unsafe gets
1818
- * through when it is skipped.
1819
- */
1820
- /**
1821
- * A host's chance to refuse the next model call.
2062
+ * The window this closes is one tick wide and it is nobody else's. An
2063
+ * awaited job that exits between the hold's grace expiring and the run
2064
+ * settling was never delivered — the hold had already looked — and is no
2065
+ * longer named either, because the exit took it off the outstanding list
2066
+ * on its way past, so `abandonedJobIds` would be lying to claim it. The
2067
+ * host's own listener is no help: the CLI queues an exit for the next
2068
+ * turn only when no run is in flight, and this one is still in flight.
2069
+ * Delivered here it reaches `Run.messages`, so the transcript has it and
2070
+ * a continued thread opens with it.
1822
2071
  *
1823
- * Fails CLOSED, which is the opposite of `prepareStep` below and the
1824
- * reason they are separate hooks rather than one with two return
1825
- * shapes. A broken step-SHAPER skipped costs a run its per-step tuning;
1826
- * a broken step-REFUSER skipped is a refusal that did not happen, which
1827
- * is precisely what the hook exists to prevent. The thrown error's
1828
- * message becomes the reason, so an operator is not left with a run
1829
- * that stopped and no account of it.
2072
+ * Before `recordAbandonedWork`, which then reports only what is still
2073
+ * running, and after `deliverArrivedCompletions`, so the two appended
2074
+ * messages land in the order the work finished in.
1830
2075
  */
2076
+ private deliverArrivedJobExits(): void {
2077
+ const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
2078
+ if (!delivered) return
2079
+
2080
+ // Fix the run's answer BEFORE appending anything after it — the same
2081
+ // `resolveResult` tail walk `deliverArrivedCompletions` explains just
2082
+ // above, and the same guard against pinning an empty one.
2083
+ const answer = this.ctx.runMgr.materializeResult()
2084
+ if (answer.length > 0) this.ctx.runMgr.setResult(answer)
2085
+
2086
+ this.ctx.log.info('Delivering a background job exit the run would have settled over', {
2087
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2088
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
2089
+ })
2090
+ this.ctx.runMgr.pushMessage(
2091
+ createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
2092
+ )
2093
+ }
2094
+
2095
+ private stepContextMessage(content: string) {
2096
+ return createRuntimeContextMessage(
2097
+ `Current step context (runtime-generated; not a new user request):\n${content}`,
2098
+ 'step-context',
2099
+ )
2100
+ }
2101
+
2102
+ /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
2103
+ private appendWorkContext(
2104
+ messages: Message[],
2105
+ stepNumber: number,
2106
+ prepared: PrepareStepResult,
2107
+ ): void {
2108
+ const contributions = [
2109
+ this.ctx.completionInbox?.describeOwnedWork(),
2110
+ this.ctx.toolExecutor.describeFileEvidence(messages),
2111
+ ].filter((content): content is string => Boolean(content))
2112
+ if (contributions.length === 0) return
2113
+ let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0
2114
+ // Leave room for the actual task; admit whole contributions, never dangling partial references.
2115
+ if (room < 1_500) return
2116
+ for (const content of contributions) {
2117
+ if (!content || content.length > 8_000) continue
2118
+ const message = this.stepContextMessage(content)
2119
+ const tokens = estimateMessageTokens(message)
2120
+ if (tokens > Math.min(2_000, room - 1_000)) continue
2121
+ messages.push(message)
2122
+ room -= tokens
2123
+ }
2124
+ }
2125
+
1831
2126
  private stepContext(stepNumber: number, prepared: PrepareStepResult): PrepareStepContext {
1832
2127
  const model = prepared.model ?? this.ctx.runConfig.model
1833
2128
  const window = resolveContextWindow(
@@ -1841,7 +2136,9 @@ export class IterationOrchestrator {
1841
2136
  )
1842
2137
  const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null
1843
2138
  const preamble = [prepared.system, skills].filter(Boolean).join('\n\n')
1844
- const preparedTokens = preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0
2139
+ const preparedTokens =
2140
+ (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
2141
+ (prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0)
1845
2142
  const responseReserve = Math.min(
1846
2143
  prepared.maxResponseTokens ??
1847
2144
  this.ctx.runConfig.maxResponseTokens ??
@@ -1852,6 +2149,7 @@ export class IterationOrchestrator {
1852
2149
  runId: this.ctx.runMgr.id,
1853
2150
  stepNumber,
1854
2151
  messages: this.ctx.runMgr.messages,
2152
+ ...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
1855
2153
  ...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
1856
2154
  signal: this.ctx.abortController.signal,
1857
2155
  contextBudget: {
@@ -1868,6 +2166,7 @@ export class IterationOrchestrator {
1868
2166
  }
1869
2167
  }
1870
2168
 
2169
+ /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
1871
2170
  private async beforeStep(stepNumber: number): Promise<StepVeto | undefined> {
1872
2171
  const configured = this.ctx.beforeStep
1873
2172
  if (!configured) return undefined
@@ -1878,11 +2177,13 @@ export class IterationOrchestrator {
1878
2177
  }
1879
2178
  }
1880
2179
 
2180
+ /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
1881
2181
  private async prepareStep(stepNumber: number): Promise<{
1882
2182
  allowedTools?: string[]
1883
2183
  toolChoice?: ToolChoice
1884
2184
  model?: string
1885
2185
  system?: string
2186
+ context?: string
1886
2187
  skills?: readonly Skill[]
1887
2188
  temperature?: number
1888
2189
  maxResponseTokens?: number
@@ -1897,8 +2198,16 @@ export class IterationOrchestrator {
1897
2198
  // rather than an accident of install history.
1898
2199
  let result: PrepareStepResult = {}
1899
2200
  for (const stage of stages) {
2201
+ const inference = createCallbackInference(
2202
+ this.ctx,
2203
+ result.model ?? this.ctx.runConfig.model,
2204
+ 'preparation',
2205
+ )
1900
2206
  try {
1901
- const decided = await stage(this.stepContext(stepNumber, result))
2207
+ const decided = await stage({
2208
+ ...this.stepContext(stepNumber, result),
2209
+ generateText: inference.generateText,
2210
+ })
1902
2211
  if (decided) result = { ...result, ...decided }
1903
2212
  await this.selectContextModel(result.model ?? this.ctx.runConfig.model)
1904
2213
  } catch (err) {
@@ -1909,6 +2218,23 @@ export class IterationOrchestrator {
1909
2218
  'namzu.runtime.step_number': stepNumber,
1910
2219
  'exception.message': toErrorMessage(err),
1911
2220
  })
2221
+ // An SDK stage may report availability and validated fallback evidence
2222
+ // without exposing its error. Preserve prior decisions and the context budget;
2223
+ // ordinary exceptions still contribute nothing to the model request.
2224
+ if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
2225
+ const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0
2226
+ if (
2227
+ typeof err.context === 'string' &&
2228
+ err.context.length > 0 &&
2229
+ err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room))
2230
+ )
2231
+ result = {
2232
+ ...result,
2233
+ context: [result.context, err.context].filter(Boolean).join('\n\n'),
2234
+ }
2235
+ }
2236
+ } finally {
2237
+ inference.close()
1912
2238
  }
1913
2239
  }
1914
2240
 
@@ -1917,6 +2243,7 @@ export class IterationOrchestrator {
1917
2243
  toolChoice?: ToolChoice
1918
2244
  model?: string
1919
2245
  system?: string
2246
+ context?: string
1920
2247
  skills?: readonly Skill[]
1921
2248
  temperature?: number
1922
2249
  maxResponseTokens?: number
@@ -1953,6 +2280,7 @@ export class IterationOrchestrator {
1953
2280
  if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
1954
2281
  if (result.model !== undefined) prepared.model = result.model
1955
2282
  if (result.system !== undefined) prepared.system = result.system
2283
+ if (result.context !== undefined) prepared.context = result.context
1956
2284
  if (result.skills !== undefined) prepared.skills = result.skills
1957
2285
  if (result.temperature !== undefined) prepared.temperature = result.temperature
1958
2286
  if (result.maxResponseTokens !== undefined) {
@@ -2237,6 +2565,8 @@ export class IterationOrchestrator {
2237
2565
  private async captureStructuredOutput(
2238
2566
  results: readonly ToolCallOutcome[],
2239
2567
  response: ChatCompletionResponse,
2568
+ reviewRequest: ReviewRequest,
2569
+ model: string,
2240
2570
  ): Promise<'absent' | 'accepted' | 'retry' | 'exhausted' | 'cancelled'> {
2241
2571
  if (!this.needsStructuredOutput() || this.ctx.structuredOutput?.mode === 'native')
2242
2572
  return 'absent'
@@ -2264,7 +2594,7 @@ export class IterationOrchestrator {
2264
2594
  )
2265
2595
  parsed = hit.output
2266
2596
  }
2267
- return this.reviewStructuredOutput(parsed)
2597
+ return this.reviewStructuredOutput(parsed, reviewRequest, model)
2268
2598
  }
2269
2599
 
2270
2600
  private publishStructuredOutput(): void {
@@ -2274,6 +2604,8 @@ export class IterationOrchestrator {
2274
2604
 
2275
2605
  private async reviewStructuredOutput(
2276
2606
  parsed: unknown,
2607
+ reviewRequest: ReviewRequest,
2608
+ model: string,
2277
2609
  ): Promise<'accepted' | 'retry' | 'exhausted' | 'cancelled'> {
2278
2610
  if (this.ctx.abortController.signal.aborted) return 'cancelled'
2279
2611
  const reviewer = this.ctx.structuredOutput?.review
@@ -2286,22 +2618,27 @@ export class IterationOrchestrator {
2286
2618
  signal.addEventListener('abort', onAbort, { once: true })
2287
2619
  })
2288
2620
  let verdict: AnswerReview
2621
+ const inference = createCallbackInference(this.ctx, model, 'review')
2289
2622
  try {
2290
2623
  verdict = await Promise.race([
2291
- Promise.resolve().then(() =>
2292
- reviewer(structuredClone(parsed), {
2624
+ Promise.resolve().then(() => {
2625
+ signal.throwIfAborted()
2626
+ return reviewer(structuredClone(parsed), {
2293
2627
  runId: this.ctx.runMgr.id,
2294
2628
  iteration: this.ctx.runMgr.currentIteration,
2295
2629
  signal,
2296
2630
  messages: this.ctx.runMgr.messages,
2297
- }),
2298
- ),
2631
+ ...reviewRequest,
2632
+ generateText: inference.generateText,
2633
+ })
2634
+ }),
2299
2635
  aborted,
2300
2636
  ])
2301
2637
  } catch (error) {
2302
2638
  if (signal.aborted) return 'cancelled'
2303
2639
  throw error
2304
2640
  } finally {
2641
+ inference.close()
2305
2642
  signal.removeEventListener('abort', onAbort)
2306
2643
  }
2307
2644
  if (signal.aborted) return 'cancelled'
@@ -2335,33 +2672,48 @@ export class IterationOrchestrator {
2335
2672
  return 'accepted'
2336
2673
  }
2337
2674
 
2338
- /**
2339
- * Ask the host whether this answer is good enough.
2340
- *
2341
- * A hook that throws **accepts**, which is the opposite of what the
2342
- * safety gates do, and deliberately so. Those are asked "is this
2343
- * dangerous", where the cost of failing closed is one refused
2344
- * operation. This is asked "is this good enough", where failing closed
2345
- * means handing the answer back forever — so a broken judge would turn
2346
- * every run into a loop that ends on a budget error naming nothing. One
2347
- * unreviewed answer is the cheaper failure, and the throw is logged at
2348
- * `error` so it is not mistaken for approval.
2349
- */
2350
- private async reviewAnswer(answer: string): Promise<AnswerReview | undefined> {
2351
- if (!this.ctx.reviewAnswer) return undefined
2675
+ /** A reviewer failure aborts settlement; only an explicit rejection requests correction. */
2676
+ private async reviewAnswer(
2677
+ answer: string,
2678
+ reviewRequest: ReviewRequest,
2679
+ model: string,
2680
+ ): Promise<AnswerReview | undefined> {
2681
+ const reviewer = this.ctx.reviewAnswer
2682
+ const signal = this.ctx.abortController.signal
2683
+ if (!reviewer || signal.aborted) return undefined
2684
+ let onAbort = () => {}
2685
+ const aborted = new Promise<never>((_resolve, reject) => {
2686
+ onAbort = () => reject(signal.reason ?? new Error('Answer review cancelled'))
2687
+ signal.addEventListener('abort', onAbort, { once: true })
2688
+ })
2689
+ const inference = createCallbackInference(this.ctx, model, 'review')
2352
2690
  try {
2353
- return await this.ctx.reviewAnswer(answer, {
2354
- runId: this.ctx.runMgr.id,
2355
- iteration: this.ctx.runMgr.currentIteration,
2356
- signal: this.ctx.abortController.signal,
2357
- messages: this.ctx.runMgr.messages,
2358
- })
2359
- } catch (err) {
2360
- this.ctx.log.error('Answer review threw — accepting the answer unreviewed', {
2361
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2362
- 'exception.message': toErrorMessage(err),
2363
- })
2364
- return { accept: true }
2691
+ const verdict = await Promise.race([
2692
+ Promise.resolve().then(() => {
2693
+ signal.throwIfAborted()
2694
+ return reviewer(answer, {
2695
+ runId: this.ctx.runMgr.id,
2696
+ iteration: this.ctx.runMgr.currentIteration,
2697
+ signal,
2698
+ messages: this.ctx.runMgr.messages,
2699
+ ...reviewRequest,
2700
+ generateText: inference.generateText,
2701
+ })
2702
+ }),
2703
+ aborted,
2704
+ ])
2705
+ if (signal.aborted) return undefined
2706
+ if (!verdict || typeof verdict.accept !== 'boolean')
2707
+ throw new Error('Answer reviewer returned an invalid verdict')
2708
+ if (!verdict.accept && (typeof verdict.feedback !== 'string' || !verdict.feedback.trim()))
2709
+ throw new Error('Answer reviewer rejection requires feedback')
2710
+ return verdict
2711
+ } catch (error) {
2712
+ if (signal.aborted) return undefined
2713
+ throw new AnswerReviewFailure(error)
2714
+ } finally {
2715
+ inference.close()
2716
+ signal.removeEventListener('abort', onAbort)
2365
2717
  }
2366
2718
  }
2367
2719
 
@@ -2445,7 +2797,7 @@ export class IterationOrchestrator {
2445
2797
  const finalHistory = [
2446
2798
  ...this.ctx.runMgr.messages,
2447
2799
  createRuntimeContextMessage(
2448
- `[SYSTEM] Run is ending due to ${reason}. You MUST provide a final response now summarizing all your findings and work so far. Do not use any tools.`,
2800
+ `[SYSTEM] Run is ending due to ${reason}. ${CLOSING_RESPONSE_GUIDANCE}`,
2449
2801
  'limit-finalization',
2450
2802
  ),
2451
2803
  ]
@@ -2453,6 +2805,7 @@ export class IterationOrchestrator {
2453
2805
  this.projectObservations(finalHistory),
2454
2806
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
2455
2807
  )
2808
+ this.appendWorkContext(finalMessages, this.steps.length + 1, { model })
2456
2809
  await this.reportUnsupportedToolResults(finalMessages)
2457
2810
 
2458
2811
  // Same cache discipline as the forced-final iteration: keep the
@@ -2517,6 +2870,7 @@ export class IterationOrchestrator {
2517
2870
  ? { replayState: response.message.replayState }
2518
2871
  : {}),
2519
2872
  },
2873
+ response.message.textParts,
2520
2874
  )
2521
2875
  this.ctx.runMgr.pushMessage(assistantMsg)
2522
2876
 
@@ -2535,6 +2889,7 @@ export class IterationOrchestrator {
2535
2889
  stopReason: 'forced_finalize',
2536
2890
  usage: response.usage,
2537
2891
  content: response.message.content ?? undefined,
2892
+ ...(response.message.textParts ? { textParts: response.message.textParts } : {}),
2538
2893
  })
2539
2894
  } catch (err) {
2540
2895
  this.ctx.log.error('Failed to get final response', {
@@ -2556,6 +2911,8 @@ export class IterationOrchestrator {
2556
2911
  * threw and is left saying so rather than dressed up as something specific.
2557
2912
  */
2558
2913
  function describeStepFailure(err: unknown, providerId: string): StepFailure {
2914
+ if (err instanceof AnswerReviewFailure)
2915
+ return { message: err.message, code: 'unknown', retryable: false }
2559
2916
  const classified = classifyProviderError(err, providerId)
2560
2917
  return {
2561
2918
  message: toErrorMessage(err),