@namzu/sdk 38.2.1 → 40.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (602) hide show
  1. package/CHANGELOG.md +851 -0
  2. package/dist/advisory/executor.d.ts +10 -1
  3. package/dist/advisory/executor.d.ts.map +1 -1
  4. package/dist/advisory/executor.js +5 -26
  5. package/dist/advisory/executor.js.map +1 -1
  6. package/dist/advisory/history.d.ts +9 -0
  7. package/dist/advisory/history.d.ts.map +1 -0
  8. package/dist/advisory/history.js +120 -0
  9. package/dist/advisory/history.js.map +1 -0
  10. package/dist/advisory/index.d.ts +1 -1
  11. package/dist/advisory/index.d.ts.map +1 -1
  12. package/dist/advisory/index.js.map +1 -1
  13. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  14. package/dist/agents/ReactiveAgent.js +3 -0
  15. package/dist/agents/ReactiveAgent.js.map +1 -1
  16. package/dist/agents/runAgent.d.ts +10 -0
  17. package/dist/agents/runAgent.d.ts.map +1 -1
  18. package/dist/agents/runAgent.js +3 -0
  19. package/dist/agents/runAgent.js.map +1 -1
  20. package/dist/compaction/manual.d.ts +6 -0
  21. package/dist/compaction/manual.d.ts.map +1 -1
  22. package/dist/compaction/manual.js +21 -2
  23. package/dist/compaction/manual.js.map +1 -1
  24. package/dist/compaction/summary.d.ts.map +1 -1
  25. package/dist/compaction/summary.js +4 -1
  26. package/dist/compaction/summary.js.map +1 -1
  27. package/dist/config/runtime.js +2 -2
  28. package/dist/config/runtime.js.map +1 -1
  29. package/dist/connector/mcp/adapter.d.ts.map +1 -1
  30. package/dist/connector/mcp/adapter.js +20 -6
  31. package/dist/connector/mcp/adapter.js.map +1 -1
  32. package/dist/contracts/schemas.js +1 -1
  33. package/dist/contracts/schemas.js.map +1 -1
  34. package/dist/eval/harness-protection.d.ts +18 -0
  35. package/dist/eval/harness-protection.d.ts.map +1 -0
  36. package/dist/eval/harness-protection.js +58 -0
  37. package/dist/eval/harness-protection.js.map +1 -0
  38. package/dist/eval/harness-verification.d.ts +6 -1
  39. package/dist/eval/harness-verification.d.ts.map +1 -1
  40. package/dist/eval/harness-verification.js +19 -2
  41. package/dist/eval/harness-verification.js.map +1 -1
  42. package/dist/eval/index.d.ts +1 -0
  43. package/dist/eval/index.d.ts.map +1 -1
  44. package/dist/eval/index.js.map +1 -1
  45. package/dist/manager/resident/activity.d.ts +50 -0
  46. package/dist/manager/resident/activity.d.ts.map +1 -0
  47. package/dist/manager/resident/activity.js +125 -0
  48. package/dist/manager/resident/activity.js.map +1 -0
  49. package/dist/manager/resident/agenda.d.ts +13 -1
  50. package/dist/manager/resident/agenda.d.ts.map +1 -1
  51. package/dist/manager/resident/agenda.js +40 -5
  52. package/dist/manager/resident/agenda.js.map +1 -1
  53. package/dist/manager/resident/consumption.d.ts +104 -0
  54. package/dist/manager/resident/consumption.d.ts.map +1 -0
  55. package/dist/manager/resident/consumption.js +233 -0
  56. package/dist/manager/resident/consumption.js.map +1 -0
  57. package/dist/manager/resident/evidence-recall.d.ts +24 -0
  58. package/dist/manager/resident/evidence-recall.d.ts.map +1 -0
  59. package/dist/manager/resident/evidence-recall.js +295 -0
  60. package/dist/manager/resident/evidence-recall.js.map +1 -0
  61. package/dist/manager/resident/history-disk.d.ts +10 -0
  62. package/dist/manager/resident/history-disk.d.ts.map +1 -0
  63. package/dist/manager/resident/history-disk.js +50 -0
  64. package/dist/manager/resident/history-disk.js.map +1 -0
  65. package/dist/manager/resident/history.d.ts +79 -0
  66. package/dist/manager/resident/history.d.ts.map +1 -0
  67. package/dist/manager/resident/history.js +203 -0
  68. package/dist/manager/resident/history.js.map +1 -0
  69. package/dist/manager/resident/initiative.d.ts.map +1 -1
  70. package/dist/manager/resident/initiative.js +11 -3
  71. package/dist/manager/resident/initiative.js.map +1 -1
  72. package/dist/manager/resident/learning-cycle.d.ts +131 -0
  73. package/dist/manager/resident/learning-cycle.d.ts.map +1 -0
  74. package/dist/manager/resident/learning-cycle.js +306 -0
  75. package/dist/manager/resident/learning-cycle.js.map +1 -0
  76. package/dist/manager/resident/learning-observation.d.ts +80 -0
  77. package/dist/manager/resident/learning-observation.d.ts.map +1 -0
  78. package/dist/manager/resident/learning-observation.js +22 -0
  79. package/dist/manager/resident/learning-observation.js.map +1 -0
  80. package/dist/manager/resident/learning-store.d.ts +106 -0
  81. package/dist/manager/resident/learning-store.d.ts.map +1 -0
  82. package/dist/manager/resident/learning-store.js +598 -0
  83. package/dist/manager/resident/learning-store.js.map +1 -0
  84. package/dist/manager/resident/learning.d.ts +246 -3
  85. package/dist/manager/resident/learning.d.ts.map +1 -1
  86. package/dist/manager/resident/learning.js +96 -6
  87. package/dist/manager/resident/learning.js.map +1 -1
  88. package/dist/manager/resident/outbox.d.ts +4 -4
  89. package/dist/manager/resident/store.d.ts +37 -4
  90. package/dist/manager/resident/store.d.ts.map +1 -1
  91. package/dist/manager/resident/store.js +27 -3
  92. package/dist/manager/resident/store.js.map +1 -1
  93. package/dist/manager/resident/tool-evidence.d.ts +71 -0
  94. package/dist/manager/resident/tool-evidence.d.ts.map +1 -0
  95. package/dist/manager/resident/tool-evidence.js +285 -0
  96. package/dist/manager/resident/tool-evidence.js.map +1 -0
  97. package/dist/manager/run/persistence.d.ts +8 -0
  98. package/dist/manager/run/persistence.d.ts.map +1 -1
  99. package/dist/manager/run/persistence.js +18 -0
  100. package/dist/manager/run/persistence.js.map +1 -1
  101. package/dist/plugin/loader.d.ts.map +1 -1
  102. package/dist/plugin/loader.js +5 -3
  103. package/dist/plugin/loader.js.map +1 -1
  104. package/dist/prompt/coding-agent-doctrine.d.ts +1 -1
  105. package/dist/prompt/coding-agent-doctrine.d.ts.map +1 -1
  106. package/dist/prompt/coding-agent-doctrine.js +2 -0
  107. package/dist/prompt/coding-agent-doctrine.js.map +1 -1
  108. package/dist/prompt/index.d.ts +2 -0
  109. package/dist/prompt/index.d.ts.map +1 -1
  110. package/dist/prompt/index.js +1 -0
  111. package/dist/prompt/index.js.map +1 -1
  112. package/dist/prompt/resident-learning.d.ts +19 -0
  113. package/dist/prompt/resident-learning.d.ts.map +1 -0
  114. package/dist/prompt/resident-learning.js +125 -0
  115. package/dist/prompt/resident-learning.js.map +1 -0
  116. package/dist/prompt/resident-step.d.ts +8 -1
  117. package/dist/prompt/resident-step.d.ts.map +1 -1
  118. package/dist/prompt/resident-step.js +65 -4
  119. package/dist/prompt/resident-step.js.map +1 -1
  120. package/dist/provider/collect-chat-completion.d.ts +2 -1
  121. package/dist/provider/collect-chat-completion.d.ts.map +1 -1
  122. package/dist/provider/collect-chat-completion.js +7 -6
  123. package/dist/provider/collect-chat-completion.js.map +1 -1
  124. package/dist/provider/fallback.d.ts.map +1 -1
  125. package/dist/provider/fallback.js +2 -1
  126. package/dist/provider/fallback.js.map +1 -1
  127. package/dist/provider/stream-text.d.ts +16 -0
  128. package/dist/provider/stream-text.d.ts.map +1 -0
  129. package/dist/provider/stream-text.js +51 -0
  130. package/dist/provider/stream-text.js.map +1 -0
  131. package/dist/public-runtime.d.ts +16 -4
  132. package/dist/public-runtime.d.ts.map +1 -1
  133. package/dist/public-runtime.js +17 -4
  134. package/dist/public-runtime.js.map +1 -1
  135. package/dist/public-tools.d.ts +13 -0
  136. package/dist/public-tools.d.ts.map +1 -1
  137. package/dist/public-tools.js +16 -0
  138. package/dist/public-tools.js.map +1 -1
  139. package/dist/public-types.d.ts +15 -3
  140. package/dist/public-types.d.ts.map +1 -1
  141. package/dist/registry/tool/execute.d.ts.map +1 -1
  142. package/dist/registry/tool/execute.js +2 -3
  143. package/dist/registry/tool/execute.js.map +1 -1
  144. package/dist/registry/tool/portable.d.ts +65 -0
  145. package/dist/registry/tool/portable.d.ts.map +1 -0
  146. package/dist/registry/tool/portable.js +244 -0
  147. package/dist/registry/tool/portable.js.map +1 -0
  148. package/dist/registry/tool/schema.d.ts +32 -5
  149. package/dist/registry/tool/schema.d.ts.map +1 -1
  150. package/dist/registry/tool/schema.js +35 -9
  151. package/dist/registry/tool/schema.js.map +1 -1
  152. package/dist/registry/toolset/catalog.js +8 -8
  153. package/dist/registry/toolset/catalog.js.map +1 -1
  154. package/dist/run/LimitChecker.js +3 -3
  155. package/dist/run/LimitChecker.js.map +1 -1
  156. package/dist/run/evidence-query.d.ts +41 -0
  157. package/dist/run/evidence-query.d.ts.map +1 -0
  158. package/dist/run/evidence-query.js +270 -0
  159. package/dist/run/evidence-query.js.map +1 -0
  160. package/dist/run/evidence-recall.d.ts +99 -0
  161. package/dist/run/evidence-recall.d.ts.map +1 -0
  162. package/dist/run/evidence-recall.js +633 -0
  163. package/dist/run/evidence-recall.js.map +1 -0
  164. package/dist/run/index.d.ts +2 -0
  165. package/dist/run/index.d.ts.map +1 -1
  166. package/dist/run/index.js +1 -0
  167. package/dist/run/index.js.map +1 -1
  168. package/dist/run/json-claim-verifier.d.ts +83 -0
  169. package/dist/run/json-claim-verifier.d.ts.map +1 -0
  170. package/dist/run/json-claim-verifier.js +200 -0
  171. package/dist/run/json-claim-verifier.js.map +1 -0
  172. package/dist/run/preparation-context-error.d.ts +10 -0
  173. package/dist/run/preparation-context-error.d.ts.map +1 -0
  174. package/dist/run/preparation-context-error.js +15 -0
  175. package/dist/run/preparation-context-error.js.map +1 -0
  176. package/dist/run-query/index.d.ts +3 -1
  177. package/dist/run-query/index.d.ts.map +1 -1
  178. package/dist/runtime/jobs/awaited-jobs.d.ts +215 -0
  179. package/dist/runtime/jobs/awaited-jobs.d.ts.map +1 -0
  180. package/dist/runtime/jobs/awaited-jobs.js +259 -0
  181. package/dist/runtime/jobs/awaited-jobs.js.map +1 -0
  182. package/dist/runtime/jobs/registry.d.ts +33 -2
  183. package/dist/runtime/jobs/registry.d.ts.map +1 -1
  184. package/dist/runtime/jobs/registry.js +37 -0
  185. package/dist/runtime/jobs/registry.js.map +1 -1
  186. package/dist/runtime/query/callback-inference.d.ts +8 -0
  187. package/dist/runtime/query/callback-inference.d.ts.map +1 -0
  188. package/dist/runtime/query/callback-inference.js +89 -0
  189. package/dist/runtime/query/callback-inference.js.map +1 -0
  190. package/dist/runtime/query/checkpoint.d.ts +4 -0
  191. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  192. package/dist/runtime/query/checkpoint.js +13 -0
  193. package/dist/runtime/query/checkpoint.js.map +1 -1
  194. package/dist/runtime/query/events.d.ts +1 -0
  195. package/dist/runtime/query/events.d.ts.map +1 -1
  196. package/dist/runtime/query/events.js +18 -0
  197. package/dist/runtime/query/events.js.map +1 -1
  198. package/dist/runtime/query/executor.d.ts +35 -1
  199. package/dist/runtime/query/executor.d.ts.map +1 -1
  200. package/dist/runtime/query/executor.js +77 -8
  201. package/dist/runtime/query/executor.js.map +1 -1
  202. package/dist/runtime/query/file-evidence-context.d.ts +5 -0
  203. package/dist/runtime/query/file-evidence-context.d.ts.map +1 -0
  204. package/dist/runtime/query/file-evidence-context.js +180 -0
  205. package/dist/runtime/query/file-evidence-context.js.map +1 -0
  206. package/dist/runtime/query/file-evidence-replay.d.ts +260 -0
  207. package/dist/runtime/query/file-evidence-replay.d.ts.map +1 -0
  208. package/dist/runtime/query/file-evidence-replay.js +647 -0
  209. package/dist/runtime/query/file-evidence-replay.js.map +1 -0
  210. package/dist/runtime/query/file-evidence-seed.d.ts +50 -0
  211. package/dist/runtime/query/file-evidence-seed.d.ts.map +1 -0
  212. package/dist/runtime/query/file-evidence-seed.js +100 -0
  213. package/dist/runtime/query/file-evidence-seed.js.map +1 -0
  214. package/dist/runtime/query/guard.d.ts.map +1 -1
  215. package/dist/runtime/query/guard.js +4 -0
  216. package/dist/runtime/query/guard.js.map +1 -1
  217. package/dist/runtime/query/index.d.ts +10 -1
  218. package/dist/runtime/query/index.d.ts.map +1 -1
  219. package/dist/runtime/query/index.js +133 -26
  220. package/dist/runtime/query/index.js.map +1 -1
  221. package/dist/runtime/query/iteration/index.d.ts +92 -36
  222. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  223. package/dist/runtime/query/iteration/index.js +420 -117
  224. package/dist/runtime/query/iteration/index.js.map +1 -1
  225. package/dist/runtime/query/iteration/phases/advisory.d.ts +2 -1
  226. package/dist/runtime/query/iteration/phases/advisory.d.ts.map +1 -1
  227. package/dist/runtime/query/iteration/phases/advisory.js +2 -1
  228. package/dist/runtime/query/iteration/phases/advisory.js.map +1 -1
  229. package/dist/runtime/query/iteration/phases/context.d.ts +12 -1
  230. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  231. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  232. package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
  233. package/dist/runtime/query/iteration/phases/tool-review.js +8 -1
  234. package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
  235. package/dist/runtime/query/iteration/provider-rejected-image.d.ts +2 -1
  236. package/dist/runtime/query/iteration/provider-rejected-image.d.ts.map +1 -1
  237. package/dist/runtime/query/iteration/provider-rejected-image.js +5 -2
  238. package/dist/runtime/query/iteration/provider-rejected-image.js.map +1 -1
  239. package/dist/runtime/query/iteration/stream-turn.d.ts +4 -1
  240. package/dist/runtime/query/iteration/stream-turn.d.ts.map +1 -1
  241. package/dist/runtime/query/iteration/stream-turn.js +22 -8
  242. package/dist/runtime/query/iteration/stream-turn.js.map +1 -1
  243. package/dist/runtime/query/plugin-hooks.d.ts +14 -0
  244. package/dist/runtime/query/plugin-hooks.d.ts.map +1 -1
  245. package/dist/runtime/query/plugin-hooks.js +18 -0
  246. package/dist/runtime/query/plugin-hooks.js.map +1 -1
  247. package/dist/runtime/query/repeat-call.d.ts +17 -4
  248. package/dist/runtime/query/repeat-call.d.ts.map +1 -1
  249. package/dist/runtime/query/repeat-call.js +26 -19
  250. package/dist/runtime/query/repeat-call.js.map +1 -1
  251. package/dist/runtime/query/resume-pending.d.ts +18 -33
  252. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  253. package/dist/runtime/query/resume-pending.js +59 -42
  254. package/dist/runtime/query/resume-pending.js.map +1 -1
  255. package/dist/runtime/query/review-policy.d.ts +4 -4
  256. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  257. package/dist/runtime/query/review-policy.js +10 -9
  258. package/dist/runtime/query/review-policy.js.map +1 -1
  259. package/dist/runtime/query/sandbox-lifecycle.d.ts.map +1 -1
  260. package/dist/runtime/query/sandbox-lifecycle.js +4 -0
  261. package/dist/runtime/query/sandbox-lifecycle.js.map +1 -1
  262. package/dist/runtime/query/steering.d.ts +11 -1
  263. package/dist/runtime/query/steering.d.ts.map +1 -1
  264. package/dist/runtime/query/steering.js +12 -1
  265. package/dist/runtime/query/steering.js.map +1 -1
  266. package/dist/runtime/query/tool-output-budget.d.ts +10 -6
  267. package/dist/runtime/query/tool-output-budget.d.ts.map +1 -1
  268. package/dist/runtime/query/tool-output-budget.js +54 -12
  269. package/dist/runtime/query/tool-output-budget.js.map +1 -1
  270. package/dist/runtime/query/tooling.d.ts +5 -1
  271. package/dist/runtime/query/tooling.d.ts.map +1 -1
  272. package/dist/runtime/query/tooling.js +5 -0
  273. package/dist/runtime/query/tooling.js.map +1 -1
  274. package/dist/scheduler/completion-inbox.d.ts +49 -0
  275. package/dist/scheduler/completion-inbox.d.ts.map +1 -1
  276. package/dist/scheduler/completion-inbox.js +122 -2
  277. package/dist/scheduler/completion-inbox.js.map +1 -1
  278. package/dist/store/evidence/compaction-archive.d.ts +109 -0
  279. package/dist/store/evidence/compaction-archive.d.ts.map +1 -0
  280. package/dist/store/evidence/compaction-archive.js +125 -0
  281. package/dist/store/evidence/compaction-archive.js.map +1 -0
  282. package/dist/store/evidence/compaction-provenance.d.ts +7 -0
  283. package/dist/store/evidence/compaction-provenance.d.ts.map +1 -0
  284. package/dist/store/evidence/compaction-provenance.js +49 -0
  285. package/dist/store/evidence/compaction-provenance.js.map +1 -0
  286. package/dist/store/evidence/compaction-text.d.ts +16 -0
  287. package/dist/store/evidence/compaction-text.d.ts.map +1 -0
  288. package/dist/store/evidence/compaction-text.js +51 -0
  289. package/dist/store/evidence/compaction-text.js.map +1 -0
  290. package/dist/store/evidence/disk.d.ts +11 -0
  291. package/dist/store/evidence/disk.d.ts.map +1 -0
  292. package/dist/store/evidence/disk.js +367 -0
  293. package/dist/store/evidence/disk.js.map +1 -0
  294. package/dist/store/evidence/format.d.ts +25 -0
  295. package/dist/store/evidence/format.d.ts.map +1 -0
  296. package/dist/store/evidence/format.js +115 -0
  297. package/dist/store/evidence/format.js.map +1 -0
  298. package/dist/store/evidence/index-page.d.ts +208 -0
  299. package/dist/store/evidence/index-page.d.ts.map +1 -0
  300. package/dist/store/evidence/index-page.js +262 -0
  301. package/dist/store/evidence/index-page.js.map +1 -0
  302. package/dist/store/evidence/io.d.ts +24 -0
  303. package/dist/store/evidence/io.d.ts.map +1 -0
  304. package/dist/store/evidence/io.js +69 -0
  305. package/dist/store/evidence/io.js.map +1 -0
  306. package/dist/store/evidence/linked.d.ts +9 -0
  307. package/dist/store/evidence/linked.d.ts.map +1 -0
  308. package/dist/store/evidence/linked.js +314 -0
  309. package/dist/store/evidence/linked.js.map +1 -0
  310. package/dist/store/evidence/passages.d.ts +15 -0
  311. package/dist/store/evidence/passages.d.ts.map +1 -0
  312. package/dist/store/evidence/passages.js +72 -0
  313. package/dist/store/evidence/passages.js.map +1 -0
  314. package/dist/store/evidence/record-chain.d.ts +42 -0
  315. package/dist/store/evidence/record-chain.d.ts.map +1 -0
  316. package/dist/store/evidence/record-chain.js +111 -0
  317. package/dist/store/evidence/record-chain.js.map +1 -0
  318. package/dist/store/evidence/search-input.d.ts +21 -0
  319. package/dist/store/evidence/search-input.d.ts.map +1 -0
  320. package/dist/store/evidence/search-input.js +44 -0
  321. package/dist/store/evidence/search-input.js.map +1 -0
  322. package/dist/store/evidence/selection.d.ts +9 -0
  323. package/dist/store/evidence/selection.d.ts.map +1 -0
  324. package/dist/store/evidence/selection.js +17 -0
  325. package/dist/store/evidence/selection.js.map +1 -0
  326. package/dist/store/evidence/source-kind.d.ts +11 -0
  327. package/dist/store/evidence/source-kind.d.ts.map +1 -0
  328. package/dist/store/evidence/source-kind.js +26 -0
  329. package/dist/store/evidence/source-kind.js.map +1 -0
  330. package/dist/store/evidence/source-text.d.ts +51 -0
  331. package/dist/store/evidence/source-text.d.ts.map +1 -0
  332. package/dist/store/evidence/source-text.js +213 -0
  333. package/dist/store/evidence/source-text.js.map +1 -0
  334. package/dist/store/evidence/types.d.ts +162 -0
  335. package/dist/store/evidence/types.d.ts.map +1 -0
  336. package/dist/store/evidence/types.js +2 -0
  337. package/dist/store/evidence/types.js.map +1 -0
  338. package/dist/store/memory/disk.d.ts +2 -0
  339. package/dist/store/memory/disk.d.ts.map +1 -1
  340. package/dist/store/memory/disk.js +2 -1
  341. package/dist/store/memory/disk.js.map +1 -1
  342. package/dist/store/run/disk.d.ts +10 -0
  343. package/dist/store/run/disk.d.ts.map +1 -1
  344. package/dist/store/run/disk.js +87 -7
  345. package/dist/store/run/disk.js.map +1 -1
  346. package/dist/store/run/memory.d.ts +1 -0
  347. package/dist/store/run/memory.d.ts.map +1 -1
  348. package/dist/store/run/memory.js +10 -0
  349. package/dist/store/run/memory.js.map +1 -1
  350. package/dist/store/run/tool-executions.d.ts +13 -0
  351. package/dist/store/run/tool-executions.d.ts.map +1 -0
  352. package/dist/store/run/tool-executions.js +99 -0
  353. package/dist/store/run/tool-executions.js.map +1 -0
  354. package/dist/store/session/index.d.ts +2 -0
  355. package/dist/store/session/index.d.ts.map +1 -1
  356. package/dist/store/session/index.js +1 -0
  357. package/dist/store/session/index.js.map +1 -1
  358. package/dist/store/session/sqlite.d.ts +57 -0
  359. package/dist/store/session/sqlite.d.ts.map +1 -0
  360. package/dist/store/session/sqlite.js +430 -0
  361. package/dist/store/session/sqlite.js.map +1 -0
  362. package/dist/tools/builtins/bash.d.ts.map +1 -1
  363. package/dist/tools/builtins/bash.js +4 -10
  364. package/dist/tools/builtins/bash.js.map +1 -1
  365. package/dist/tools/builtins/edit-apply.d.ts +126 -0
  366. package/dist/tools/builtins/edit-apply.d.ts.map +1 -0
  367. package/dist/tools/builtins/edit-apply.js +360 -0
  368. package/dist/tools/builtins/edit-apply.js.map +1 -0
  369. package/dist/tools/builtins/edit.d.ts +143 -1
  370. package/dist/tools/builtins/edit.d.ts.map +1 -1
  371. package/dist/tools/builtins/edit.js +37 -219
  372. package/dist/tools/builtins/edit.js.map +1 -1
  373. package/dist/tools/builtins/index.d.ts +1 -0
  374. package/dist/tools/builtins/index.d.ts.map +1 -1
  375. package/dist/tools/builtins/index.js +9 -3
  376. package/dist/tools/builtins/index.js.map +1 -1
  377. package/dist/tools/builtins/job.d.ts.map +1 -1
  378. package/dist/tools/builtins/job.js +5 -6
  379. package/dist/tools/builtins/job.js.map +1 -1
  380. package/dist/tools/builtins/read-file.d.ts +2 -2
  381. package/dist/tools/builtins/read-file.d.ts.map +1 -1
  382. package/dist/tools/builtins/read-file.js +50 -65
  383. package/dist/tools/builtins/read-file.js.map +1 -1
  384. package/dist/tools/builtins/read-render.d.ts +56 -0
  385. package/dist/tools/builtins/read-render.d.ts.map +1 -0
  386. package/dist/tools/builtins/read-render.js +73 -0
  387. package/dist/tools/builtins/read-render.js.map +1 -0
  388. package/dist/tools/builtins/wait-for-job-bounds.d.ts +67 -0
  389. package/dist/tools/builtins/wait-for-job-bounds.d.ts.map +1 -0
  390. package/dist/tools/builtins/wait-for-job-bounds.js +108 -0
  391. package/dist/tools/builtins/wait-for-job-bounds.js.map +1 -0
  392. package/dist/tools/builtins/wait-for-job.d.ts +6 -0
  393. package/dist/tools/builtins/wait-for-job.d.ts.map +1 -0
  394. package/dist/tools/builtins/wait-for-job.js +162 -0
  395. package/dist/tools/builtins/wait-for-job.js.map +1 -0
  396. package/dist/tools/builtins/write-file.js +7 -2
  397. package/dist/tools/builtins/write-file.js.map +1 -1
  398. package/dist/tools/coordinator/index.d.ts.map +1 -1
  399. package/dist/tools/coordinator/index.js +1 -7
  400. package/dist/tools/coordinator/index.js.map +1 -1
  401. package/dist/tools/defineTool.d.ts +2 -1
  402. package/dist/tools/defineTool.d.ts.map +1 -1
  403. package/dist/tools/defineTool.js +1 -1
  404. package/dist/tools/defineTool.js.map +1 -1
  405. package/dist/tools/file-read-tracker.d.ts.map +1 -1
  406. package/dist/tools/file-read-tracker.js +90 -3
  407. package/dist/tools/file-read-tracker.js.map +1 -1
  408. package/dist/tools/resident-history.d.ts +10 -0
  409. package/dist/tools/resident-history.d.ts.map +1 -0
  410. package/dist/tools/resident-history.js +80 -0
  411. package/dist/tools/resident-history.js.map +1 -0
  412. package/dist/tools/resident-tool-evidence.d.ts +5 -0
  413. package/dist/tools/resident-tool-evidence.d.ts.map +1 -0
  414. package/dist/tools/resident-tool-evidence.js +57 -0
  415. package/dist/tools/resident-tool-evidence.js.map +1 -0
  416. package/dist/types/advisory/config.d.ts +7 -0
  417. package/dist/types/advisory/config.d.ts.map +1 -1
  418. package/dist/types/agent/reactive.d.ts +1 -0
  419. package/dist/types/agent/reactive.d.ts.map +1 -1
  420. package/dist/types/authorization/index.d.ts +9 -9
  421. package/dist/types/authorization/index.d.ts.map +1 -1
  422. package/dist/types/authorization/index.js +1 -1
  423. package/dist/types/authorization/index.js.map +1 -1
  424. package/dist/types/hitl/index.d.ts +4 -0
  425. package/dist/types/hitl/index.d.ts.map +1 -1
  426. package/dist/types/hitl/index.js.map +1 -1
  427. package/dist/types/message/index.d.ts +16 -2
  428. package/dist/types/message/index.d.ts.map +1 -1
  429. package/dist/types/message/index.js +10 -1
  430. package/dist/types/message/index.js.map +1 -1
  431. package/dist/types/provider/chat.d.ts +2 -0
  432. package/dist/types/provider/chat.d.ts.map +1 -1
  433. package/dist/types/provider/stream.d.ts +4 -0
  434. package/dist/types/provider/stream.d.ts.map +1 -1
  435. package/dist/types/run/answer-review.d.ts +34 -3
  436. package/dist/types/run/answer-review.d.ts.map +1 -1
  437. package/dist/types/run/config.d.ts +3 -0
  438. package/dist/types/run/config.d.ts.map +1 -1
  439. package/dist/types/run/entity.d.ts +20 -0
  440. package/dist/types/run/entity.d.ts.map +1 -1
  441. package/dist/types/run/events.d.ts +31 -8
  442. package/dist/types/run/events.d.ts.map +1 -1
  443. package/dist/types/run/events.js.map +1 -1
  444. package/dist/types/run/prepare-step.d.ts +53 -8
  445. package/dist/types/run/prepare-step.d.ts.map +1 -1
  446. package/dist/types/run/store.d.ts +29 -4
  447. package/dist/types/run/store.d.ts.map +1 -1
  448. package/dist/types/run/store.js +0 -28
  449. package/dist/types/run/store.js.map +1 -1
  450. package/dist/types/sandbox/index.d.ts +15 -14
  451. package/dist/types/sandbox/index.d.ts.map +1 -1
  452. package/dist/types/sandbox/index.js.map +1 -1
  453. package/dist/types/tool/index.d.ts +124 -1
  454. package/dist/types/tool/index.d.ts.map +1 -1
  455. package/dist/types/tool/index.js.map +1 -1
  456. package/dist/utils/await-with-abort.d.ts +8 -0
  457. package/dist/utils/await-with-abort.d.ts.map +1 -0
  458. package/dist/utils/await-with-abort.js +28 -0
  459. package/dist/utils/await-with-abort.js.map +1 -0
  460. package/dist/utils/env.d.ts +19 -0
  461. package/dist/utils/env.d.ts.map +1 -0
  462. package/dist/utils/env.js +25 -0
  463. package/dist/utils/env.js.map +1 -0
  464. package/dist/utils/evidence-time.d.ts +3 -0
  465. package/dist/utils/evidence-time.d.ts.map +1 -0
  466. package/dist/utils/evidence-time.js +10 -0
  467. package/dist/utils/evidence-time.js.map +1 -0
  468. package/dist/utils/evidence-tokens.d.ts +13 -0
  469. package/dist/utils/evidence-tokens.d.ts.map +1 -0
  470. package/dist/utils/evidence-tokens.js +29 -0
  471. package/dist/utils/evidence-tokens.js.map +1 -0
  472. package/package.json +1 -1
  473. package/src/advisory/executor.ts +15 -30
  474. package/src/advisory/history.ts +126 -0
  475. package/src/advisory/index.ts +5 -1
  476. package/src/agents/ReactiveAgent.ts +3 -0
  477. package/src/agents/runAgent.ts +14 -1
  478. package/src/compaction/manual.ts +30 -2
  479. package/src/compaction/summary.ts +4 -1
  480. package/src/config/runtime.ts +2 -2
  481. package/src/connector/mcp/adapter.ts +20 -6
  482. package/src/contracts/schemas.ts +1 -1
  483. package/src/eval/harness-protection.ts +79 -0
  484. package/src/eval/harness-verification.ts +31 -1
  485. package/src/eval/index.ts +1 -0
  486. package/src/manager/resident/activity.ts +187 -0
  487. package/src/manager/resident/agenda.ts +59 -5
  488. package/src/manager/resident/consumption.ts +311 -0
  489. package/src/manager/resident/evidence-recall.ts +363 -0
  490. package/src/manager/resident/history-disk.ts +63 -0
  491. package/src/manager/resident/history.ts +312 -0
  492. package/src/manager/resident/initiative.ts +13 -3
  493. package/src/manager/resident/learning-cycle.ts +499 -0
  494. package/src/manager/resident/learning-observation.ts +39 -0
  495. package/src/manager/resident/learning-store.ts +813 -0
  496. package/src/manager/resident/learning.ts +132 -8
  497. package/src/manager/resident/store.ts +31 -3
  498. package/src/manager/resident/tool-evidence.ts +412 -0
  499. package/src/manager/run/persistence.ts +18 -0
  500. package/src/plugin/loader.ts +8 -3
  501. package/src/prompt/coding-agent-doctrine.ts +2 -0
  502. package/src/prompt/index.ts +2 -0
  503. package/src/prompt/resident-learning.ts +143 -0
  504. package/src/prompt/resident-step.ts +83 -3
  505. package/src/provider/collect-chat-completion.ts +7 -6
  506. package/src/provider/fallback.ts +2 -1
  507. package/src/provider/stream-text.ts +56 -0
  508. package/src/public-runtime.ts +39 -1
  509. package/src/public-tools.ts +21 -0
  510. package/src/public-types.ts +98 -0
  511. package/src/registry/tool/execute.ts +2 -4
  512. package/src/registry/tool/portable.ts +264 -0
  513. package/src/registry/tool/schema.ts +38 -8
  514. package/src/registry/toolset/catalog.ts +8 -9
  515. package/src/run/LimitChecker.ts +3 -3
  516. package/src/run/evidence-query.ts +333 -0
  517. package/src/run/evidence-recall.ts +854 -0
  518. package/src/run/index.ts +11 -0
  519. package/src/run/json-claim-verifier.ts +298 -0
  520. package/src/run/preparation-context-error.ts +16 -0
  521. package/src/run-query/index.ts +1 -1
  522. package/src/runtime/jobs/awaited-jobs.ts +271 -0
  523. package/src/runtime/jobs/registry.ts +50 -0
  524. package/src/runtime/query/callback-inference.ts +94 -0
  525. package/src/runtime/query/checkpoint.ts +14 -0
  526. package/src/runtime/query/events.ts +22 -0
  527. package/src/runtime/query/executor.ts +100 -10
  528. package/src/runtime/query/file-evidence-context.ts +209 -0
  529. package/src/runtime/query/file-evidence-replay.ts +776 -0
  530. package/src/runtime/query/file-evidence-seed.ts +126 -0
  531. package/src/runtime/query/guard.ts +2 -0
  532. package/src/runtime/query/index.ts +153 -27
  533. package/src/runtime/query/iteration/index.ts +467 -110
  534. package/src/runtime/query/iteration/phases/advisory.ts +3 -0
  535. package/src/runtime/query/iteration/phases/context.ts +12 -0
  536. package/src/runtime/query/iteration/phases/tool-review.ts +7 -0
  537. package/src/runtime/query/iteration/provider-rejected-image.ts +6 -1
  538. package/src/runtime/query/iteration/stream-turn.ts +35 -8
  539. package/src/runtime/query/plugin-hooks.ts +20 -0
  540. package/src/runtime/query/repeat-call.ts +28 -18
  541. package/src/runtime/query/resume-pending.ts +65 -40
  542. package/src/runtime/query/review-policy.ts +13 -9
  543. package/src/runtime/query/sandbox-lifecycle.ts +3 -0
  544. package/src/runtime/query/steering.ts +11 -0
  545. package/src/runtime/query/tool-output-budget.ts +65 -13
  546. package/src/runtime/query/tooling.ts +10 -1
  547. package/src/scheduler/completion-inbox.ts +124 -2
  548. package/src/store/evidence/compaction-archive.ts +139 -0
  549. package/src/store/evidence/compaction-provenance.ts +52 -0
  550. package/src/store/evidence/compaction-text.ts +61 -0
  551. package/src/store/evidence/disk.ts +461 -0
  552. package/src/store/evidence/format.ts +126 -0
  553. package/src/store/evidence/index-page.ts +292 -0
  554. package/src/store/evidence/io.ts +95 -0
  555. package/src/store/evidence/linked.ts +365 -0
  556. package/src/store/evidence/passages.ts +88 -0
  557. package/src/store/evidence/record-chain.ts +108 -0
  558. package/src/store/evidence/search-input.ts +62 -0
  559. package/src/store/evidence/selection.ts +25 -0
  560. package/src/store/evidence/source-kind.ts +36 -0
  561. package/src/store/evidence/source-text.ts +285 -0
  562. package/src/store/evidence/types.ts +177 -0
  563. package/src/store/memory/disk.ts +4 -1
  564. package/src/store/run/disk.ts +110 -8
  565. package/src/store/run/memory.ts +10 -0
  566. package/src/store/run/tool-executions.ts +112 -0
  567. package/src/store/session/index.ts +2 -0
  568. package/src/store/session/sqlite.ts +584 -0
  569. package/src/tools/builtins/bash.ts +4 -10
  570. package/src/tools/builtins/edit-apply.ts +456 -0
  571. package/src/tools/builtins/edit.ts +39 -270
  572. package/src/tools/builtins/index.ts +9 -3
  573. package/src/tools/builtins/job.ts +5 -6
  574. package/src/tools/builtins/read-file.ts +56 -77
  575. package/src/tools/builtins/read-render.ts +104 -0
  576. package/src/tools/builtins/wait-for-job-bounds.ts +179 -0
  577. package/src/tools/builtins/wait-for-job.ts +184 -0
  578. package/src/tools/builtins/write-file.ts +7 -2
  579. package/src/tools/coordinator/index.ts +1 -7
  580. package/src/tools/defineTool.ts +4 -2
  581. package/src/tools/file-read-tracker.ts +86 -2
  582. package/src/tools/resident-history.ts +86 -0
  583. package/src/tools/resident-tool-evidence.ts +68 -0
  584. package/src/types/advisory/config.ts +7 -0
  585. package/src/types/agent/reactive.ts +1 -0
  586. package/src/types/authorization/index.ts +2 -2
  587. package/src/types/hitl/index.ts +4 -0
  588. package/src/types/message/index.ts +22 -0
  589. package/src/types/provider/chat.ts +2 -0
  590. package/src/types/provider/stream.ts +4 -0
  591. package/src/types/run/answer-review.ts +34 -3
  592. package/src/types/run/config.ts +3 -0
  593. package/src/types/run/entity.ts +21 -0
  594. package/src/types/run/events.ts +31 -8
  595. package/src/types/run/prepare-step.ts +59 -8
  596. package/src/types/run/store.ts +35 -4
  597. package/src/types/sandbox/index.ts +15 -14
  598. package/src/types/tool/index.ts +123 -1
  599. package/src/utils/await-with-abort.ts +26 -0
  600. package/src/utils/env.ts +23 -0
  601. package/src/utils/evidence-time.ts +9 -0
  602. package/src/utils/evidence-tokens.ts +32 -0
@@ -8,6 +8,7 @@ import { renderSkillsSection } from '../../../persona/assembler.js';
8
8
  import { resolveProviderCapabilities } from '../../../provider/capabilities.js';
9
9
  import { collectChatCompletion } from '../../../provider/collect-chat-completion.js';
10
10
  import { renderToolSchema } from '../../../registry/tool/schema.js';
11
+ import { PreparationContextError } from '../../../run/preparation-context-error.js';
11
12
  import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js';
12
13
  import { GENAI, NAMZU, agentIterationSpanName, parentContext, } from '../../../telemetry/attributes.js';
13
14
  import { getTracer } from '../../../telemetry/runtime-accessors.js';
@@ -16,14 +17,16 @@ import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js';
16
17
  import { NamzuError } from '../../../types/errors/index.js';
17
18
  import { createAssistantMessage, createRuntimeContextMessage, createSystemMessage, } from '../../../types/message/index.js';
18
19
  import { classifyProviderError } from '../../../types/provider/errors.js';
20
+ import { readPositiveIntEnv } from '../../../utils/env.js';
19
21
  import { toErrorMessage } from '../../../utils/error.js';
20
22
  import { stableDigest } from '../../../utils/hash.js';
21
23
  import { generateMessageId } from '../../../utils/id.js';
24
+ import { createCallbackInference } from '../callback-inference.js';
22
25
  import { projectObservationContext } from '../observation-context.js';
23
26
  import { applyLifecycleHookResults } from '../plugin-hooks.js';
24
27
  import { diffRequestContext, snapshotRequestContext, } from '../request-context.js';
25
28
  import { DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES, markProviderRejectedImage, projectRequestRichContent, } from '../request-rich-content.js';
26
- import { formatSteeringNote, isOperatorUserMessage } from '../steering.js';
29
+ import { formatJobNote, formatSteeringNote, isOperatorUserMessage } from '../steering.js';
27
30
  import { parseNativeCandidate } from './native-output.js';
28
31
  import { runAdvisoryPhase } from './phases/advisory.js';
29
32
  import { runIterationCheckpoint } from './phases/checkpoint.js';
@@ -33,6 +36,18 @@ import { runToolReview } from './phases/tool-review.js';
33
36
  import { refreshWorkingMemory } from './phases/working-memory.js';
34
37
  import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js';
35
38
  import { streamProviderTurn } from './stream-turn.js';
39
+ /** A host reviewer is not the model transport, even when its cause is an HTTP failure. */
40
+ class AnswerReviewFailure extends NamzuError {
41
+ constructor(cause) {
42
+ super({
43
+ code: 'unknown',
44
+ message: `Answer review failed: ${toErrorMessage(cause)}`,
45
+ retryable: false,
46
+ details: { phase: 'answer-review' },
47
+ cause,
48
+ });
49
+ }
50
+ }
36
51
  /**
37
52
  * How many times an answer may be handed back before the run stops.
38
53
  *
@@ -42,6 +57,9 @@ import { streamProviderTurn } from './stream-turn.js';
42
57
  * on the thing that actually went wrong.
43
58
  */
44
59
  const DEFAULT_ANSWER_REVIEW_LIMIT = 3;
60
+ // Ending a run changes the available actions, not the strength of its evidence.
61
+ // Use the same standard for warning closure and empty-completion recovery.
62
+ const CLOSING_RESPONSE_GUIDANCE = 'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.';
45
63
  /**
46
64
  * The share of a run's REMAINING time a settle-hold may take.
47
65
  *
@@ -99,8 +117,59 @@ const SETTLE_GRACE_FRACTION = 0.5;
99
117
  export function settleGraceMs(remainingBeforeFinalizeMs) {
100
118
  return Math.min(Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION), DELEGATION_TIMEOUT_MS);
101
119
  }
120
+ /**
121
+ * The ceiling on the job half of that grace, in milliseconds.
122
+ *
123
+ * `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
124
+ * only opens where it matters most: a run with no `timeoutMs` — the CLI's
125
+ * shipping default, `No run deadline by default` — has infinite time before
126
+ * it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
127
+ * delegated task that is sound, because the hour is the longest the task
128
+ * itself may live: the hold cannot outlast the work. A background job has no
129
+ * such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
130
+ * the same arithmetic parks an interactive session for an hour on a job that
131
+ * was never going to exit.
132
+ *
133
+ * So the job leg gets its own bound, and it is sized to what the wait buys
134
+ * rather than to how long a job may live: a turn in which to use the exit.
135
+ * A model that already waited its `wait_for_job` bound out and saw nothing is
136
+ * not usually two minutes from an exit, and the run ending is not the news
137
+ * being lost — with no run in flight the session announces the exit itself
138
+ * (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
139
+ * cheaper of the two places to hear it.
140
+ */
141
+ const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000;
142
+ /**
143
+ * The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
144
+ *
145
+ * `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
146
+ * longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
147
+ * `wait_for_job`'s own bound — and it is the same parse, so a value that is
148
+ * not a positive whole number of milliseconds leaves the default standing
149
+ * rather than holding a run for `NaN`. Called here rather than at module
150
+ * load, because a host that sets it after import is not ignored.
151
+ */
152
+ export function awaitedJobGraceMs(remainingBeforeFinalizeMs) {
153
+ const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS);
154
+ return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling);
155
+ }
102
156
  export class IterationOrchestrator {
103
157
  ctx;
158
+ advisoryTurn;
159
+ /** Live only within its iteration; never joined by a guessed array offset. */
160
+ getAdvisoryTurnContext() {
161
+ const turn = this.advisoryTurn;
162
+ if (!turn || turn.iteration !== this.ctx.runMgr.currentIteration)
163
+ return undefined;
164
+ const start = this.ctx.runMgr.messages.indexOf(turn.response);
165
+ if (start < 0)
166
+ return undefined;
167
+ return {
168
+ iteration: turn.iteration,
169
+ requestMessages: turn.requestMessages,
170
+ subsequentMessages: this.ctx.runMgr.messages.slice(start),
171
+ };
172
+ }
104
173
  /** Rejections so far. See {@link DEFAULT_ANSWER_REVIEW_LIMIT}. */
105
174
  answerReviewAttempts = 0;
106
175
  /**
@@ -143,6 +212,7 @@ export class IterationOrchestrator {
143
212
  },
144
213
  };
145
214
  ctx.checkpointMgr.setLatestUserMessageSource(() => this.latestUserMessage);
215
+ ctx.checkpointMgr.setAnswerReviewAttemptsSource?.(() => this.answerReviewAttempts);
146
216
  ctx.checkpointMgr.setStructuredReviewAttemptsSource?.(() => this.structuredReviewAttempts);
147
217
  ctx.checkpointMgr.setNativeStructuredAttemptsSource?.(() => this.nativeStructuredAttempts);
148
218
  if (ctx.structuredOutput?.mode === 'native') {
@@ -156,6 +226,9 @@ export class IterationOrchestrator {
156
226
  const maxReviews = ctx.structuredOutput?.maxReviews;
157
227
  if (maxReviews !== undefined && (!Number.isSafeInteger(maxReviews) || maxReviews < 0))
158
228
  throw new RangeError('structuredOutput.maxReviews must be a nonnegative safe integer');
229
+ if (ctx.maxAnswerReviews !== undefined &&
230
+ (!Number.isSafeInteger(ctx.maxAnswerReviews) || ctx.maxAnswerReviews < 0))
231
+ throw new RangeError('maxAnswerReviews must be a nonnegative safe integer');
159
232
  }
160
233
  /**
161
234
  * Check the exact post-budget request for tool-result shapes the active driver
@@ -233,6 +306,7 @@ export class IterationOrchestrator {
233
306
  const tracer = getTracer();
234
307
  // Resume hydration happens after construction, before the loop starts.
235
308
  this.latestUserMessage = this.ctx.checkpointMgr.restoredLatestUserMessage;
309
+ this.answerReviewAttempts = this.ctx.checkpointMgr.restoredAnswerReviewAttempts ?? 0;
236
310
  this.structuredReviewAttempts = this.ctx.checkpointMgr.restoredStructuredReviewAttempts ?? 0;
237
311
  this.nativeStructuredAttempts = this.ctx.checkpointMgr.restoredNativeStructuredAttempts ?? 0;
238
312
  if (!this.latestUserMessage) {
@@ -278,6 +352,11 @@ export class IterationOrchestrator {
278
352
  runMgr.setStopReason('structured_output_failed');
279
353
  break;
280
354
  }
355
+ if (this.ctx.reviewAnswer &&
356
+ this.answerReviewAttempts > (this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT)) {
357
+ runMgr.setStopReason('answer_rejected');
358
+ break;
359
+ }
281
360
  if (this.ctx.structuredOutput?.review &&
282
361
  this.structuredReviewAttempts >
283
362
  (this.ctx.structuredOutput.maxReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT)) {
@@ -434,13 +513,14 @@ export class IterationOrchestrator {
434
513
  // Snapshot the cumulative counters so the step can report ITS
435
514
  // own usage rather than the run total.
436
515
  stepStartedAt = Date.now();
437
- usageBefore = { ...runMgr.tokenUsage };
438
- costBefore = { ...runMgr.costInfo };
439
516
  // Shape this step before calling the model. `stopWhen` decides
440
517
  // whether to keep going; this decides HOW. No-op when the host
441
518
  // supplied no hook.
442
519
  const contextModelBeforePreparation = this.ctx.contextModel ?? model;
443
520
  const step = await this.prepareStep(iterationNum);
521
+ // Preparation inference belongs to the run, not the main-model step.
522
+ usageBefore = { ...runMgr.tokenUsage };
523
+ costBefore = { ...runMgr.costInfo };
444
524
  stepModel = step.model ?? model;
445
525
  await this.selectContextModel(stepModel);
446
526
  // Preserve post-compaction preparation/recall semantics. A changed
@@ -465,7 +545,7 @@ export class IterationOrchestrator {
465
545
  const baseMessages = forceFinalize
466
546
  ? [
467
547
  ...runMgr.messages,
468
- createRuntimeContextMessage('[SYSTEM] You are approaching your resource limits. Provide your final, comprehensive response now based on everything you have gathered so far. Do not request any more tool calls.', 'limit-finalization'),
548
+ createRuntimeContextMessage(`[SYSTEM] You are approaching your resource limits. ${CLOSING_RESPONSE_GUIDANCE}`, 'limit-finalization'),
469
549
  ]
470
550
  : runMgr.messages;
471
551
  // Step guidance is appended to the REQUEST, never pushed onto
@@ -481,9 +561,9 @@ export class IterationOrchestrator {
481
561
  // mutation, and per-iteration this is trivial next to the model
482
562
  // call it precedes.
483
563
  // A step's skills and its guidance ride the same ephemeral
484
- // trailing system message. Appending leaves the cached prefix
485
- // intact; rewriting the run's own prompt to carry a phase's
486
- // skills would invalidate it on every iteration.
564
+ // system message. A driver may move it before history; changing
565
+ // system guidance can therefore affect prefix caching. Observations
566
+ // that need no system authority use step.context below.
487
567
  // `renderSkillsSection` already answers null for an empty list, so
488
568
  // there is no length check here — a second guard for the same
489
569
  // case is one more thing to keep in agreement with the first.
@@ -504,10 +584,9 @@ export class IterationOrchestrator {
504
584
  ? `Approval policy changed from "${policyChange.from}" to "${policyChange.to}" (${policyChange.reason}). Tool calls from here on are reviewed under the new policy.`
505
585
  : null;
506
586
  // State that changed during the run, reported once per turn.
507
- // `turn` contributions land HERE and nowhere else: in the
508
- // system prompt they would be cached for the run or read as
509
- // a standing instruction, and either way the state they
510
- // exist to report goes stale silently.
587
+ // `turn` contributions are recomputed here, not fixed when the
588
+ // run's prompt is assembled. They retain system authority and
589
+ // may affect caching just like the other system contributions.
511
590
  const turnSections = this.ctx.promptContributions?.render('turn', {
512
591
  iteration: iterationNum,
513
592
  }) ?? [];
@@ -517,7 +596,10 @@ export class IterationOrchestrator {
517
596
  const requestHistory = stepPreamble
518
597
  ? [...baseMessages, createSystemMessage(stepPreamble)]
519
598
  : [...baseMessages];
599
+ if (step.context)
600
+ requestHistory.push(this.stepContextMessage(step.context));
520
601
  const messages = projectRequestRichContent(this.projectObservations(requestHistory), this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES);
602
+ this.appendWorkContext(messages, iterationNum, step);
521
603
  await this.reportUnsupportedToolResults(messages);
522
604
  yield* this.ctx.drainPending();
523
605
  // What the model is about to be ASKED, recorded when it
@@ -619,7 +701,11 @@ export class IterationOrchestrator {
619
701
  model: requestedMember.model ?? stepModel,
620
702
  chainIndex: requestedMember.index,
621
703
  };
622
- const { response, messageId } = yield* streamProviderTurn(this.ctx.provider, {
704
+ const operatorInputAtDispatch = this.latestUserMessage;
705
+ const latestReviewUserMessage = (this.ctx.reviewAnswer || this.ctx.structuredOutput?.review) && operatorInputAtDispatch
706
+ ? structuredClone(operatorInputAtDispatch)
707
+ : undefined;
708
+ const { response, messageId, requestMessages } = yield* streamProviderTurn(this.ctx.provider, {
623
709
  model: stepModel,
624
710
  ...(this.ctx.structuredOutput?.mode === 'native'
625
711
  ? {
@@ -659,8 +745,12 @@ export class IterationOrchestrator {
659
745
  signal: this.ctx.abortController.signal,
660
746
  }, this.ctx.emitEvent, this.ctx.drainPending, runMgr.id, iterationNum, forceFinalize, this.ctx.log, iterSpan, stepMessageId, {
661
747
  onAccepted: (identity) => this.acceptProviderRejectedImage(identity),
662
- });
748
+ }, Boolean(this.ctx.reviewAnswer || this.ctx.structuredOutput?.review || this.ctx.advisoryCtx));
663
749
  stepResponse = response;
750
+ const reviewRequest = {
751
+ ...(requestMessages ? { requestMessages } : {}),
752
+ ...(latestReviewUserMessage ? { latestUserMessage: latestReviewUserMessage } : {}),
753
+ };
664
754
  // Who answered THIS turn.
665
755
  //
666
756
  // The read is exact at this point and stays exact: a chain that
@@ -668,18 +758,9 @@ export class IterationOrchestrator {
668
758
  // request, so the member at the cursor when the stream ends is
669
759
  // the one whose bytes are in `response`.
670
760
  //
671
- // It is taken here rather than at `recordStep` several hundred
672
- // lines below, and the honest account of that is defence in
673
- // depth, not a defect it currently prevents. Moving it down
674
- // fails no test, because nothing between the two asks this
675
- // provider for anything: compaction and working memory run
676
- // BEFORE the turn, the advisory phase runs after the step is
677
- // already recorded, and the only thing in between is tool
678
- // execution. That is a fact about today's phase order, which a
679
- // later phase inserted here would change silently — and the
680
- // symptom would be a step attributed to a member that first
681
- // served the turn after it, which is the class of wrongness
682
- // this whole field exists to end.
761
+ // Capture before host review: its auxiliary inference can move
762
+ // the fallback cursor. Main-step usage and provenance must keep
763
+ // naming the provider that produced this candidate.
683
764
  const servedBy = (() => {
684
765
  const member = this.ctx.servingMember?.() ?? {
685
766
  index: 0,
@@ -780,8 +861,11 @@ export class IterationOrchestrator {
780
861
  ...(response.message.replayState !== undefined
781
862
  ? { replayState: response.message.replayState }
782
863
  : {}),
783
- });
864
+ }, response.message.textParts);
784
865
  runMgr.pushMessage(assistantMsg);
866
+ if (this.ctx.advisoryCtx && requestMessages) {
867
+ this.advisoryTurn = { iteration: iterationNum, requestMessages, response: assistantMsg };
868
+ }
785
869
  if (this.ctx.workingStateManager && this.ctx.compactionConfig && assistantMsg.content) {
786
870
  extractFromAssistantMessage(this.ctx.workingStateManager, assistantMsg.content, this.ctx.compactionConfig);
787
871
  }
@@ -887,7 +971,7 @@ export class IterationOrchestrator {
887
971
  const candidate = await parseNativeCandidate(this.ctx.structuredOutput.schema, response, this.ctx.abortController.signal);
888
972
  let outcome;
889
973
  if (candidate.success)
890
- outcome = await this.reviewStructuredOutput(candidate.value);
974
+ outcome = await this.reviewStructuredOutput(candidate.value, reviewRequest, stepModel);
891
975
  else {
892
976
  this.nativeStructuredAttempts++;
893
977
  runMgr.pushMessage(createRuntimeContextMessage('Return a complete JSON value matching the supplied response schema. Do not continue a partial JSON fragment.', 'structured-output'));
@@ -1018,7 +1102,7 @@ export class IterationOrchestrator {
1018
1102
  // judge: bounded attempts, feedback as a user message, and
1019
1103
  // a loud stop rather than a loop.
1020
1104
  if (!forceFinalize && this.ctx.reviewAnswer) {
1021
- const review = await this.reviewAnswer(response.message.content ?? '');
1105
+ const review = await this.reviewAnswer(response.message.content ?? '', reviewRequest, stepModel);
1022
1106
  if (this.ctx.abortController.signal.aborted) {
1023
1107
  runMgr.setStopReason('cancelled');
1024
1108
  runMgr.markCancelled();
@@ -1026,6 +1110,21 @@ export class IterationOrchestrator {
1026
1110
  }
1027
1111
  if (review && !review.accept) {
1028
1112
  const attempt = ++this.answerReviewAttempts;
1113
+ runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'));
1114
+ // Commit the consumed allowance with its feedback before another
1115
+ // request, including exhaustion. Compaction cannot reset this quota.
1116
+ const checkpoint = await this.ctx.checkpointMgr.create(runMgr, iterationNum);
1117
+ await this.ctx.emitEvent({
1118
+ type: 'checkpoint_created',
1119
+ runId: runMgr.id,
1120
+ checkpointId: checkpoint.id,
1121
+ iteration: iterationNum,
1122
+ });
1123
+ if (this.ctx.abortController.signal.aborted) {
1124
+ runMgr.setStopReason('cancelled');
1125
+ runMgr.markCancelled();
1126
+ break;
1127
+ }
1029
1128
  const limit = this.ctx.maxAnswerReviews ?? DEFAULT_ANSWER_REVIEW_LIMIT;
1030
1129
  if (attempt > limit) {
1031
1130
  this.ctx.log.warn('Answer rejected more times than the run allows', {
@@ -1041,7 +1140,6 @@ export class IterationOrchestrator {
1041
1140
  'namzu.retry.attempt': attempt,
1042
1141
  'namzu.runtime.limit': limit,
1043
1142
  });
1044
- runMgr.pushMessage(createRuntimeContextMessage(review.feedback, 'answer-review'));
1045
1143
  await this.ctx.emitEvent({
1046
1144
  type: 'iteration_completed',
1047
1145
  runId: runMgr.id,
@@ -1080,7 +1178,12 @@ export class IterationOrchestrator {
1080
1178
  yield* this.ctx.drainPending();
1081
1179
  continue;
1082
1180
  }
1083
- let closingStopReason;
1181
+ // A limit-requested summary bypasses prose review and further
1182
+ // work. Preserve that limit on settlement, even if the provider
1183
+ // reports a normal text completion and headroom still remains.
1184
+ let closingStopReason = forceFinalize
1185
+ ? guardResult.stopReason
1186
+ : undefined;
1084
1187
  if (!hasContent && !forceFinalize) {
1085
1188
  this.ctx.log.warn('Empty completion detected — requesting final summary', {
1086
1189
  [NAMZU.ITERATION]: iterationNum,
@@ -1150,7 +1253,7 @@ export class IterationOrchestrator {
1150
1253
  // run ends here rather than paying for another turn whose only
1151
1254
  // job would be to restate it — unless it shared its turn with
1152
1255
  // other calls, which relays instead. See the method.
1153
- const structuredOutcome = await this.captureStructuredOutput(reviewOutcome.results, response);
1256
+ const structuredOutcome = await this.captureStructuredOutput(reviewOutcome.results, response, reviewRequest, stepModel);
1154
1257
  if (structuredOutcome === 'retry' ||
1155
1258
  structuredOutcome === 'exhausted' ||
1156
1259
  structuredOutcome === 'cancelled') {
@@ -1174,10 +1277,6 @@ export class IterationOrchestrator {
1174
1277
  break;
1175
1278
  }
1176
1279
  if (structuredOutcome === 'accepted') {
1177
- this.ctx.log.info('Structured output produced — ending run', {
1178
- [NAMZU.RUN_ID]: runMgr.id,
1179
- [NAMZU.ITERATION]: iterationNum,
1180
- });
1181
1280
  await this.ctx.emitEvent({
1182
1281
  type: 'iteration_completed',
1183
1282
  runId: runMgr.id,
@@ -1190,6 +1289,22 @@ export class IterationOrchestrator {
1190
1289
  runMgr.markCancelled();
1191
1290
  break;
1192
1291
  }
1292
+ if (!forceFinalize) {
1293
+ const inbound = this.deliverInbound();
1294
+ // Tool-result steering may already have been delivered by
1295
+ // runToolReview. Its candidate still answers the older input.
1296
+ if (inbound > 0 || this.latestUserMessage !== operatorInputAtDispatch)
1297
+ continue;
1298
+ }
1299
+ if (this.ctx.abortController.signal.aborted) {
1300
+ runMgr.setStopReason('cancelled');
1301
+ runMgr.markCancelled();
1302
+ break;
1303
+ }
1304
+ this.ctx.log.info('Structured output produced — ending run', {
1305
+ [NAMZU.RUN_ID]: runMgr.id,
1306
+ [NAMZU.ITERATION]: iterationNum,
1307
+ });
1193
1308
  this.publishStructuredOutput();
1194
1309
  runMgr.setStopReason('end_turn');
1195
1310
  break;
@@ -1223,8 +1338,10 @@ export class IterationOrchestrator {
1223
1338
  // returned — which is what makes a terminal submit_answer tool
1224
1339
  // usable without discarding its output.
1225
1340
  if (await this.shouldStop()) {
1226
- // Outstanding delegated work outranks the host's stop
1227
- // predicate, exactly once.
1341
+ // Outstanding work outranks the host's stop predicate —
1342
+ // a delegated task the completion inbox is expecting, or
1343
+ // a background job the model told `wait_for_job` it is
1344
+ // waiting on.
1228
1345
  //
1229
1346
  // This is a precedence rule chosen here, not something
1230
1347
  // `stopWhen` implies — a stop predicate is a programmable
@@ -1233,13 +1350,20 @@ export class IterationOrchestrator {
1233
1350
  // tool or a captured structured output. Those decide the
1234
1351
  // result, so no turn follows and a hold would buy nothing.
1235
1352
  // This one only says "stop", and stopping one turn later
1236
- // with the worker's result in hand is a better reading of
1237
- // the host's intent than stopping now and discarding it.
1353
+ // with the result in hand is a better reading of the
1354
+ // host's intent than stopping now and discarding it.
1238
1355
  //
1239
- // Bounded: after the notification is delivered the inbox
1240
- // is drained, so the predicate fires again next turn with
1241
- // nothing pending and the run stops. Exactly one extra
1242
- // turn, and `maxIterations` bounds it regardless.
1356
+ // Bounded by what is left to deliver, not by a count.
1357
+ // Each delivery consumes what it delivered — the inbox is
1358
+ // drained, and a job exit's notice is taken with the
1359
+ // record of the exits it accounts for — so the predicate
1360
+ // is asked again next turn against whatever is still
1361
+ // outstanding. One task deferred it once; two awaited
1362
+ // jobs exiting a minute apart defer it twice, each time
1363
+ // for a turn the model spends on news it has not read.
1364
+ // `maxIterations` and the run's own deadline bound all of
1365
+ // it regardless, and a leg with nothing pending never
1366
+ // opens a hold at all.
1243
1367
  if (yield* this.holdForOutstandingWork(iterationNum, true)) {
1244
1368
  // Remember WHY the next turn exists, so the turn that
1245
1369
  // ends the run can name the host's decision instead of
@@ -1298,7 +1422,7 @@ export class IterationOrchestrator {
1298
1422
  // the next iteration so a message queued during THIS turn is
1299
1423
  // in the history the next request is built from.
1300
1424
  this.deliverInbound();
1301
- await runAdvisoryPhase(this.ctx, iterationNum, response);
1425
+ await runAdvisoryPhase(this.ctx, iterationNum, response, this.getAdvisoryTurnContext());
1302
1426
  if (this.ctx.pluginManager) {
1303
1427
  const hookResults = await this.ctx.pluginManager.executeHooks('iteration_end', {
1304
1428
  runId: runMgr.id,
@@ -1411,6 +1535,7 @@ export class IterationOrchestrator {
1411
1535
  // compaction means the prompt is irreducible, and looping on it
1412
1536
  // would burn the budget to arrive at the same error.
1413
1537
  if (!overflowRelieved &&
1538
+ !(err instanceof AnswerReviewFailure) &&
1414
1539
  classifyProviderError(err, this.ctx.provider.id).code === 'context_length_exceeded') {
1415
1540
  overflowRelieved = true;
1416
1541
  const shed = await relieveOverflow(this.ctx);
@@ -1436,6 +1561,7 @@ export class IterationOrchestrator {
1436
1561
  throw err;
1437
1562
  }
1438
1563
  finally {
1564
+ this.advisoryTurn = undefined;
1439
1565
  // The only place the iteration span ends. It used to be ended at each of
1440
1566
  // seventeen exits, which is a rule every future edit has to
1441
1567
  // remember; a generator abandoned by its consumer never reached
@@ -1449,27 +1575,60 @@ export class IterationOrchestrator {
1449
1575
  }
1450
1576
  }
1451
1577
  /**
1452
- * Hold the run open for a worker that has not finished, and deliver it.
1578
+ * Hold the run open for work that has not finished, and deliver it.
1453
1579
  *
1454
- * Returns whether a completion or operator message entered the transcript —
1455
- * the caller continues on `true`, so the model gets a turn to respond.
1456
- * That turn is the entire justification for waiting, which
1580
+ * Returns whether a completion, a job exit or an operator message entered
1581
+ * the transcript — the caller continues on `true`, so the model gets a turn
1582
+ * to respond. That turn is the entire justification for waiting, which
1457
1583
  * is why only the exits that can still take one call this.
1458
1584
  *
1459
- * Bounded by `settleGraceMs` and by `maxIterations`, so a worker that never
1460
- * finishes cannot keep the run open.
1585
+ * Two kinds of work qualify and they are raced together, because a run has
1586
+ * one settle point and one grace period to spend at it:
1587
+ *
1588
+ * - a delegated task the `CompletionInbox` is still expecting;
1589
+ * - a background job the model told `wait_for_job` it is waiting on.
1590
+ *
1591
+ * The job half is deliberately narrow. Intent comes from the wait and from
1592
+ * nothing else — a dev server the model started and never waited on is
1593
+ * running because somebody wanted it running, and a hold for it would add
1594
+ * the grace period to the end of every turn for the rest of the session.
1595
+ *
1596
+ * Each leg is opened only when it has something pending: both
1597
+ * `waitForArrival` implementations resolve immediately when their own side
1598
+ * is idle, so racing an idle one would end the hold before it began.
1599
+ *
1600
+ * Bounded by `settleGraceMs` and by `maxIterations`, so work that never
1601
+ * finishes cannot keep the run open. On a run with a deadline the grace is
1602
+ * a share of what is LEFT of it rather than a fresh allowance, so a
1603
+ * `wait_for_job` call that already spent minutes has shortened this hold
1604
+ * by the same minutes. On a run without one — the CLI's default — there is
1605
+ * no remainder to take a share of, and the job leg's own ceiling
1606
+ * (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
1607
+ * by an hour of silence.
1461
1608
  */
1462
1609
  async *holdForOutstandingWork(iterationNum, hasToolCalls) {
1463
- if (!this.ctx.completionInbox?.hasPendingWork)
1610
+ const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined;
1611
+ const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined;
1612
+ if (!inbox && !jobs)
1464
1613
  return false;
1465
1614
  // Read HERE rather than from `forceFinalize`, which was sampled at the
1466
1615
  // top of the iteration: one that has since crossed the finalize point
1467
1616
  // must not open a wait against a reserve it has already entered.
1468
- const graceMs = settleGraceMs(this.ctx.guard.remainingBeforeFinalizeMs());
1469
- this.ctx.log.info('Holding the run open for a background task', {
1617
+ const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs();
1618
+ // One deadline for the race, and it is the LONGEST ceiling any pending
1619
+ // leg justifies. A leg resolving on its own timer ends the whole race,
1620
+ // so handing the job leg its shorter ceiling while a task was also
1621
+ // outstanding would cut the task's hold down to the job's — a run
1622
+ // walking away from a worker it had time for, because a job happened
1623
+ // to be running. A job therefore never shortens a wait, and it never
1624
+ // lengthens one either: where a task is outstanding too, that is how
1625
+ // long this run was waiting anyway.
1626
+ const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs);
1627
+ this.ctx.log.info('Holding the run open for outstanding work', {
1470
1628
  [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1471
1629
  [NAMZU.ITERATION]: iterationNum,
1472
1630
  'namzu.runtime.grace_ms': graceMs,
1631
+ 'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
1473
1632
  });
1474
1633
  // User input releases this wait without cancelling any child. Both waits
1475
1634
  // share a disposable signal so the losing arrival listener cannot leak.
@@ -1481,7 +1640,8 @@ export class IterationOrchestrator {
1481
1640
  cancelWait();
1482
1641
  try {
1483
1642
  await Promise.race([
1484
- this.ctx.completionInbox.waitForArrival(graceMs, waiting.signal),
1643
+ ...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
1644
+ ...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
1485
1645
  ...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
1486
1646
  ]);
1487
1647
  }
@@ -1494,12 +1654,13 @@ export class IterationOrchestrator {
1494
1654
  runSignal.removeEventListener('abort', cancelWait);
1495
1655
  }
1496
1656
  runSignal.throwIfAborted();
1497
- const arrived = this.ctx.completionInbox.drain();
1657
+ const arrived = this.ctx.completionInbox?.drain() ?? [];
1498
1658
  if (arrived.length > 0) {
1499
1659
  this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'));
1500
1660
  }
1661
+ const exited = this.deliverAwaitedJobExits();
1501
1662
  const inbound = this.deliverInbound();
1502
- if (arrived.length === 0 && inbound === 0)
1663
+ if (arrived.length === 0 && !exited && inbound === 0)
1503
1664
  return false;
1504
1665
  await this.ctx.emitEvent({
1505
1666
  type: 'iteration_completed',
@@ -1511,8 +1672,45 @@ export class IterationOrchestrator {
1511
1672
  return true;
1512
1673
  }
1513
1674
  /**
1514
- * Account for delegated work on the way out: deliver what arrived, and say
1515
- * what did not.
1675
+ * Put the job exits this hold was waiting for in front of the model.
1676
+ *
1677
+ * Through `jobNotices`, which is the channel a job exit already travels on
1678
+ * — `attachNotice` rides it out on the next tool result — rather than a
1679
+ * second one built for this path. A turn that called no tools has no such
1680
+ * result, so the queued text becomes a `runtime-context` message instead,
1681
+ * exactly as `deliverInbound` does for steering that found no tool result
1682
+ * to attach to.
1683
+ *
1684
+ * That drain is also what keeps one exit from being delivered twice: the
1685
+ * channel hands its text over once, so an exit already attached to a tool
1686
+ * result earlier in the turn leaves nothing here — and the record of it
1687
+ * went with that delivery, so this returns `false` rather than buying a
1688
+ * turn to re-read what the model has read.
1689
+ *
1690
+ * `takeDelivery` is what pairs the two. Taking the exits first and then
1691
+ * finding no notice would discard them, which is the one way this path
1692
+ * can lose an exit outright; neither is taken unless both are there.
1693
+ *
1694
+ * The channel is not per-job, so the text taken here can include a notice
1695
+ * for a job nobody awaited that ended while the hold was open. Delivering
1696
+ * it is right — it is unread either way, and the alternative is stranding
1697
+ * it — but it is not a reason to WAIT, which is why what opens this hold
1698
+ * is `AwaitedJobs`, and the two are asked separately.
1699
+ */
1700
+ deliverAwaitedJobExits() {
1701
+ const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain());
1702
+ if (!delivered)
1703
+ return false;
1704
+ this.ctx.log.info('Delivering a background job exit the run held open for', {
1705
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1706
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1707
+ });
1708
+ this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
1709
+ return true;
1710
+ }
1711
+ /**
1712
+ * Account for outstanding work on the way out: deliver what arrived, and
1713
+ * say what did not.
1516
1714
  *
1517
1715
  * A run that ends with a worker outstanding must not leave the impression
1518
1716
  * that the worker's result was delivered. There are exactly two honest
@@ -1536,18 +1734,31 @@ export class IterationOrchestrator {
1536
1734
  */
1537
1735
  settleOutstandingWork() {
1538
1736
  this.deliverArrivedCompletions();
1737
+ this.deliverArrivedJobExits();
1539
1738
  this.recordAbandonedWork();
1540
1739
  }
1541
- /** Delegated work this run walked away from. See {@link settleOutstandingWork}. */
1740
+ /** Work this run walked away from. See {@link settleOutstandingWork}. */
1542
1741
  recordAbandonedWork() {
1543
1742
  const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? [];
1544
- if (abandoned.length === 0)
1743
+ if (abandoned.length > 0) {
1744
+ this.ctx.log.warn('Run ended with delegated work still running', {
1745
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1746
+ 'namzu.runtime.tasks': abandoned,
1747
+ });
1748
+ this.ctx.runMgr.setAbandonedTaskIds(abandoned);
1749
+ }
1750
+ // The same statement for a job the model was waiting on when the grace
1751
+ // ran out. Only awaited ones: a job nobody waited for was never work
1752
+ // this run was holding, so naming it would report an abandonment that
1753
+ // did not happen.
1754
+ const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? [];
1755
+ if (abandonedJobs.length === 0)
1545
1756
  return;
1546
- this.ctx.log.warn('Run ended with delegated work still running', {
1757
+ this.ctx.log.warn('Run ended with an awaited background job still running', {
1547
1758
  [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1548
- 'namzu.runtime.tasks': abandoned,
1759
+ 'namzu.runtime.jobs': abandonedJobs,
1549
1760
  });
1550
- this.ctx.runMgr.setAbandonedTaskIds(abandoned);
1761
+ this.ctx.runMgr.setAbandonedJobIds(abandonedJobs);
1551
1762
  }
1552
1763
  deliverArrivedCompletions() {
1553
1764
  const unheard = this.ctx.completionInbox?.drain() ?? [];
@@ -1579,24 +1790,65 @@ export class IterationOrchestrator {
1579
1790
  this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'));
1580
1791
  }
1581
1792
  /**
1582
- * Ask the host how to shape this step.
1793
+ * The job half of {@link deliverArrivedCompletions}: an exit that arrived
1794
+ * too late to earn a turn is still delivered on the way out.
1583
1795
  *
1584
- * Fails OPEN on a throw — same reasoning as `stopWhen` and deliberately
1585
- * opposite to a guardrail: a broken step-shaping hook should not kill an
1586
- * otherwise healthy run, and unlike a safety check, nothing unsafe gets
1587
- * through when it is skipped.
1588
- */
1589
- /**
1590
- * A host's chance to refuse the next model call.
1796
+ * The window this closes is one tick wide and it is nobody else's. An
1797
+ * awaited job that exits between the hold's grace expiring and the run
1798
+ * settling was never delivered — the hold had already looked — and is no
1799
+ * longer named either, because the exit took it off the outstanding list
1800
+ * on its way past, so `abandonedJobIds` would be lying to claim it. The
1801
+ * host's own listener is no help: the CLI queues an exit for the next
1802
+ * turn only when no run is in flight, and this one is still in flight.
1803
+ * Delivered here it reaches `Run.messages`, so the transcript has it and
1804
+ * a continued thread opens with it.
1591
1805
  *
1592
- * Fails CLOSED, which is the opposite of `prepareStep` below and the
1593
- * reason they are separate hooks rather than one with two return
1594
- * shapes. A broken step-SHAPER skipped costs a run its per-step tuning;
1595
- * a broken step-REFUSER skipped is a refusal that did not happen, which
1596
- * is precisely what the hook exists to prevent. The thrown error's
1597
- * message becomes the reason, so an operator is not left with a run
1598
- * that stopped and no account of it.
1806
+ * Before `recordAbandonedWork`, which then reports only what is still
1807
+ * running, and after `deliverArrivedCompletions`, so the two appended
1808
+ * messages land in the order the work finished in.
1599
1809
  */
1810
+ deliverArrivedJobExits() {
1811
+ const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain());
1812
+ if (!delivered)
1813
+ return;
1814
+ // Fix the run's answer BEFORE appending anything after it — the same
1815
+ // `resolveResult` tail walk `deliverArrivedCompletions` explains just
1816
+ // above, and the same guard against pinning an empty one.
1817
+ const answer = this.ctx.runMgr.materializeResult();
1818
+ if (answer.length > 0)
1819
+ this.ctx.runMgr.setResult(answer);
1820
+ this.ctx.log.info('Delivering a background job exit the run would have settled over', {
1821
+ [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1822
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1823
+ });
1824
+ this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
1825
+ }
1826
+ stepContextMessage(content) {
1827
+ return createRuntimeContextMessage(`Current step context (runtime-generated; not a new user request):\n${content}`, 'step-context');
1828
+ }
1829
+ /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
1830
+ appendWorkContext(messages, stepNumber, prepared) {
1831
+ const contributions = [
1832
+ this.ctx.completionInbox?.describeOwnedWork(),
1833
+ this.ctx.toolExecutor.describeFileEvidence(messages),
1834
+ ].filter((content) => Boolean(content));
1835
+ if (contributions.length === 0)
1836
+ return;
1837
+ let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0;
1838
+ // Leave room for the actual task; admit whole contributions, never dangling partial references.
1839
+ if (room < 1_500)
1840
+ return;
1841
+ for (const content of contributions) {
1842
+ if (!content || content.length > 8_000)
1843
+ continue;
1844
+ const message = this.stepContextMessage(content);
1845
+ const tokens = estimateMessageTokens(message);
1846
+ if (tokens > Math.min(2_000, room - 1_000))
1847
+ continue;
1848
+ messages.push(message);
1849
+ room -= tokens;
1850
+ }
1851
+ }
1600
1852
  stepContext(stepNumber, prepared) {
1601
1853
  const model = prepared.model ?? this.ctx.runConfig.model;
1602
1854
  const window = resolveContextWindow(this.ctx.compactionConfig?.contextWindowTokens, model, model === this.ctx.runConfig.model
@@ -1606,7 +1858,8 @@ export class IterationOrchestrator {
1606
1858
  : undefined);
1607
1859
  const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null;
1608
1860
  const preamble = [prepared.system, skills].filter(Boolean).join('\n\n');
1609
- const preparedTokens = preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0;
1861
+ const preparedTokens = (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
1862
+ (prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0);
1610
1863
  const responseReserve = Math.min(prepared.maxResponseTokens ??
1611
1864
  this.ctx.runConfig.maxResponseTokens ??
1612
1865
  Math.floor(window.tokens / 4), Math.floor(window.tokens / 4));
@@ -1614,6 +1867,7 @@ export class IterationOrchestrator {
1614
1867
  runId: this.ctx.runMgr.id,
1615
1868
  stepNumber,
1616
1869
  messages: this.ctx.runMgr.messages,
1870
+ ...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
1617
1871
  ...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
1618
1872
  signal: this.ctx.abortController.signal,
1619
1873
  contextBudget: {
@@ -1624,6 +1878,7 @@ export class IterationOrchestrator {
1624
1878
  prepared,
1625
1879
  };
1626
1880
  }
1881
+ /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
1627
1882
  async beforeStep(stepNumber) {
1628
1883
  const configured = this.ctx.beforeStep;
1629
1884
  if (!configured)
@@ -1635,6 +1890,7 @@ export class IterationOrchestrator {
1635
1890
  return { reason: `beforeStep threw: ${toErrorMessage(err)}` };
1636
1891
  }
1637
1892
  }
1893
+ /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
1638
1894
  async prepareStep(stepNumber) {
1639
1895
  const configured = this.ctx.prepareStep;
1640
1896
  if (!configured)
@@ -1646,8 +1902,12 @@ export class IterationOrchestrator {
1646
1902
  // rather than an accident of install history.
1647
1903
  let result = {};
1648
1904
  for (const stage of stages) {
1905
+ const inference = createCallbackInference(this.ctx, result.model ?? this.ctx.runConfig.model, 'preparation');
1649
1906
  try {
1650
- const decided = await stage(this.stepContext(stepNumber, result));
1907
+ const decided = await stage({
1908
+ ...this.stepContext(stepNumber, result),
1909
+ generateText: inference.generateText,
1910
+ });
1651
1911
  if (decided)
1652
1912
  result = { ...result, ...decided };
1653
1913
  await this.selectContextModel(result.model ?? this.ctx.runConfig.model);
@@ -1660,6 +1920,22 @@ export class IterationOrchestrator {
1660
1920
  'namzu.runtime.step_number': stepNumber,
1661
1921
  'exception.message': toErrorMessage(err),
1662
1922
  });
1923
+ // An SDK stage may report availability and validated fallback evidence
1924
+ // without exposing its error. Preserve prior decisions and the context budget;
1925
+ // ordinary exceptions still contribute nothing to the model request.
1926
+ if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
1927
+ const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0;
1928
+ if (typeof err.context === 'string' &&
1929
+ err.context.length > 0 &&
1930
+ err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room)))
1931
+ result = {
1932
+ ...result,
1933
+ context: [result.context, err.context].filter(Boolean).join('\n\n'),
1934
+ };
1935
+ }
1936
+ }
1937
+ finally {
1938
+ inference.close();
1663
1939
  }
1664
1940
  }
1665
1941
  const prepared = {};
@@ -1696,6 +1972,8 @@ export class IterationOrchestrator {
1696
1972
  prepared.model = result.model;
1697
1973
  if (result.system !== undefined)
1698
1974
  prepared.system = result.system;
1975
+ if (result.context !== undefined)
1976
+ prepared.context = result.context;
1699
1977
  if (result.skills !== undefined)
1700
1978
  prepared.skills = result.skills;
1701
1979
  if (result.temperature !== undefined)
@@ -1937,7 +2215,7 @@ export class IterationOrchestrator {
1937
2215
  * doing work, and it is the bound the neighbour relies on for the
1938
2216
  * identical pathology.
1939
2217
  */
1940
- async captureStructuredOutput(results, response) {
2218
+ async captureStructuredOutput(results, response, reviewRequest, model) {
1941
2219
  if (!this.needsStructuredOutput() || this.ctx.structuredOutput?.mode === 'native')
1942
2220
  return 'absent';
1943
2221
  const hit = results.find((r) => r.toolName === STRUCTURED_OUTPUT_TOOL_NAME && !r.isError);
@@ -1962,13 +2240,13 @@ export class IterationOrchestrator {
1962
2240
  throw new Error('Structured review requires an intact JSON tool result; check tool-output limits and result transformations');
1963
2241
  parsed = hit.output;
1964
2242
  }
1965
- return this.reviewStructuredOutput(parsed);
2243
+ return this.reviewStructuredOutput(parsed, reviewRequest, model);
1966
2244
  }
1967
2245
  publishStructuredOutput() {
1968
2246
  this.ctx.runMgr.setStructuredOutput(this.pendingStructuredOutput);
1969
2247
  this.structuredOutputDone = true;
1970
2248
  }
1971
- async reviewStructuredOutput(parsed) {
2249
+ async reviewStructuredOutput(parsed, reviewRequest, model) {
1972
2250
  if (this.ctx.abortController.signal.aborted)
1973
2251
  return 'cancelled';
1974
2252
  const reviewer = this.ctx.structuredOutput?.review;
@@ -1982,14 +2260,20 @@ export class IterationOrchestrator {
1982
2260
  signal.addEventListener('abort', onAbort, { once: true });
1983
2261
  });
1984
2262
  let verdict;
2263
+ const inference = createCallbackInference(this.ctx, model, 'review');
1985
2264
  try {
1986
2265
  verdict = await Promise.race([
1987
- Promise.resolve().then(() => reviewer(structuredClone(parsed), {
1988
- runId: this.ctx.runMgr.id,
1989
- iteration: this.ctx.runMgr.currentIteration,
1990
- signal,
1991
- messages: this.ctx.runMgr.messages,
1992
- })),
2266
+ Promise.resolve().then(() => {
2267
+ signal.throwIfAborted();
2268
+ return reviewer(structuredClone(parsed), {
2269
+ runId: this.ctx.runMgr.id,
2270
+ iteration: this.ctx.runMgr.currentIteration,
2271
+ signal,
2272
+ messages: this.ctx.runMgr.messages,
2273
+ ...reviewRequest,
2274
+ generateText: inference.generateText,
2275
+ });
2276
+ }),
1993
2277
  aborted,
1994
2278
  ]);
1995
2279
  }
@@ -1999,6 +2283,7 @@ export class IterationOrchestrator {
1999
2283
  throw error;
2000
2284
  }
2001
2285
  finally {
2286
+ inference.close();
2002
2287
  signal.removeEventListener('abort', onAbort);
2003
2288
  }
2004
2289
  if (signal.aborted)
@@ -2030,35 +2315,49 @@ export class IterationOrchestrator {
2030
2315
  this.pendingStructuredOutput = parsed;
2031
2316
  return 'accepted';
2032
2317
  }
2033
- /**
2034
- * Ask the host whether this answer is good enough.
2035
- *
2036
- * A hook that throws **accepts**, which is the opposite of what the
2037
- * safety gates do, and deliberately so. Those are asked "is this
2038
- * dangerous", where the cost of failing closed is one refused
2039
- * operation. This is asked "is this good enough", where failing closed
2040
- * means handing the answer back forever — so a broken judge would turn
2041
- * every run into a loop that ends on a budget error naming nothing. One
2042
- * unreviewed answer is the cheaper failure, and the throw is logged at
2043
- * `error` so it is not mistaken for approval.
2044
- */
2045
- async reviewAnswer(answer) {
2046
- if (!this.ctx.reviewAnswer)
2318
+ /** A reviewer failure aborts settlement; only an explicit rejection requests correction. */
2319
+ async reviewAnswer(answer, reviewRequest, model) {
2320
+ const reviewer = this.ctx.reviewAnswer;
2321
+ const signal = this.ctx.abortController.signal;
2322
+ if (!reviewer || signal.aborted)
2047
2323
  return undefined;
2324
+ let onAbort = () => { };
2325
+ const aborted = new Promise((_resolve, reject) => {
2326
+ onAbort = () => reject(signal.reason ?? new Error('Answer review cancelled'));
2327
+ signal.addEventListener('abort', onAbort, { once: true });
2328
+ });
2329
+ const inference = createCallbackInference(this.ctx, model, 'review');
2048
2330
  try {
2049
- return await this.ctx.reviewAnswer(answer, {
2050
- runId: this.ctx.runMgr.id,
2051
- iteration: this.ctx.runMgr.currentIteration,
2052
- signal: this.ctx.abortController.signal,
2053
- messages: this.ctx.runMgr.messages,
2054
- });
2331
+ const verdict = await Promise.race([
2332
+ Promise.resolve().then(() => {
2333
+ signal.throwIfAborted();
2334
+ return reviewer(answer, {
2335
+ runId: this.ctx.runMgr.id,
2336
+ iteration: this.ctx.runMgr.currentIteration,
2337
+ signal,
2338
+ messages: this.ctx.runMgr.messages,
2339
+ ...reviewRequest,
2340
+ generateText: inference.generateText,
2341
+ });
2342
+ }),
2343
+ aborted,
2344
+ ]);
2345
+ if (signal.aborted)
2346
+ return undefined;
2347
+ if (!verdict || typeof verdict.accept !== 'boolean')
2348
+ throw new Error('Answer reviewer returned an invalid verdict');
2349
+ if (!verdict.accept && (typeof verdict.feedback !== 'string' || !verdict.feedback.trim()))
2350
+ throw new Error('Answer reviewer rejection requires feedback');
2351
+ return verdict;
2055
2352
  }
2056
- catch (err) {
2057
- this.ctx.log.error('Answer review threw — accepting the answer unreviewed', {
2058
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2059
- 'exception.message': toErrorMessage(err),
2060
- });
2061
- return { accept: true };
2353
+ catch (error) {
2354
+ if (signal.aborted)
2355
+ return undefined;
2356
+ throw new AnswerReviewFailure(error);
2357
+ }
2358
+ finally {
2359
+ inference.close();
2360
+ signal.removeEventListener('abort', onAbort);
2062
2361
  }
2063
2362
  }
2064
2363
  /** Evaluate the caller's halt predicate, if there is one. */
@@ -2131,9 +2430,10 @@ export class IterationOrchestrator {
2131
2430
  try {
2132
2431
  const finalHistory = [
2133
2432
  ...this.ctx.runMgr.messages,
2134
- createRuntimeContextMessage(`[SYSTEM] Run is ending due to ${reason}. You MUST provide a final response now summarizing all your findings and work so far. Do not use any tools.`, 'limit-finalization'),
2433
+ createRuntimeContextMessage(`[SYSTEM] Run is ending due to ${reason}. ${CLOSING_RESPONSE_GUIDANCE}`, 'limit-finalization'),
2135
2434
  ];
2136
2435
  const finalMessages = projectRequestRichContent(this.projectObservations(finalHistory), this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES);
2436
+ this.appendWorkContext(finalMessages, this.steps.length + 1, { model });
2137
2437
  await this.reportUnsupportedToolResults(finalMessages);
2138
2438
  // Same cache discipline as the forced-final iteration: keep the
2139
2439
  // tools param identical to prior iterations (cache prefix intact,
@@ -2185,7 +2485,7 @@ export class IterationOrchestrator {
2185
2485
  ...(response.message.replayState !== undefined
2186
2486
  ? { replayState: response.message.replayState }
2187
2487
  : {}),
2188
- });
2488
+ }, response.message.textParts);
2189
2489
  this.ctx.runMgr.pushMessage(assistantMsg);
2190
2490
  const finalMessageId = generateMessageId();
2191
2491
  await this.ctx.emitEvent({
@@ -2202,6 +2502,7 @@ export class IterationOrchestrator {
2202
2502
  stopReason: 'forced_finalize',
2203
2503
  usage: response.usage,
2204
2504
  content: response.message.content ?? undefined,
2505
+ ...(response.message.textParts ? { textParts: response.message.textParts } : {}),
2205
2506
  });
2206
2507
  }
2207
2508
  catch (err) {
@@ -2223,6 +2524,8 @@ export class IterationOrchestrator {
2223
2524
  * threw and is left saying so rather than dressed up as something specific.
2224
2525
  */
2225
2526
  function describeStepFailure(err, providerId) {
2527
+ if (err instanceof AnswerReviewFailure)
2528
+ return { message: err.message, code: 'unknown', retryable: false };
2226
2529
  const classified = classifyProviderError(err, providerId);
2227
2530
  return {
2228
2531
  message: toErrorMessage(err),