@namzu/sdk 3.1.0 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (257) hide show
  1. package/CHANGELOG.md +177 -0
  2. package/dist/advisory/__tests__/consultation-context.test.d.ts +2 -0
  3. package/dist/advisory/__tests__/consultation-context.test.d.ts.map +1 -0
  4. package/dist/advisory/__tests__/consultation-context.test.js +124 -0
  5. package/dist/advisory/__tests__/consultation-context.test.js.map +1 -0
  6. package/dist/advisory/context.d.ts +25 -0
  7. package/dist/advisory/context.d.ts.map +1 -1
  8. package/dist/advisory/context.js +18 -0
  9. package/dist/advisory/context.js.map +1 -1
  10. package/dist/advisory/executor.d.ts.map +1 -1
  11. package/dist/advisory/executor.js +25 -3
  12. package/dist/advisory/executor.js.map +1 -1
  13. package/dist/compaction/__tests__/context-reducer.test.d.ts +2 -0
  14. package/dist/compaction/__tests__/context-reducer.test.d.ts.map +1 -0
  15. package/dist/compaction/__tests__/context-reducer.test.js +197 -0
  16. package/dist/compaction/__tests__/context-reducer.test.js.map +1 -0
  17. package/dist/compaction/factory.d.ts +7 -0
  18. package/dist/compaction/factory.d.ts.map +1 -1
  19. package/dist/compaction/factory.js +7 -0
  20. package/dist/compaction/factory.js.map +1 -1
  21. package/dist/compaction/index.d.ts +2 -0
  22. package/dist/compaction/index.d.ts.map +1 -1
  23. package/dist/compaction/index.js +1 -0
  24. package/dist/compaction/index.js.map +1 -1
  25. package/dist/compaction/interface.d.ts +13 -0
  26. package/dist/compaction/interface.d.ts.map +1 -1
  27. package/dist/compaction/managers/null.d.ts +3 -0
  28. package/dist/compaction/managers/null.d.ts.map +1 -1
  29. package/dist/compaction/managers/null.js +3 -0
  30. package/dist/compaction/managers/null.js.map +1 -1
  31. package/dist/compaction/managers/slidingWindow.d.ts +6 -0
  32. package/dist/compaction/managers/slidingWindow.d.ts.map +1 -1
  33. package/dist/compaction/managers/slidingWindow.js +6 -0
  34. package/dist/compaction/managers/slidingWindow.js.map +1 -1
  35. package/dist/compaction/managers/structured.d.ts +10 -0
  36. package/dist/compaction/managers/structured.d.ts.map +1 -1
  37. package/dist/compaction/managers/structured.js +10 -0
  38. package/dist/compaction/managers/structured.js.map +1 -1
  39. package/dist/compaction/reducer.d.ts +86 -0
  40. package/dist/compaction/reducer.d.ts.map +1 -0
  41. package/dist/compaction/reducer.js +77 -0
  42. package/dist/compaction/reducer.js.map +1 -0
  43. package/dist/connector/builtins/__tests__/oauth2-auth.test.d.ts +2 -0
  44. package/dist/connector/builtins/__tests__/oauth2-auth.test.d.ts.map +1 -0
  45. package/dist/connector/builtins/__tests__/oauth2-auth.test.js +54 -0
  46. package/dist/connector/builtins/__tests__/oauth2-auth.test.js.map +1 -0
  47. package/dist/connector/builtins/http.d.ts.map +1 -1
  48. package/dist/connector/builtins/http.js +24 -2
  49. package/dist/connector/builtins/http.js.map +1 -1
  50. package/dist/connector/builtins/http.test.js +18 -2
  51. package/dist/connector/builtins/http.test.js.map +1 -1
  52. package/dist/connector/index.d.ts +2 -2
  53. package/dist/connector/index.d.ts.map +1 -1
  54. package/dist/connector/index.js +1 -1
  55. package/dist/connector/index.js.map +1 -1
  56. package/dist/connector/mcp/__tests__/prompts-and-lifecycle.test.d.ts +2 -0
  57. package/dist/connector/mcp/__tests__/prompts-and-lifecycle.test.d.ts.map +1 -0
  58. package/dist/connector/mcp/__tests__/prompts-and-lifecycle.test.js +214 -0
  59. package/dist/connector/mcp/__tests__/prompts-and-lifecycle.test.js.map +1 -0
  60. package/dist/connector/mcp/client.d.ts +52 -1
  61. package/dist/connector/mcp/client.d.ts.map +1 -1
  62. package/dist/connector/mcp/client.js +86 -0
  63. package/dist/connector/mcp/client.js.map +1 -1
  64. package/dist/connector/mcp/discovery.d.ts +12 -1
  65. package/dist/connector/mcp/discovery.d.ts.map +1 -1
  66. package/dist/connector/mcp/discovery.js +19 -4
  67. package/dist/connector/mcp/discovery.js.map +1 -1
  68. package/dist/connector/mcp/index.d.ts +2 -2
  69. package/dist/connector/mcp/index.d.ts.map +1 -1
  70. package/dist/connector/mcp/index.js +1 -1
  71. package/dist/connector/mcp/index.js.map +1 -1
  72. package/dist/connector/mcp/server.d.ts +42 -1
  73. package/dist/connector/mcp/server.d.ts.map +1 -1
  74. package/dist/connector/mcp/server.js +77 -4
  75. package/dist/connector/mcp/server.js.map +1 -1
  76. package/dist/manager/agent/__tests__/depth-limit-authority.test.d.ts +2 -0
  77. package/dist/manager/agent/__tests__/depth-limit-authority.test.d.ts.map +1 -0
  78. package/dist/manager/agent/__tests__/depth-limit-authority.test.js +58 -0
  79. package/dist/manager/agent/__tests__/depth-limit-authority.test.js.map +1 -0
  80. package/dist/plugin/__tests__/discovery-scopes.test.d.ts +2 -0
  81. package/dist/plugin/__tests__/discovery-scopes.test.d.ts.map +1 -0
  82. package/dist/plugin/__tests__/discovery-scopes.test.js +97 -0
  83. package/dist/plugin/__tests__/discovery-scopes.test.js.map +1 -0
  84. package/dist/plugin/__tests__/enable-contributions.test.js +5 -1
  85. package/dist/plugin/__tests__/enable-contributions.test.js.map +1 -1
  86. package/dist/plugin/__tests__/mcp-admission.test.d.ts +2 -0
  87. package/dist/plugin/__tests__/mcp-admission.test.d.ts.map +1 -0
  88. package/dist/plugin/__tests__/mcp-admission.test.js +192 -0
  89. package/dist/plugin/__tests__/mcp-admission.test.js.map +1 -0
  90. package/dist/plugin/lifecycle.d.ts +41 -0
  91. package/dist/plugin/lifecycle.d.ts.map +1 -1
  92. package/dist/plugin/lifecycle.js +29 -1
  93. package/dist/plugin/lifecycle.js.map +1 -1
  94. package/dist/plugin/loader.d.ts +39 -3
  95. package/dist/plugin/loader.d.ts.map +1 -1
  96. package/dist/plugin/loader.js +37 -4
  97. package/dist/plugin/loader.js.map +1 -1
  98. package/dist/public-runtime.d.ts +4 -2
  99. package/dist/public-runtime.d.ts.map +1 -1
  100. package/dist/public-runtime.js +5 -2
  101. package/dist/public-runtime.js.map +1 -1
  102. package/dist/public-types.d.ts +2 -2
  103. package/dist/public-types.d.ts.map +1 -1
  104. package/dist/rag/__tests__/namespace-isolation.test.d.ts +2 -0
  105. package/dist/rag/__tests__/namespace-isolation.test.d.ts.map +1 -0
  106. package/dist/rag/__tests__/namespace-isolation.test.js +80 -0
  107. package/dist/rag/__tests__/namespace-isolation.test.js.map +1 -0
  108. package/dist/rag/ingestion.d.ts.map +1 -1
  109. package/dist/rag/ingestion.js +1 -0
  110. package/dist/rag/ingestion.js.map +1 -1
  111. package/dist/rag/retriever.d.ts.map +1 -1
  112. package/dist/rag/retriever.js +2 -0
  113. package/dist/rag/retriever.js.map +1 -1
  114. package/dist/rag/vector-store.d.ts.map +1 -1
  115. package/dist/rag/vector-store.js +6 -0
  116. package/dist/rag/vector-store.js.map +1 -1
  117. package/dist/registry/tool/execute.d.ts.map +1 -1
  118. package/dist/registry/tool/execute.js +113 -109
  119. package/dist/registry/tool/execute.js.map +1 -1
  120. package/dist/runtime/query/__tests__/per-step-skills.test.d.ts +10 -0
  121. package/dist/runtime/query/__tests__/per-step-skills.test.d.ts.map +1 -0
  122. package/dist/runtime/query/__tests__/per-step-skills.test.js +122 -0
  123. package/dist/runtime/query/__tests__/per-step-skills.test.js.map +1 -0
  124. package/dist/runtime/query/__tests__/per-step-tool-choice.test.d.ts +2 -0
  125. package/dist/runtime/query/__tests__/per-step-tool-choice.test.d.ts.map +1 -0
  126. package/dist/runtime/query/__tests__/per-step-tool-choice.test.js +153 -0
  127. package/dist/runtime/query/__tests__/per-step-tool-choice.test.js.map +1 -0
  128. package/dist/runtime/query/__tests__/resume-run.test.d.ts +2 -0
  129. package/dist/runtime/query/__tests__/resume-run.test.d.ts.map +1 -0
  130. package/dist/runtime/query/__tests__/resume-run.test.js +211 -0
  131. package/dist/runtime/query/__tests__/resume-run.test.js.map +1 -0
  132. package/dist/runtime/query/index.d.ts +10 -0
  133. package/dist/runtime/query/index.d.ts.map +1 -1
  134. package/dist/runtime/query/index.js +26 -0
  135. package/dist/runtime/query/index.js.map +1 -1
  136. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  137. package/dist/runtime/query/iteration/index.js +96 -34
  138. package/dist/runtime/query/iteration/index.js.map +1 -1
  139. package/dist/runtime/query/iteration/phases/compaction-model-routing.test.d.ts +2 -0
  140. package/dist/runtime/query/iteration/phases/compaction-model-routing.test.d.ts.map +1 -0
  141. package/dist/runtime/query/iteration/phases/compaction-model-routing.test.js +96 -0
  142. package/dist/runtime/query/iteration/phases/compaction-model-routing.test.js.map +1 -0
  143. package/dist/runtime/query/iteration/phases/compaction.d.ts.map +1 -1
  144. package/dist/runtime/query/iteration/phases/compaction.js +95 -5
  145. package/dist/runtime/query/iteration/phases/compaction.js.map +1 -1
  146. package/dist/runtime/query/iteration/phases/context-reducer-dispatch.test.d.ts +2 -0
  147. package/dist/runtime/query/iteration/phases/context-reducer-dispatch.test.d.ts.map +1 -0
  148. package/dist/runtime/query/iteration/phases/context-reducer-dispatch.test.js +180 -0
  149. package/dist/runtime/query/iteration/phases/context-reducer-dispatch.test.js.map +1 -0
  150. package/dist/runtime/query/iteration/phases/context.d.ts +9 -0
  151. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  152. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  153. package/dist/runtime/query/resume-run.d.ts +70 -0
  154. package/dist/runtime/query/resume-run.d.ts.map +1 -0
  155. package/dist/runtime/query/resume-run.js +46 -0
  156. package/dist/runtime/query/resume-run.js.map +1 -0
  157. package/dist/telemetry/__tests__/model-call-span.test.d.ts +2 -0
  158. package/dist/telemetry/__tests__/model-call-span.test.d.ts.map +1 -0
  159. package/dist/telemetry/__tests__/model-call-span.test.js +147 -0
  160. package/dist/telemetry/__tests__/model-call-span.test.js.map +1 -0
  161. package/dist/telemetry/__tests__/span-closure.test.d.ts +2 -0
  162. package/dist/telemetry/__tests__/span-closure.test.d.ts.map +1 -0
  163. package/dist/telemetry/__tests__/span-closure.test.js +124 -0
  164. package/dist/telemetry/__tests__/span-closure.test.js.map +1 -0
  165. package/dist/tools/advisory/index.js +1 -1
  166. package/dist/tools/advisory/index.js.map +1 -1
  167. package/dist/tools/coordinator/__tests__/plan-dependencies.test.d.ts +2 -0
  168. package/dist/tools/coordinator/__tests__/plan-dependencies.test.d.ts.map +1 -0
  169. package/dist/tools/coordinator/__tests__/plan-dependencies.test.js +126 -0
  170. package/dist/tools/coordinator/__tests__/plan-dependencies.test.js.map +1 -0
  171. package/dist/tools/coordinator/index.d.ts.map +1 -1
  172. package/dist/tools/coordinator/index.js +13 -1
  173. package/dist/tools/coordinator/index.js.map +1 -1
  174. package/dist/tools/coordinator/plan-dependencies.d.ts +43 -0
  175. package/dist/tools/coordinator/plan-dependencies.d.ts.map +1 -0
  176. package/dist/tools/coordinator/plan-dependencies.js +148 -0
  177. package/dist/tools/coordinator/plan-dependencies.js.map +1 -0
  178. package/dist/types/agent/supervisor.d.ts +15 -0
  179. package/dist/types/agent/supervisor.d.ts.map +1 -1
  180. package/dist/types/connector/core.d.ts +34 -0
  181. package/dist/types/connector/core.d.ts.map +1 -1
  182. package/dist/types/connector/definition.d.ts +10 -0
  183. package/dist/types/connector/definition.d.ts.map +1 -1
  184. package/dist/types/connector/mcp.d.ts +13 -0
  185. package/dist/types/connector/mcp.d.ts.map +1 -1
  186. package/dist/types/rag/retrieval.d.ts +16 -0
  187. package/dist/types/rag/retrieval.d.ts.map +1 -1
  188. package/dist/types/rag/storage.d.ts +9 -0
  189. package/dist/types/rag/storage.d.ts.map +1 -1
  190. package/dist/types/rag/vector.d.ts +11 -0
  191. package/dist/types/rag/vector.d.ts.map +1 -1
  192. package/dist/types/router/task-router.d.ts +19 -0
  193. package/dist/types/router/task-router.d.ts.map +1 -1
  194. package/dist/types/run/prepare-step.d.ts +56 -3
  195. package/dist/types/run/prepare-step.d.ts.map +1 -1
  196. package/dist/types/toolset/index.d.ts +22 -0
  197. package/dist/types/toolset/index.d.ts.map +1 -1
  198. package/package.json +1 -1
  199. package/src/advisory/__tests__/consultation-context.test.ts +191 -0
  200. package/src/advisory/context.ts +32 -0
  201. package/src/advisory/executor.ts +30 -3
  202. package/src/compaction/__tests__/context-reducer.test.ts +239 -0
  203. package/src/compaction/factory.ts +7 -0
  204. package/src/compaction/index.ts +8 -0
  205. package/src/compaction/interface.ts +13 -0
  206. package/src/compaction/managers/null.ts +3 -0
  207. package/src/compaction/managers/slidingWindow.ts +6 -0
  208. package/src/compaction/managers/structured.ts +10 -0
  209. package/src/compaction/reducer.ts +154 -0
  210. package/src/connector/builtins/__tests__/oauth2-auth.test.ts +73 -0
  211. package/src/connector/builtins/http.test.ts +28 -2
  212. package/src/connector/builtins/http.ts +26 -2
  213. package/src/connector/index.ts +6 -2
  214. package/src/connector/mcp/__tests__/prompts-and-lifecycle.test.ts +286 -0
  215. package/src/connector/mcp/client.ts +95 -0
  216. package/src/connector/mcp/discovery.ts +19 -4
  217. package/src/connector/mcp/index.ts +6 -2
  218. package/src/connector/mcp/server.ts +101 -3
  219. package/src/manager/agent/__tests__/depth-limit-authority.test.ts +74 -0
  220. package/src/plugin/__tests__/discovery-scopes.test.ts +133 -0
  221. package/src/plugin/__tests__/enable-contributions.test.ts +5 -1
  222. package/src/plugin/__tests__/mcp-admission.test.ts +242 -0
  223. package/src/plugin/lifecycle.ts +57 -1
  224. package/src/plugin/loader.ts +57 -3
  225. package/src/public-runtime.ts +8 -0
  226. package/src/public-types.ts +5 -0
  227. package/src/rag/__tests__/namespace-isolation.test.ts +109 -0
  228. package/src/rag/ingestion.ts +1 -0
  229. package/src/rag/retriever.ts +2 -0
  230. package/src/rag/vector-store.ts +5 -0
  231. package/src/registry/tool/execute.ts +123 -119
  232. package/src/runtime/query/__tests__/per-step-skills.test.ts +154 -0
  233. package/src/runtime/query/__tests__/per-step-tool-choice.test.ts +180 -0
  234. package/src/runtime/query/__tests__/resume-run.test.ts +262 -0
  235. package/src/runtime/query/index.ts +39 -0
  236. package/src/runtime/query/iteration/index.ts +103 -34
  237. package/src/runtime/query/iteration/phases/compaction-model-routing.test.ts +125 -0
  238. package/src/runtime/query/iteration/phases/compaction.ts +106 -5
  239. package/src/runtime/query/iteration/phases/context-reducer-dispatch.test.ts +238 -0
  240. package/src/runtime/query/iteration/phases/context.ts +11 -0
  241. package/src/runtime/query/resume-run.ts +93 -0
  242. package/src/telemetry/__tests__/model-call-span.test.ts +189 -0
  243. package/src/telemetry/__tests__/span-closure.test.ts +153 -0
  244. package/src/tools/advisory/index.ts +1 -1
  245. package/src/tools/coordinator/__tests__/plan-dependencies.test.ts +186 -0
  246. package/src/tools/coordinator/index.ts +14 -1
  247. package/src/tools/coordinator/plan-dependencies.ts +175 -0
  248. package/src/types/agent/supervisor.ts +15 -0
  249. package/src/types/connector/core.ts +34 -0
  250. package/src/types/connector/definition.ts +10 -0
  251. package/src/types/connector/mcp.ts +14 -0
  252. package/src/types/rag/retrieval.ts +16 -0
  253. package/src/types/rag/storage.ts +9 -0
  254. package/src/types/rag/vector.ts +11 -0
  255. package/src/types/router/task-router.ts +19 -0
  256. package/src/types/run/prepare-step.ts +58 -3
  257. package/src/types/toolset/index.ts +22 -0
@@ -0,0 +1,262 @@
1
+ import { mkdtemp, rm } from 'node:fs/promises'
2
+ import { tmpdir } from 'node:os'
3
+ import { join } from 'node:path'
4
+ import { afterEach, describe, expect, it } from 'vitest'
5
+
6
+ import { MockLLMProvider } from '../../../provider/mock.js'
7
+ import { ToolRegistry } from '../../../registry/tool/execute.js'
8
+ import type { CheckpointId, IterationCheckpoint } from '../../../types/hitl/index.js'
9
+ import type { RunId, SessionId, TenantId } from '../../../types/ids/index.js'
10
+ import { createUserMessage } from '../../../types/message/index.js'
11
+ import type { CheckpointRunScope, CheckpointStore } from '../../../types/run/checkpoint-store.js'
12
+ import type { ProjectId, ThreadId } from '../../../types/session/ids.js'
13
+ import { resumeRun } from '../resume-run.js'
14
+ import type { RunStateScope } from '../run-state.js'
15
+
16
+ /**
17
+ * The pieces of a cross-process resume all existed and nothing joined them.
18
+ * `CheckpointManager` wrote the history, budgets, working state and any
19
+ * park; `loadRunState` read them back; `query` accepted `runId` +
20
+ * `resumeFromCheckpoint` and restored all of it. But `resumeFromCheckpoint`
21
+ * had no caller anywhere outside `packages/sdk/src`, so the whole path
22
+ * shipped untravelled — every host was expected to write the same wiring
23
+ * and none did.
24
+ *
25
+ * These cover the join, and especially its two refusals: a resume must not
26
+ * quietly become a fresh run under a recycled id, and it must not step past
27
+ * a park without the answer that park is waiting for.
28
+ */
29
+
30
+ const SCOPE: RunStateScope = {
31
+ tenantId: 'tnt_resume' as TenantId,
32
+ projectId: 'prj_resume' as ProjectId,
33
+ sessionId: 'ses_resume' as SessionId,
34
+ runId: 'run_resume' as RunId,
35
+ threadId: 'thd_resume' as ThreadId,
36
+ }
37
+
38
+ const ZERO_USAGE = {
39
+ promptTokens: 0,
40
+ completionTokens: 0,
41
+ totalTokens: 0,
42
+ cachedTokens: 0,
43
+ cacheWriteTokens: 0,
44
+ }
45
+
46
+ const ZERO_COST = {
47
+ inputCostPer1M: 0,
48
+ outputCostPer1M: 0,
49
+ totalCost: 0,
50
+ cacheDiscount: 0,
51
+ }
52
+
53
+ class InMemoryCheckpointStore implements CheckpointStore {
54
+ readonly rows = new Map<string, IterationCheckpoint>()
55
+
56
+ private key(scope: CheckpointRunScope, id: CheckpointId): string {
57
+ return [scope.tenantId, scope.projectId, scope.sessionId, scope.runId, id].join('/')
58
+ }
59
+
60
+ async writeCheckpoint(scope: CheckpointRunScope, checkpoint: IterationCheckpoint): Promise<void> {
61
+ this.rows.set(this.key(scope, checkpoint.id), checkpoint)
62
+ }
63
+
64
+ async readCheckpoint(
65
+ scope: CheckpointRunScope,
66
+ id: CheckpointId,
67
+ ): Promise<IterationCheckpoint | null> {
68
+ return this.rows.get(this.key(scope, id)) ?? null
69
+ }
70
+
71
+ async listCheckpoints(scope: CheckpointRunScope): Promise<IterationCheckpoint[]> {
72
+ const prefix = `${[scope.tenantId, scope.projectId, scope.sessionId, scope.runId].join('/')}/`
73
+ return [...this.rows.entries()]
74
+ .filter(([key]) => key.startsWith(prefix))
75
+ .map(([, cp]) => cp)
76
+ .sort((a, b) => a.createdAt - b.createdAt)
77
+ }
78
+
79
+ async deleteCheckpoint(scope: CheckpointRunScope, id: CheckpointId): Promise<void> {
80
+ this.rows.delete(this.key(scope, id))
81
+ }
82
+ }
83
+
84
+ function checkpoint(overrides: Partial<IterationCheckpoint> = {}): IterationCheckpoint {
85
+ return {
86
+ id: 'ckpt_1' as CheckpointId,
87
+ runId: SCOPE.runId,
88
+ iteration: 2,
89
+ messages: [createUserMessage('the work so far')],
90
+ tokenUsage: { ...ZERO_USAGE, promptTokens: 120, totalTokens: 120 },
91
+ costInfo: { ...ZERO_COST, totalCost: 0.4 },
92
+ guardState: { iterationCount: 2, elapsedMs: 9_000 },
93
+ createdAt: Date.now(),
94
+ ...overrides,
95
+ } as IterationCheckpoint
96
+ }
97
+
98
+ let workdirs: string[] = []
99
+
100
+ afterEach(async () => {
101
+ await Promise.all(workdirs.map((d) => rm(d, { recursive: true, force: true })))
102
+ workdirs = []
103
+ })
104
+
105
+ async function mkWorkdir(): Promise<string> {
106
+ const dir = await mkdtemp(join(tmpdir(), 'namzu-resume-run-'))
107
+ workdirs.push(dir)
108
+ return dir
109
+ }
110
+
111
+ async function baseParams(store: CheckpointStore) {
112
+ return {
113
+ scope: SCOPE,
114
+ checkpointStore: store,
115
+ provider: new MockLLMProvider({ turns: [{ text: 'continued' }] }),
116
+ tools: new ToolRegistry(),
117
+ runConfig: {
118
+ model: 'mock-model',
119
+ timeoutMs: 30_000,
120
+ tokenBudget: 100_000,
121
+ maxIterations: 2,
122
+ maxResponseTokens: 256,
123
+ },
124
+ agentId: 'agent_resume',
125
+ agentName: 'Resume Agent',
126
+ workingDirectory: await mkWorkdir(),
127
+ sessionId: SCOPE.sessionId,
128
+ threadId: SCOPE.threadId,
129
+ projectId: SCOPE.projectId,
130
+ tenantId: SCOPE.tenantId,
131
+ // Required by the contract, and rightly so: a resume that lands on a
132
+ // park has to have somewhere to ask.
133
+ resumeHandler: async () => ({ action: 'continue' as const }),
134
+ }
135
+ }
136
+
137
+ describe('a run is picked back up from its store', () => {
138
+ it('continues the same run id rather than starting a new one', async () => {
139
+ const store = new InMemoryCheckpointStore()
140
+ await store.writeCheckpoint(SCOPE, checkpoint())
141
+
142
+ const outcome = await resumeRun(await baseParams(store))
143
+
144
+ expect(outcome.resumed).toBe(true)
145
+ if (!outcome.resumed) return
146
+ // The whole point: a resume is the same run in a different process,
147
+ // so its id, budgets and trace all have to carry across.
148
+ expect(outcome.run.id).toBe(SCOPE.runId)
149
+ expect(outcome.state.checkpointId).toBe('ckpt_1')
150
+ })
151
+
152
+ it('carries the spent budget forward instead of granting a fresh one', async () => {
153
+ const store = new InMemoryCheckpointStore()
154
+ await store.writeCheckpoint(SCOPE, checkpoint())
155
+
156
+ const outcome = await resumeRun(await baseParams(store))
157
+
158
+ expect(outcome.resumed).toBe(true)
159
+ if (!outcome.resumed) return
160
+ // A run recalled at 120 tokens must not come back at zero — the
161
+ // budget belongs to the run, not to the process hosting it.
162
+ expect(outcome.run.tokenUsage.totalTokens).toBeGreaterThanOrEqual(120)
163
+ })
164
+
165
+ it('picks the newest checkpoint when the caller names none', async () => {
166
+ const store = new InMemoryCheckpointStore()
167
+ await store.writeCheckpoint(SCOPE, checkpoint({ id: 'ckpt_old' as CheckpointId, createdAt: 1 }))
168
+ await store.writeCheckpoint(
169
+ SCOPE,
170
+ checkpoint({ id: 'ckpt_new' as CheckpointId, createdAt: 2_000 }),
171
+ )
172
+
173
+ const outcome = await resumeRun(await baseParams(store))
174
+
175
+ expect(outcome.resumed).toBe(true)
176
+ if (!outcome.resumed) return
177
+ expect(outcome.state.checkpointId).toBe('ckpt_new')
178
+ })
179
+
180
+ it('honours an explicitly named checkpoint', async () => {
181
+ const store = new InMemoryCheckpointStore()
182
+ await store.writeCheckpoint(SCOPE, checkpoint({ id: 'ckpt_old' as CheckpointId, createdAt: 1 }))
183
+ await store.writeCheckpoint(
184
+ SCOPE,
185
+ checkpoint({ id: 'ckpt_new' as CheckpointId, createdAt: 2_000 }),
186
+ )
187
+
188
+ const outcome = await resumeRun({
189
+ ...(await baseParams(store)),
190
+ checkpointId: 'ckpt_old' as CheckpointId,
191
+ })
192
+
193
+ expect(outcome.resumed).toBe(true)
194
+ if (!outcome.resumed) return
195
+ expect(outcome.state.checkpointId).toBe('ckpt_old')
196
+ })
197
+ })
198
+
199
+ describe('it refuses rather than guessing', () => {
200
+ it('reports no checkpoint instead of silently starting fresh', async () => {
201
+ const outcome = await resumeRun(await baseParams(new InMemoryCheckpointStore()))
202
+
203
+ // Starting a new run here would be the worst outcome: a different
204
+ // run wearing a recycled id, with the original's budget reset.
205
+ expect(outcome).toEqual({ resumed: false, reason: 'no-checkpoint' })
206
+ })
207
+
208
+ it('hands back the outstanding question instead of resuming past it', async () => {
209
+ const store = new InMemoryCheckpointStore()
210
+ await store.writeCheckpoint(
211
+ SCOPE,
212
+ checkpoint({
213
+ pending: {
214
+ request: {
215
+ type: 'tool_review',
216
+ runId: SCOPE.runId,
217
+ checkpointId: 'ckpt_1' as CheckpointId,
218
+ toolCalls: [{ id: 'call_1', name: 'write', input: {} }],
219
+ },
220
+ parkedAt: Date.now(),
221
+ deadlineAt: Date.now() + 60_000,
222
+ },
223
+ } as unknown as Partial<IterationCheckpoint>),
224
+ )
225
+
226
+ const outcome = await resumeRun(await baseParams(store))
227
+
228
+ expect(outcome.resumed).toBe(false)
229
+ if (outcome.resumed || outcome.reason !== 'awaiting-decision') {
230
+ throw new Error(`expected awaiting-decision, got ${JSON.stringify(outcome)}`)
231
+ }
232
+ // The host needs the request itself to put in front of a person.
233
+ expect(outcome.pending.request.type).toBe('tool_review')
234
+ expect(outcome.state.runId).toBe(SCOPE.runId)
235
+ })
236
+
237
+ it('treats an already-answered park as an ordinary resume', async () => {
238
+ const store = new InMemoryCheckpointStore()
239
+ await store.writeCheckpoint(
240
+ SCOPE,
241
+ checkpoint({
242
+ pending: {
243
+ request: {
244
+ type: 'tool_review',
245
+ runId: SCOPE.runId,
246
+ checkpointId: 'ckpt_1' as CheckpointId,
247
+ toolCalls: [{ id: 'call_1', name: 'write', input: {} }],
248
+ },
249
+ parkedAt: Date.now(),
250
+ deadlineAt: Date.now() + 60_000,
251
+ resolvedAt: Date.now(),
252
+ },
253
+ } as unknown as Partial<IterationCheckpoint>),
254
+ )
255
+
256
+ const outcome = await resumeRun(await baseParams(store))
257
+
258
+ // `resolvedAt` is what makes a park answered. Blocking on one that
259
+ // already has its answer would strand the run permanently.
260
+ expect(outcome.resumed).toBe(true)
261
+ })
262
+ })
@@ -9,6 +9,8 @@ import {
9
9
  import { findDanglingMessages, removeDanglingMessages } from '../../compaction/dangling.js'
10
10
  import { extractFromUserMessage } from '../../compaction/extractor.js'
11
11
  import { WorkingStateManager } from '../../compaction/manager.js'
12
+ import type { ContextReducer } from '../../compaction/reducer.js'
13
+ import { serializeState as serializeWorkingState } from '../../compaction/serializer.js'
12
14
  import { restoreWorkingState, snapshotWorkingState } from '../../compaction/wire.js'
13
15
  import type { CompactionConfig } from '../../config/runtime.js'
14
16
  import { TOOL_OUTPUT_DIR_NAME } from '../../constants/tools/index.js'
@@ -421,6 +423,16 @@ export interface QueryParams {
421
423
  */
422
424
  workingMemoryProvider?: WorkingMemoryProvider
423
425
 
426
+ /**
427
+ * Replace context reduction for this run.
428
+ *
429
+ * Outranks `compactionConfig.strategy`, and the built-in structured pass
430
+ * does not also run: two mechanisms editing one history in the same pass
431
+ * cannot both be reasoned about. See `ContextReducer` for the invariants a
432
+ * reducer is expected to keep.
433
+ */
434
+ contextReducer?: ContextReducer
435
+
424
436
  agentBus?: import('../../bus/index.js').AgentBus
425
437
 
426
438
  verificationGate?: VerificationGateConfig
@@ -752,6 +764,31 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
752
764
  params.advisory.budget,
753
765
  )
754
766
 
767
+ // What the run looks like when the MODEL consults an advisor, as
768
+ // opposed to when a trigger does. The trigger path has always passed
769
+ // this; the tool path passed an empty context, so an advisor the model
770
+ // asked for help saw the question and nothing else.
771
+ //
772
+ // `includeToolCatalog` and `useCompactedContext` are read here and
773
+ // nowhere else. Both were declared on `AdvisoryConfig` /
774
+ // `AdvisorDefinition` and consulted by nothing, so a host who turned
775
+ // the catalogue off still paid for it in every advisory prompt.
776
+ const advisoryConfig = params.advisory
777
+ advisoryCtx.setCallContextProvider(() => {
778
+ const summary =
779
+ workingStateManager && advisoryConfig.advisors.some((a) => a.useCompactedContext)
780
+ ? serializeWorkingState(workingStateManager.getState())
781
+ : undefined
782
+ return {
783
+ messages: ctx.runMgr.messages,
784
+ ...(summary !== undefined ? { workingStateSummary: summary } : {}),
785
+ ...(advisoryConfig.includeToolCatalog
786
+ ? { toolCatalog: params.tools.toLLMTools(effectiveAllowedTools) }
787
+ : {}),
788
+ iteration: ctx.runMgr.currentIteration,
789
+ }
790
+ })
791
+
755
792
  if (params.advisory.enableAgentTool) {
756
793
  const advisoryTools = buildAdvisoryTools({ advisoryCtx })
757
794
  const overrides = params.runtimeToolOverrides
@@ -800,6 +837,8 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
800
837
  toolGrants: new ToolGrantSet(),
801
838
  compactionConfig: params.compactionConfig,
802
839
  workingStateManager,
840
+ taskRouter: params.taskRouter,
841
+ contextReducer: params.contextReducer,
803
842
  workingMemoryProvider: params.workingMemoryProvider,
804
843
  advisoryCtx,
805
844
  agentBus: params.agentBus,
@@ -5,11 +5,13 @@ import {
5
5
  DEFAULT_STRUCTURED_OUTPUT_RETRIES,
6
6
  STRUCTURED_OUTPUT_REPROMPT,
7
7
  } from '../../../constants/tools/index.js'
8
+ import { renderSkillsSection } from '../../../persona/assembler.js'
8
9
  import { collect } from '../../../provider/collect.js'
9
10
  import {
10
11
  GENAI,
11
12
  NAMZU,
12
13
  agentIterationSpanName,
14
+ chatSpanName,
13
15
  parentContext,
14
16
  } from '../../../telemetry/attributes.js'
15
17
  import { getTracer } from '../../../telemetry/runtime-accessors.js'
@@ -21,6 +23,7 @@ import {
21
23
  createSystemMessage,
22
24
  createUserMessage,
23
25
  } from '../../../types/message/index.js'
26
+ import type { ToolChoice } from '../../../types/provider/chat.js'
24
27
  import { classifyProviderError } from '../../../types/provider/errors.js'
25
28
  import type { ChatCompletionResponse } from '../../../types/provider/index.js'
26
29
  import type { AnswerReview } from '../../../types/run/answer-review.js'
@@ -30,6 +33,7 @@ import type {
30
33
  StepResult,
31
34
  StopReason,
32
35
  } from '../../../types/run/index.js'
36
+ import type { Skill } from '../../../types/skills/index.js'
33
37
  import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/index.js'
34
38
  import { toErrorMessage } from '../../../utils/error.js'
35
39
  import { generateMessageId } from '../../../utils/id.js'
@@ -150,23 +154,30 @@ export class IterationOrchestrator {
150
154
  {},
151
155
  parentContext(this.ctx.rootSpan),
152
156
  )
153
- // Tool spans for this turn belong under this iteration.
154
- this.ctx.toolExecutor.setParentSpan(iterSpan)
157
+ // Declared out here so the iteration's own finally can close it on
158
+ // any path that does not reach its success branch.
159
+ let chatSpan: Span | undefined
160
+ try {
161
+ // Tool spans for this turn belong under this iteration. Inside
162
+ // the try rather than before it: a throw from any of these left
163
+ // the span open, and an iteration span that never ends is a
164
+ // trace that never closes — the export is incomplete for exactly
165
+ // the run that failed.
166
+ this.ctx.toolExecutor.setParentSpan(iterSpan)
155
167
 
156
- iterSpan.setAttributes({
157
- [NAMZU.ITERATION]: iterationNum,
158
- [NAMZU.RUN_ID]: runMgr.id,
159
- [GENAI.REQUEST_MODEL]: model,
160
- })
168
+ iterSpan.setAttributes({
169
+ [NAMZU.ITERATION]: iterationNum,
170
+ [NAMZU.RUN_ID]: runMgr.id,
171
+ [GENAI.REQUEST_MODEL]: model,
172
+ })
161
173
 
162
- await this.ctx.emitEvent({
163
- type: 'iteration_started',
164
- runId: runMgr.id,
165
- iteration: iterationNum,
166
- })
167
- yield* this.ctx.drainPending()
174
+ await this.ctx.emitEvent({
175
+ type: 'iteration_started',
176
+ runId: runMgr.id,
177
+ iteration: iterationNum,
178
+ })
179
+ yield* this.ctx.drainPending()
168
180
 
169
- try {
170
181
  if (this.ctx.pluginManager) {
171
182
  const hookResults = await this.ctx.pluginManager.executeHooks(
172
183
  'iteration_start',
@@ -225,8 +236,17 @@ export class IterationOrchestrator {
225
236
  // exactly this reason. Shallow is enough: the defect is array
226
237
  // mutation, and per-iteration this is trivial next to the model
227
238
  // call it precedes.
228
- const messages = step.system
229
- ? [...baseMessages, createSystemMessage(step.system)]
239
+ // A step's skills and its guidance ride the same ephemeral
240
+ // trailing system message. Appending leaves the cached prefix
241
+ // intact; rewriting the run's own prompt to carry a phase's
242
+ // skills would invalidate it on every iteration.
243
+ // `renderSkillsSection` already answers null for an empty list, so
244
+ // there is no length check here — a second guard for the same
245
+ // case is one more thing to keep in agreement with the first.
246
+ const stepSkills = step.skills ? renderSkillsSection([...step.skills]) : null
247
+ const stepPreamble = [step.system, stepSkills].filter(Boolean).join('\n\n')
248
+ const messages = stepPreamble
249
+ ? [...baseMessages, createSystemMessage(stepPreamble)]
230
250
  : [...baseMessages]
231
251
 
232
252
  if (this.ctx.pluginManager) {
@@ -262,6 +282,27 @@ export class IterationOrchestrator {
262
282
  // aggregated `ChatCompletionResponse` for the legacy
263
283
  // downstream paths (assistantMsg construction, working
264
284
  // state extraction, telemetry attribute stamping).
285
+ // The model call gets its own span. There was none at all —
286
+ // `chatSpanName` shipped with zero call sites — so a run's traces
287
+ // carried no LLM latency whatsoever, and the one thing anybody
288
+ // opens a trace to find (which turn was slow, and why) was the
289
+ // one thing not in it.
290
+ chatSpan = tracer.startSpan(chatSpanName(stepModel), {}, parentContext(iterSpan))
291
+ chatSpan.setAttributes({
292
+ [GENAI.OPERATION_NAME]: 'chat',
293
+ [GENAI.SYSTEM]: this.ctx.provider.id,
294
+ [GENAI.REQUEST_MODEL]: stepModel,
295
+ ...((step.temperature ?? runConfig.temperature) !== undefined
296
+ ? { [GENAI.REQUEST_TEMPERATURE]: (step.temperature ?? runConfig.temperature) as number }
297
+ : {}),
298
+ ...((step.maxResponseTokens ?? runConfig.maxResponseTokens) !== undefined
299
+ ? {
300
+ [GENAI.REQUEST_MAX_TOKENS]: (step.maxResponseTokens ??
301
+ runConfig.maxResponseTokens) as number,
302
+ }
303
+ : {}),
304
+ })
305
+
265
306
  const { response, messageId } = yield* streamProviderTurn(
266
307
  this.ctx.provider,
267
308
  {
@@ -269,7 +310,17 @@ export class IterationOrchestrator {
269
310
  messages,
270
311
  tools: llmTools.length > 0 ? llmTools : undefined,
271
312
  ...(enforceToolInputSchema ? { enforceToolInputSchema } : {}),
272
- toolChoice: forceFinalize && llmTools.length > 0 ? 'none' : undefined,
313
+ // The forced-final turn wins: a step that asked to force a
314
+ // tool cannot override the loop's own decision to stop
315
+ // asking for them. Otherwise the step's choice applies —
316
+ // and only to this step, because the next one is prepared
317
+ // from scratch.
318
+ toolChoice:
319
+ forceFinalize && llmTools.length > 0
320
+ ? 'none'
321
+ : llmTools.length > 0
322
+ ? step.toolChoice
323
+ : undefined,
273
324
  temperature: step.temperature ?? runConfig.temperature,
274
325
  maxTokens: step.maxResponseTokens ?? runConfig.maxResponseTokens,
275
326
  cacheControl: { type: 'auto' },
@@ -288,6 +339,27 @@ export class IterationOrchestrator {
288
339
  iterSpan,
289
340
  )
290
341
 
342
+ // Stamped on the call that produced them. The token counts also
343
+ // stay on the iteration span below, where they have always been:
344
+ // moving them would silently break whatever reads them today,
345
+ // and one turn per iteration makes the two agree.
346
+ chatSpan.setAttributes({
347
+ [GENAI.RESPONSE_MODEL]: response.model || stepModel,
348
+ [GENAI.RESPONSE_ID]: response.id,
349
+ [GENAI.USAGE_INPUT_TOKENS]: response.usage.promptTokens,
350
+ [GENAI.USAGE_OUTPUT_TOKENS]: response.usage.completionTokens,
351
+ // An array, per the semantic convention: one call can finish
352
+ // several ways when a provider returns more than one choice.
353
+ [GENAI.RESPONSE_FINISH_REASONS]: [response.finishReason ?? 'stop'],
354
+ [NAMZU.CACHE_READ_TOKENS]: response.usage.cachedTokens ?? 0,
355
+ [NAMZU.CACHE_WRITE_TOKENS]: response.usage.cacheWriteTokens ?? 0,
356
+ })
357
+ chatSpan.setStatus({ code: SpanStatusCode.OK })
358
+ chatSpan.end()
359
+ // Closed here for an accurate duration, and cleared so the
360
+ // iteration finally does not close it a second time.
361
+ chatSpan = undefined
362
+
291
363
  // Main-loop turn: also records the prompt size compaction reads.
292
364
  runMgr.recordTurnUsage(response.usage)
293
365
 
@@ -446,7 +518,6 @@ export class IterationOrchestrator {
446
518
  hasToolCalls: false,
447
519
  })
448
520
  yield* this.ctx.drainPending()
449
- iterSpan.end()
450
521
  continue
451
522
  }
452
523
 
@@ -464,7 +535,6 @@ export class IterationOrchestrator {
464
535
  attempts: attempt - 1,
465
536
  })
466
537
  runMgr.setStopReason('structured_output_failed')
467
- iterSpan.end()
468
538
  break
469
539
  }
470
540
  this.ctx.log.info('Re-prompting for structured output', {
@@ -480,7 +550,6 @@ export class IterationOrchestrator {
480
550
  hasToolCalls: false,
481
551
  })
482
552
  yield* this.ctx.drainPending()
483
- iterSpan.end()
484
553
  continue
485
554
  }
486
555
 
@@ -509,7 +578,6 @@ export class IterationOrchestrator {
509
578
  limit,
510
579
  })
511
580
  runMgr.setStopReason('answer_rejected')
512
- iterSpan.end()
513
581
  break
514
582
  }
515
583
  this.ctx.log.info('Answer rejected — returning it to the model', {
@@ -525,7 +593,6 @@ export class IterationOrchestrator {
525
593
  hasToolCalls: false,
526
594
  })
527
595
  yield* this.ctx.drainPending()
528
- iterSpan.end()
529
596
  continue
530
597
  }
531
598
  }
@@ -553,11 +620,9 @@ export class IterationOrchestrator {
553
620
  if (this.ctx.abortController.signal.aborted) {
554
621
  runMgr.setStopReason('cancelled')
555
622
  runMgr.markCancelled()
556
- iterSpan.end()
557
623
  break
558
624
  }
559
625
  runMgr.setStopReason('end_turn')
560
- iterSpan.end()
561
626
  break
562
627
  }
563
628
 
@@ -579,12 +644,10 @@ export class IterationOrchestrator {
579
644
  })
580
645
 
581
646
  if (reviewOutcome.decision === 'stop') {
582
- iterSpan.end()
583
647
  return
584
648
  }
585
649
 
586
650
  if (reviewOutcome.decision === 'rejected') {
587
- iterSpan.end()
588
651
  continue
589
652
  }
590
653
 
@@ -604,7 +667,6 @@ export class IterationOrchestrator {
604
667
  hasToolCalls: true,
605
668
  })
606
669
  yield* this.ctx.drainPending()
607
- iterSpan.end()
608
670
  break
609
671
  }
610
672
 
@@ -631,7 +693,6 @@ export class IterationOrchestrator {
631
693
  hasToolCalls: true,
632
694
  })
633
695
  yield* this.ctx.drainPending()
634
- iterSpan.end()
635
696
  break
636
697
  }
637
698
 
@@ -651,13 +712,11 @@ export class IterationOrchestrator {
651
712
  hasToolCalls: true,
652
713
  })
653
714
  yield* this.ctx.drainPending()
654
- iterSpan.end()
655
715
  break
656
716
  }
657
717
 
658
718
  const checkpointSignal = yield* runIterationCheckpoint(this.ctx, iterationNum)
659
719
  if (checkpointSignal === 'stop') {
660
- iterSpan.end()
661
720
  return
662
721
  }
663
722
 
@@ -680,7 +739,6 @@ export class IterationOrchestrator {
680
739
  hasToolCalls: true,
681
740
  })
682
741
  yield* this.ctx.drainPending()
683
- iterSpan.end()
684
742
  } catch (err) {
685
743
  // A Stop that aborted the in-flight turn surfaces here as a
686
744
  // thrown abort (the provider stream was raced against the run
@@ -692,7 +750,6 @@ export class IterationOrchestrator {
692
750
  if (this.ctx.abortController.signal.aborted) {
693
751
  runMgr.setStopReason('cancelled')
694
752
  runMgr.markCancelled()
695
- iterSpan.end()
696
753
  break
697
754
  }
698
755
 
@@ -722,7 +779,6 @@ export class IterationOrchestrator {
722
779
  if (iterationActivity) {
723
780
  this.ctx.activityStore.complete(iterationActivity.id)
724
781
  }
725
- iterSpan.end()
726
782
  continue
727
783
  }
728
784
  }
@@ -736,8 +792,15 @@ export class IterationOrchestrator {
736
792
  message: toErrorMessage(err),
737
793
  })
738
794
  iterSpan.recordException(err instanceof Error ? err : new Error(String(err)))
739
- iterSpan.end()
740
795
  throw err
796
+ } finally {
797
+ // A model call that threw never reached its own close above.
798
+ chatSpan?.end()
799
+ // The only place the iteration span ends. It used to be ended at each of
800
+ // seventeen exits, which is a rule every future edit has to
801
+ // remember; a generator abandoned by its consumer never reached
802
+ // any of them.
803
+ iterSpan.end()
741
804
  }
742
805
  }
743
806
  }
@@ -752,8 +815,10 @@ export class IterationOrchestrator {
752
815
  */
753
816
  private async prepareStep(stepNumber: number): Promise<{
754
817
  allowedTools?: string[]
818
+ toolChoice?: ToolChoice
755
819
  model?: string
756
820
  system?: string
821
+ skills?: readonly Skill[]
757
822
  temperature?: number
758
823
  maxResponseTokens?: number
759
824
  }> {
@@ -789,8 +854,10 @@ export class IterationOrchestrator {
789
854
 
790
855
  const prepared: {
791
856
  allowedTools?: string[]
857
+ toolChoice?: ToolChoice
792
858
  model?: string
793
859
  system?: string
860
+ skills?: readonly Skill[]
794
861
  temperature?: number
795
862
  maxResponseTokens?: number
796
863
  } = {}
@@ -809,8 +876,10 @@ export class IterationOrchestrator {
809
876
  }
810
877
  prepared.allowedTools = known
811
878
  }
879
+ if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
812
880
  if (result.model !== undefined) prepared.model = result.model
813
881
  if (result.system !== undefined) prepared.system = result.system
882
+ if (result.skills !== undefined) prepared.skills = result.skills
814
883
  if (result.temperature !== undefined) prepared.temperature = result.temperature
815
884
  if (result.maxResponseTokens !== undefined) {
816
885
  prepared.maxResponseTokens = result.maxResponseTokens