@namzu/sdk 42.0.0 → 42.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +174 -0
  2. package/dist/manager/resident/outbox.d.ts +8 -8
  3. package/dist/manager/resident/store.d.ts +4 -4
  4. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  5. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  6. package/dist/runtime/query/cancelled-before-start.js +152 -0
  7. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  8. package/dist/runtime/query/checkpoint.d.ts +21 -0
  9. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  10. package/dist/runtime/query/checkpoint.js +23 -0
  11. package/dist/runtime/query/checkpoint.js.map +1 -1
  12. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  13. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  14. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  15. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  16. package/dist/runtime/query/executor.d.ts +70 -35
  17. package/dist/runtime/query/executor.d.ts.map +1 -1
  18. package/dist/runtime/query/executor.js +46 -380
  19. package/dist/runtime/query/executor.js.map +1 -1
  20. package/dist/runtime/query/finalize-run.d.ts +55 -0
  21. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  22. package/dist/runtime/query/finalize-run.js +113 -0
  23. package/dist/runtime/query/finalize-run.js.map +1 -0
  24. package/dist/runtime/query/index.d.ts +4 -9
  25. package/dist/runtime/query/index.d.ts.map +1 -1
  26. package/dist/runtime/query/index.js +238 -893
  27. package/dist/runtime/query/index.js.map +1 -1
  28. package/dist/runtime/query/iteration/index.d.ts +6 -161
  29. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  30. package/dist/runtime/query/iteration/index.js +23 -523
  31. package/dist/runtime/query/iteration/index.js.map +1 -1
  32. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  33. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  34. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  35. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  36. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  37. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  38. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  39. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  40. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  41. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  42. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  43. package/dist/runtime/query/prepare-run.d.ts +94 -0
  44. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  45. package/dist/runtime/query/prepare-run.js +589 -0
  46. package/dist/runtime/query/prepare-run.js.map +1 -0
  47. package/dist/runtime/query/release-run.d.ts +56 -0
  48. package/dist/runtime/query/release-run.d.ts.map +1 -0
  49. package/dist/runtime/query/release-run.js +101 -0
  50. package/dist/runtime/query/release-run.js.map +1 -0
  51. package/dist/runtime/query/resume-pending.d.ts +112 -1
  52. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  53. package/dist/runtime/query/resume-pending.js +133 -0
  54. package/dist/runtime/query/resume-pending.js.map +1 -1
  55. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  56. package/dist/types/run/config.d.ts +12 -5
  57. package/dist/types/run/config.d.ts.map +1 -1
  58. package/package.json +1 -1
  59. package/src/runtime/query/cancelled-before-start.ts +189 -0
  60. package/src/runtime/query/checkpoint.ts +22 -0
  61. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  62. package/src/runtime/query/executor.ts +63 -442
  63. package/src/runtime/query/finalize-run.ts +192 -0
  64. package/src/runtime/query/index.ts +270 -1011
  65. package/src/runtime/query/iteration/index.ts +40 -586
  66. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  67. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  68. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  69. package/src/runtime/query/prepare-run.ts +718 -0
  70. package/src/runtime/query/release-run.ts +168 -0
  71. package/src/runtime/query/resume-pending.ts +158 -0
  72. package/src/types/run/config.ts +12 -5
@@ -0,0 +1,168 @@
1
+ import type { Span } from '@opentelemetry/api'
2
+ import type { WorkingStateManager } from '../../compaction/manager.js'
3
+ import { NAMZU } from '../../constants/telemetry/index.js'
4
+ import type { EmergencySaveManager } from '../../manager/run/emergency.js'
5
+ import { recordRunDuration } from '../../telemetry/metrics.js'
6
+ import type { RunEvent } from '../../types/run/index.js'
7
+ import { type PromoteMemory, memoryCandidateFor } from '../../types/run/memory-promotion.js'
8
+ import type { Sandbox } from '../../types/sandbox/index.js'
9
+ import { errorAttributes } from '../../utils/log/exception.js'
10
+ import type { AwaitedJobs } from '../jobs/awaited-jobs.js'
11
+ import type { BackgroundJobRegistry } from '../jobs/registry.js'
12
+ import type { RunContext } from './context.js'
13
+ import type { EventTranslator } from './events.js'
14
+ import type { QuestionParkBinding } from './question-park.js'
15
+ import { teardownSandbox } from './sandbox-lifecycle.js'
16
+
17
+ /**
18
+ * Everything a run borrows, handed back when it ends.
19
+ *
20
+ * A run attaches to process-wide things it does not own — crash handlers, a
21
+ * shared background-job registry, a question channel a tool outlived, a task
22
+ * store's listener, a sandbox — and every one of them has to be released on
23
+ * the way out, including the exits a `try` never reaches. So this is the body
24
+ * of `query()`'s `finally`: the keyword stays where it is, which is what makes
25
+ * abandonment run this just as settlement does.
26
+ *
27
+ * The order is the contract, and the awaits are in it: detach the emergency
28
+ * handlers, unsubscribe from job exits, close the wait-intent recorder, kill
29
+ * only this run's jobs, unbind the question channel, promote what the run
30
+ * learned, tear the sandbox down, unsubscribe from the task store, record the
31
+ * duration under the status the run actually settled with, and end the root
32
+ * span last.
33
+ */
34
+ export interface RunResources {
35
+ readonly ctx: RunContext
36
+ readonly eventTranslator: EventTranslator
37
+ readonly emergencyManager: EmergencySaveManager | undefined
38
+ readonly unsubscribeJobExits: (() => void) | undefined
39
+ readonly unsubscribeTaskStore: (() => void) | undefined
40
+ readonly awaitedJobs: AwaitedJobs | undefined
41
+ /** Jobs a host bound to its session are the host's to stop, not this run's. */
42
+ readonly backgroundJobs: BackgroundJobRegistry | undefined
43
+ readonly backgroundJobOwner: string | undefined
44
+ readonly questionParks: QuestionParkBinding
45
+ readonly workingStateManager: WorkingStateManager | undefined
46
+ readonly promoteMemory: PromoteMemory | undefined
47
+ readonly sandbox: Sandbox | undefined
48
+ readonly sandboxTeardownTimeoutMs: number
49
+ readonly runStartedAt: number
50
+ readonly rootSpan: Span
51
+ }
52
+
53
+ /**
54
+ * Release them, in that order.
55
+ *
56
+ * A generator rather than a plain async function because the sandbox teardown
57
+ * reports `sandbox_destroyed`, and that event has to reach the host at the
58
+ * position it always did — `yield*` from the caller's `finally` keeps it
59
+ * exactly there.
60
+ */
61
+ export async function* releaseRunResources(
62
+ resources: RunResources,
63
+ ): AsyncGenerator<RunEvent, void> {
64
+ const {
65
+ ctx,
66
+ eventTranslator,
67
+ emergencyManager,
68
+ unsubscribeJobExits,
69
+ unsubscribeTaskStore,
70
+ awaitedJobs,
71
+ backgroundJobs,
72
+ backgroundJobOwner,
73
+ questionParks,
74
+ workingStateManager,
75
+ promoteMemory,
76
+ sandbox,
77
+ sandboxTeardownTimeoutMs,
78
+ runStartedAt,
79
+ rootSpan,
80
+ } = resources
81
+
82
+ // Release the process's termination path as soon as this run is
83
+ // done with it. Leaving the handlers installed would keep a
84
+ // WeakRef'd, settled run as the crash target for the rest of the
85
+ // process's life.
86
+ emergencyManager?.detach()
87
+
88
+ // A background job outlives the tool call that started it — that
89
+ // is what it is for — so nothing but this stops it outliving the
90
+ // RUN. Scoped to this run's id: a shared registry serving several
91
+ // runs must not have one of them tear down another's work.
92
+ //
93
+ // Awaited, and its failure swallowed. A job that would not die is
94
+ // worth a log line, and is not worth retracting a run's answer.
95
+ unsubscribeJobExits?.()
96
+ // The wait-intent recorder listens on the same shared registry and
97
+ // leaks the same way if it is left attached.
98
+ awaitedJobs?.close()
99
+ // Only jobs bound to this run. Jobs a host bound to its session are
100
+ // the host's to stop, when the session ends.
101
+ if (backgroundJobs && (backgroundJobOwner ?? ctx.runId) === ctx.runId) {
102
+ try {
103
+ const stopped = await backgroundJobs.killOwner(ctx.runId)
104
+ if (stopped.length > 0) {
105
+ ctx.log.info('Background jobs stopped with the run', {
106
+ [NAMZU.RUN_ID]: ctx.runId,
107
+ 'namzu.jobs.stopped': stopped.length,
108
+ })
109
+ }
110
+ } catch (jobErr) {
111
+ ctx.log.error('A background job did not stop cleanly', {
112
+ [NAMZU.RUN_ID]: ctx.runId,
113
+ ...errorAttributes(jobErr),
114
+ })
115
+ }
116
+ }
117
+
118
+ // Same reasoning for the question channel: the tools outlive the
119
+ // run that bound them, so leaving it attached would have a later
120
+ // run's question written into this run's checkpoint store.
121
+ questionParks.unbind()
122
+
123
+ // Offer what the run learned to whoever decides what is worth
124
+ // keeping. In `finally` and awaited: a run that failed still
125
+ // discovered things, and a fire-and-forget write would race the
126
+ // process exiting on a one-shot CLI run. A throw here is
127
+ // swallowed — a memory that failed to form must not retract an
128
+ // answer that was already produced.
129
+ const candidate = memoryCandidateFor(ctx.runId, workingStateManager)
130
+ if (promoteMemory && candidate) {
131
+ try {
132
+ await promoteMemory(candidate)
133
+ } catch (promoteErr) {
134
+ ctx.log.error('Memory promotion threw — the run is unaffected', {
135
+ [NAMZU.RUN_ID]: ctx.runId,
136
+ 'exception.message': promoteErr instanceof Error ? promoteErr.message : String(promoteErr),
137
+ })
138
+ }
139
+ }
140
+
141
+ // --- Sandbox lifecycle: destroy after run ---
142
+ if (sandbox) {
143
+ const sandboxId = sandbox.id
144
+ const teardown = await teardownSandbox(sandbox, sandboxTeardownTimeoutMs)
145
+ if (teardown.kind === 'destroyed') {
146
+ await eventTranslator.emitEvent({
147
+ type: 'sandbox_destroyed',
148
+ runId: ctx.runId,
149
+ sandboxId,
150
+ })
151
+ yield* eventTranslator.drainPending()
152
+ ctx.log.info('Sandbox destroyed', { 'namzu.sandbox.id': sandboxId })
153
+ } else {
154
+ ctx.log.error('Sandbox destroy failed', {
155
+ 'namzu.sandbox.id': sandboxId,
156
+ ...errorAttributes(teardown.error),
157
+ })
158
+ }
159
+ }
160
+
161
+ unsubscribeTaskStore?.()
162
+ // Keyed by HOW it settled, not just that it did: a run that was
163
+ // cancelled and a run that hit its budget have very different
164
+ // duration distributions, and averaging them together describes
165
+ // neither.
166
+ recordRunDuration(ctx.runMgr.getRun().status ?? 'unknown', Date.now() - runStartedAt)
167
+ rootSpan.end()
168
+ }
@@ -4,6 +4,7 @@ import type { RunPersistence } from '../../manager/run/persistence.js'
4
4
  import { ToolExecutionCollector } from '../../store/run/tool-executions.js'
5
5
  import type {
6
6
  CheckpointId,
7
+ HITLDecisionRequest,
7
8
  HITLResumeDecision,
8
9
  IterationCheckpoint,
9
10
  ToolCallSummary,
@@ -31,6 +32,19 @@ import { isPauseForCall } from './tool-pause.js'
31
32
  * keeps the existing repair-and-re-decide behavior.
32
33
  */
33
34
  export interface PendingResumePlan {
35
+ /**
36
+ * What produced this plan, and therefore whether it carries the human's
37
+ * decision out or stands in for it.
38
+ *
39
+ * `'decision'` — the answer, applied to the calls the park was about.
40
+ * `'recovery'` — the checkpoint's recorded and explicitly unknown outcomes,
41
+ * replayed so that nothing runs twice. The caller resolves the park in both
42
+ * cases — recovery answering the batch is what makes the question moot —
43
+ * but only the first may write the human's decision down as what ended it.
44
+ * Recording a decision recovery stood in for says the run carried out
45
+ * something it did not.
46
+ */
47
+ readonly source: 'decision' | 'recovery'
34
48
  /**
35
49
  * The checkpoint the park was recorded on, so the caller can clear it
36
50
  * once the decision has actually been applied. Leaving it outstanding
@@ -122,6 +136,7 @@ export function planPendingResume(
122
136
  if (!denials) return null
123
137
 
124
138
  return {
139
+ source: 'decision',
125
140
  checkpointId: checkpoint.id,
126
141
  assistant,
127
142
  response: synthesizeResponse(assistant),
@@ -131,6 +146,144 @@ export function planPendingResume(
131
146
  }
132
147
  }
133
148
 
149
+ /**
150
+ * The stable marker `supersededByRecovery` puts at the head of its reason.
151
+ *
152
+ * `resolvedAt` says a park ENDED; it does not say HOW, and `pause` is the
153
+ * action both endings share — `CheckpointManager.expire` records one for a
154
+ * park that ran out of time, this one records another for a park whose
155
+ * question crash recovery answered instead. A reader that tests
156
+ * `pending.decision.action` alone can tell neither from a run still holding
157
+ * the park, and the reason is the only field left to carry the difference.
158
+ *
159
+ * A constant rather than a sentence written at the call site, and a PREFIX
160
+ * rather than the whole string, because the sentence names which decision was
161
+ * superseded — informative to a person, unstable to a comparison. A consumer
162
+ * tests this; the tail is prose.
163
+ *
164
+ * Exported for the SDK's own readers. It is not on the package's public
165
+ * surface: a new field on the recorded decision would be, and this branch
166
+ * ships as a `patch`.
167
+ */
168
+ export const PARK_SUPERSEDED_BY_RECOVERY = 'crash-recovery-superseded'
169
+
170
+ /** Whether a recorded decision is the supersede marker rather than an answer. */
171
+ export function isSupersededByRecovery(decision: HITLResumeDecision | undefined): boolean {
172
+ return (
173
+ decision?.action === 'pause' && (decision.reason ?? '').startsWith(PARK_SUPERSEDED_BY_RECOVERY)
174
+ )
175
+ }
176
+
177
+ /**
178
+ * What to record on a park whose batch crash recovery answered instead of the
179
+ * decision.
180
+ *
181
+ * Neither half of the obvious record is honest. Writing the human's decision
182
+ * down would say the run carried it out, when the calls it named were answered
183
+ * with an explicitly UNKNOWN outcome and nothing they asked for happened —
184
+ * `planPendingResume` refused that decision in the first place, which is why
185
+ * recovery spoke at all. Writing nothing would lose the fact that somebody
186
+ * answered, and leave `pending.decision` meaning two different things.
187
+ *
188
+ * The vocabulary already has one shape for "this park ended and no decision
189
+ * was carried out": `CheckpointManager.expire` records a `pause` carrying the
190
+ * reason, and says why it is not an `abort` ("that would read as somebody
191
+ * having refused it"). This is that shape, with
192
+ * {@link PARK_SUPERSEDED_BY_RECOVERY} at the head of the reason so the fact is
193
+ * comparable rather than prose, and the decision it superseded after it so the
194
+ * answer a human gave is still on the record.
195
+ */
196
+ export function supersededByRecovery(decision: HITLResumeDecision): HITLResumeDecision {
197
+ return {
198
+ action: 'pause',
199
+ reason: `${PARK_SUPERSEDED_BY_RECOVERY}: crash recovery answered the tool batch this park asked about, so the decision that was given (${decision.action}) was not applied.`,
200
+ }
201
+ }
202
+
203
+ /**
204
+ * Whether the ordinary continue path carries `decision` out for a park that
205
+ * has no batch of tool calls to apply it to.
206
+ *
207
+ * {@link planPendingResume} covers the two arms whose decision has to REACH
208
+ * something: the calls a `tool_review` park is about, and the tool a
209
+ * `user_question` park is inside. An `iteration_checkpoint` park has neither,
210
+ * so it produces no plan — and "no plan" must not be read as "nothing
211
+ * happened". For this arm the decision IS the run's next move, and the loop
212
+ * that resumes carries it out by continuing; the park it answered therefore
213
+ * has to be resolved exactly as the other arms' are.
214
+ *
215
+ * The set is `handleHITLDecision`'s continue arm, deliberately: these are the
216
+ * decisions a resumed run carries out by going on. `pause` is not among them
217
+ * — it is "hold this, I am not answering now", which the live path leaves the
218
+ * park standing for, and a resumed run does not honour it either. Neither are
219
+ * `abort` and `reject_plan`: nothing on the resume path acts on them, so
220
+ * recording one as the park's answer would say the run carried out something
221
+ * it did not.
222
+ */
223
+ export function isCarriedOutByContinue(decision: HITLResumeDecision): boolean {
224
+ switch (decision.action) {
225
+ case 'continue':
226
+ case 'approve_tools':
227
+ case 'modify_tools':
228
+ case 'reject_tools':
229
+ case 'answer_question':
230
+ return true
231
+ default:
232
+ return false
233
+ }
234
+ }
235
+
236
+ /**
237
+ * Whether `decision` is a verdict on the question a `plan_approval` park asks.
238
+ *
239
+ * The plan arm was the one park `isCarriedOutByContinue` did not cover and
240
+ * nothing else did either, so a run resumed with `{action: 'approve_plan'}`
241
+ * completed with its park still outstanding: `findPendingCheckpoint` kept
242
+ * serving a plan nobody was waiting on, a second resume of the FINISHED run
243
+ * was refused `awaiting-decision`, and `prune`'s refusal to collect an
244
+ * unresolved park left the row uncollectable — with no `hitlParkTtlMs` there
245
+ * is no `deadlineAt`, so `expire` cannot reach it either.
246
+ *
247
+ * Both verdicts answer it, and that is the difference from `pause` (which
248
+ * HOLDS the park rather than answering it, on the live path and here) and
249
+ * from `abort` (which is not a verdict on the plan at all). What the resumed
250
+ * run can do about the answer afterwards is a separate question and not this
251
+ * predicate's: the record's `decision` is what the HUMAN answered, and a park
252
+ * is not made outstanding again by the new process having no plan to act on.
253
+ */
254
+ export function isPlanVerdict(decision: HITLResumeDecision): boolean {
255
+ return decision.action === 'approve_plan' || decision.action === 'reject_plan'
256
+ }
257
+
258
+ /**
259
+ * Whether a park of `type` is ANSWERED by `decision` on the resume path.
260
+ *
261
+ * The map from park to the decision that answers it, in one place, because
262
+ * each arm was added by a different fix and the two that were missed were
263
+ * missed by being absent rather than wrong. `tool_review` and `user_question`
264
+ * are deliberately not here: their decisions have to REACH something —
265
+ * `planPendingResume` applies them to the parked batch — and their parks are
266
+ * resolved where that plan is applied, not by this predicate.
267
+ *
268
+ * "Answered" is the ordinary continue path carrying the decision out, which
269
+ * for the cadence arm means the loop going on and for the plan arm means the
270
+ * verdict having been given. It does not mean the run did everything the
271
+ * decision implies — see {@link isPlanVerdict}.
272
+ */
273
+ export function answersParkOf(
274
+ parkType: HITLDecisionRequest['type'] | undefined,
275
+ decision: HITLResumeDecision,
276
+ ): boolean {
277
+ switch (parkType) {
278
+ case 'iteration_checkpoint':
279
+ return isCarriedOutByContinue(decision)
280
+ case 'plan_approval':
281
+ return isPlanVerdict(decision)
282
+ default:
283
+ return false
284
+ }
285
+ }
286
+
134
287
  /**
135
288
  * Resume a batch that parked inside a tool asking the user a question.
136
289
  *
@@ -183,6 +336,7 @@ function planQuestionResume(
183
336
  }
184
337
 
185
338
  return {
339
+ source: 'decision',
186
340
  checkpointId: checkpoint.id,
187
341
  assistant,
188
342
  response: synthesizeResponse(assistant),
@@ -228,6 +382,10 @@ export function planCrashResume(
228
382
  })
229
383
 
230
384
  return {
385
+ // Not the human's decision — recovery's own reading of the batch. The
386
+ // caller resolves the park either way and must not write the decision
387
+ // down as what ended it; see {@link supersededByRecovery}.
388
+ source: 'recovery',
231
389
  checkpointId: checkpoint.id,
232
390
  assistant,
233
391
  response: synthesizeResponse(assistant),
@@ -101,11 +101,18 @@ export interface AgentRunConfig {
101
101
 
102
102
  /**
103
103
  * After creating an iteration checkpoint, prune the run's checkpoint
104
- * set down to the newest N (oldest-first deletion across ALL of the
105
- * run's checkpoints, including tool-review/plan ones). Default
106
- * `undefined` — never prune, today's behavior. Each checkpoint copies
107
- * the full message array, so long tool-heavy runs grow O(iterations ×
108
- * history) without this.
104
+ * set down to the newest N. Default `undefined` — never prune, today's
105
+ * behavior. Each checkpoint copies the full message array, so long
106
+ * tool-heavy runs grow O(iterations × history) without this.
107
+ *
108
+ * Oldest-first by `createdAt`, across all of the run's checkpoints — but
109
+ * a checkpoint whose park is UNRESOLVED is never collected, whatever its
110
+ * age. Those rows are what `findPendingCheckpoint` serves to an approval
111
+ * queue and what `listExpiredParks` enumerates for a sweep, so pruning
112
+ * briefly holds more than N while a park is outstanding; the next prune
113
+ * after the park resolves — by `unpark`, or by `expire` for one that ran
114
+ * out of time — collects them. A host that needs the bound to hold
115
+ * regardless should sweep expired parks itself.
109
116
  */
110
117
  pruneKeepLast?: number
111
118