@namzu/sdk 39.0.0 → 40.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (183) hide show
  1. package/CHANGELOG.md +151 -0
  2. package/dist/connector/mcp/adapter.d.ts.map +1 -1
  3. package/dist/connector/mcp/adapter.js +20 -6
  4. package/dist/connector/mcp/adapter.js.map +1 -1
  5. package/dist/manager/run/persistence.d.ts +8 -0
  6. package/dist/manager/run/persistence.d.ts.map +1 -1
  7. package/dist/manager/run/persistence.js +12 -0
  8. package/dist/manager/run/persistence.js.map +1 -1
  9. package/dist/public-runtime.d.ts +3 -1
  10. package/dist/public-runtime.d.ts.map +1 -1
  11. package/dist/public-runtime.js +5 -1
  12. package/dist/public-runtime.js.map +1 -1
  13. package/dist/public-tools.d.ts +11 -0
  14. package/dist/public-tools.d.ts.map +1 -1
  15. package/dist/public-tools.js +14 -0
  16. package/dist/public-tools.js.map +1 -1
  17. package/dist/registry/tool/execute.d.ts.map +1 -1
  18. package/dist/registry/tool/execute.js +2 -3
  19. package/dist/registry/tool/execute.js.map +1 -1
  20. package/dist/registry/tool/portable.d.ts +65 -0
  21. package/dist/registry/tool/portable.d.ts.map +1 -0
  22. package/dist/registry/tool/portable.js +244 -0
  23. package/dist/registry/tool/portable.js.map +1 -0
  24. package/dist/registry/tool/schema.d.ts +32 -5
  25. package/dist/registry/tool/schema.d.ts.map +1 -1
  26. package/dist/registry/tool/schema.js +35 -9
  27. package/dist/registry/tool/schema.js.map +1 -1
  28. package/dist/registry/toolset/catalog.js +8 -8
  29. package/dist/registry/toolset/catalog.js.map +1 -1
  30. package/dist/runtime/jobs/awaited-jobs.d.ts +215 -0
  31. package/dist/runtime/jobs/awaited-jobs.d.ts.map +1 -0
  32. package/dist/runtime/jobs/awaited-jobs.js +259 -0
  33. package/dist/runtime/jobs/awaited-jobs.js.map +1 -0
  34. package/dist/runtime/jobs/registry.d.ts +33 -2
  35. package/dist/runtime/jobs/registry.d.ts.map +1 -1
  36. package/dist/runtime/jobs/registry.js +37 -0
  37. package/dist/runtime/jobs/registry.js.map +1 -1
  38. package/dist/runtime/query/executor.d.ts +28 -0
  39. package/dist/runtime/query/executor.d.ts.map +1 -1
  40. package/dist/runtime/query/executor.js +39 -1
  41. package/dist/runtime/query/executor.js.map +1 -1
  42. package/dist/runtime/query/file-evidence-context.d.ts.map +1 -1
  43. package/dist/runtime/query/file-evidence-context.js +159 -43
  44. package/dist/runtime/query/file-evidence-context.js.map +1 -1
  45. package/dist/runtime/query/file-evidence-replay.d.ts +260 -0
  46. package/dist/runtime/query/file-evidence-replay.d.ts.map +1 -0
  47. package/dist/runtime/query/file-evidence-replay.js +647 -0
  48. package/dist/runtime/query/file-evidence-replay.js.map +1 -0
  49. package/dist/runtime/query/file-evidence-seed.d.ts +50 -0
  50. package/dist/runtime/query/file-evidence-seed.d.ts.map +1 -0
  51. package/dist/runtime/query/file-evidence-seed.js +100 -0
  52. package/dist/runtime/query/file-evidence-seed.js.map +1 -0
  53. package/dist/runtime/query/index.d.ts.map +1 -1
  54. package/dist/runtime/query/index.js +94 -2
  55. package/dist/runtime/query/index.js.map +1 -1
  56. package/dist/runtime/query/iteration/index.d.ts +87 -9
  57. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  58. package/dist/runtime/query/iteration/index.js +193 -28
  59. package/dist/runtime/query/iteration/index.js.map +1 -1
  60. package/dist/runtime/query/iteration/phases/context.d.ts +10 -0
  61. package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
  62. package/dist/runtime/query/iteration/phases/context.js.map +1 -1
  63. package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
  64. package/dist/runtime/query/iteration/phases/tool-review.js +5 -1
  65. package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
  66. package/dist/runtime/query/plugin-hooks.d.ts +14 -0
  67. package/dist/runtime/query/plugin-hooks.d.ts.map +1 -1
  68. package/dist/runtime/query/plugin-hooks.js +18 -0
  69. package/dist/runtime/query/plugin-hooks.js.map +1 -1
  70. package/dist/runtime/query/repeat-call.d.ts +17 -4
  71. package/dist/runtime/query/repeat-call.d.ts.map +1 -1
  72. package/dist/runtime/query/repeat-call.js +26 -19
  73. package/dist/runtime/query/repeat-call.js.map +1 -1
  74. package/dist/runtime/query/steering.d.ts +11 -1
  75. package/dist/runtime/query/steering.d.ts.map +1 -1
  76. package/dist/runtime/query/steering.js +12 -1
  77. package/dist/runtime/query/steering.js.map +1 -1
  78. package/dist/runtime/query/tooling.d.ts +2 -0
  79. package/dist/runtime/query/tooling.d.ts.map +1 -1
  80. package/dist/runtime/query/tooling.js +1 -0
  81. package/dist/runtime/query/tooling.js.map +1 -1
  82. package/dist/scheduler/completion-inbox.d.ts +48 -2
  83. package/dist/scheduler/completion-inbox.d.ts.map +1 -1
  84. package/dist/scheduler/completion-inbox.js +102 -10
  85. package/dist/scheduler/completion-inbox.js.map +1 -1
  86. package/dist/tools/builtins/bash.d.ts.map +1 -1
  87. package/dist/tools/builtins/bash.js +4 -10
  88. package/dist/tools/builtins/bash.js.map +1 -1
  89. package/dist/tools/builtins/edit-apply.d.ts +126 -0
  90. package/dist/tools/builtins/edit-apply.d.ts.map +1 -0
  91. package/dist/tools/builtins/edit-apply.js +360 -0
  92. package/dist/tools/builtins/edit-apply.js.map +1 -0
  93. package/dist/tools/builtins/edit.d.ts +143 -1
  94. package/dist/tools/builtins/edit.d.ts.map +1 -1
  95. package/dist/tools/builtins/edit.js +37 -219
  96. package/dist/tools/builtins/edit.js.map +1 -1
  97. package/dist/tools/builtins/index.d.ts +1 -0
  98. package/dist/tools/builtins/index.d.ts.map +1 -1
  99. package/dist/tools/builtins/index.js +9 -3
  100. package/dist/tools/builtins/index.js.map +1 -1
  101. package/dist/tools/builtins/job.js +1 -1
  102. package/dist/tools/builtins/job.js.map +1 -1
  103. package/dist/tools/builtins/read-file.d.ts +2 -2
  104. package/dist/tools/builtins/read-file.d.ts.map +1 -1
  105. package/dist/tools/builtins/read-file.js +50 -65
  106. package/dist/tools/builtins/read-file.js.map +1 -1
  107. package/dist/tools/builtins/read-render.d.ts +56 -0
  108. package/dist/tools/builtins/read-render.d.ts.map +1 -0
  109. package/dist/tools/builtins/read-render.js +73 -0
  110. package/dist/tools/builtins/read-render.js.map +1 -0
  111. package/dist/tools/builtins/wait-for-job-bounds.d.ts +67 -0
  112. package/dist/tools/builtins/wait-for-job-bounds.d.ts.map +1 -0
  113. package/dist/tools/builtins/wait-for-job-bounds.js +108 -0
  114. package/dist/tools/builtins/wait-for-job-bounds.js.map +1 -0
  115. package/dist/tools/builtins/wait-for-job.d.ts +6 -0
  116. package/dist/tools/builtins/wait-for-job.d.ts.map +1 -0
  117. package/dist/tools/builtins/wait-for-job.js +162 -0
  118. package/dist/tools/builtins/wait-for-job.js.map +1 -0
  119. package/dist/tools/builtins/write-file.js +5 -0
  120. package/dist/tools/builtins/write-file.js.map +1 -1
  121. package/dist/tools/coordinator/index.d.ts.map +1 -1
  122. package/dist/tools/coordinator/index.js +1 -7
  123. package/dist/tools/coordinator/index.js.map +1 -1
  124. package/dist/tools/file-read-tracker.d.ts.map +1 -1
  125. package/dist/tools/file-read-tracker.js +88 -10
  126. package/dist/tools/file-read-tracker.js.map +1 -1
  127. package/dist/types/message/index.d.ts +1 -1
  128. package/dist/types/message/index.d.ts.map +1 -1
  129. package/dist/types/message/index.js +2 -0
  130. package/dist/types/message/index.js.map +1 -1
  131. package/dist/types/run/entity.d.ts +13 -0
  132. package/dist/types/run/entity.d.ts.map +1 -1
  133. package/dist/types/sandbox/index.d.ts +15 -14
  134. package/dist/types/sandbox/index.d.ts.map +1 -1
  135. package/dist/types/sandbox/index.js.map +1 -1
  136. package/dist/types/tool/index.d.ts +109 -0
  137. package/dist/types/tool/index.d.ts.map +1 -1
  138. package/dist/types/tool/index.js.map +1 -1
  139. package/dist/utils/env.d.ts +19 -0
  140. package/dist/utils/env.d.ts.map +1 -0
  141. package/dist/utils/env.js +25 -0
  142. package/dist/utils/env.js.map +1 -0
  143. package/package.json +1 -1
  144. package/src/connector/mcp/adapter.ts +20 -6
  145. package/src/manager/run/persistence.ts +12 -0
  146. package/src/public-runtime.ts +9 -1
  147. package/src/public-tools.ts +18 -0
  148. package/src/registry/tool/execute.ts +2 -4
  149. package/src/registry/tool/portable.ts +264 -0
  150. package/src/registry/tool/schema.ts +38 -8
  151. package/src/registry/toolset/catalog.ts +8 -9
  152. package/src/runtime/jobs/awaited-jobs.ts +271 -0
  153. package/src/runtime/jobs/registry.ts +50 -0
  154. package/src/runtime/query/executor.ts +49 -1
  155. package/src/runtime/query/file-evidence-context.ts +190 -46
  156. package/src/runtime/query/file-evidence-replay.ts +776 -0
  157. package/src/runtime/query/file-evidence-seed.ts +126 -0
  158. package/src/runtime/query/index.ts +104 -2
  159. package/src/runtime/query/iteration/index.ts +202 -28
  160. package/src/runtime/query/iteration/phases/context.ts +10 -0
  161. package/src/runtime/query/iteration/phases/tool-review.ts +4 -0
  162. package/src/runtime/query/plugin-hooks.ts +20 -0
  163. package/src/runtime/query/repeat-call.ts +28 -18
  164. package/src/runtime/query/steering.ts +11 -0
  165. package/src/runtime/query/tooling.ts +3 -0
  166. package/src/scheduler/completion-inbox.ts +105 -9
  167. package/src/tools/builtins/bash.ts +4 -10
  168. package/src/tools/builtins/edit-apply.ts +456 -0
  169. package/src/tools/builtins/edit.ts +39 -270
  170. package/src/tools/builtins/index.ts +9 -3
  171. package/src/tools/builtins/job.ts +1 -1
  172. package/src/tools/builtins/read-file.ts +56 -77
  173. package/src/tools/builtins/read-render.ts +104 -0
  174. package/src/tools/builtins/wait-for-job-bounds.ts +179 -0
  175. package/src/tools/builtins/wait-for-job.ts +184 -0
  176. package/src/tools/builtins/write-file.ts +5 -0
  177. package/src/tools/coordinator/index.ts +1 -7
  178. package/src/tools/file-read-tracker.ts +85 -7
  179. package/src/types/message/index.ts +2 -0
  180. package/src/types/run/entity.ts +14 -0
  181. package/src/types/sandbox/index.ts +15 -14
  182. package/src/types/tool/index.ts +104 -0
  183. package/src/utils/env.ts +23 -0
@@ -59,3 +59,23 @@ export function applyLifecycleHookResults(
59
59
  }
60
60
  return annotations
61
61
  }
62
+
63
+ /**
64
+ * What a `pre_tool_use` hook's SKIP goes back to the model as.
65
+ *
66
+ * A function rather than a template at the one call site, because the text is
67
+ * a SIGNAL as well as prose. A skipped call gets a non-error receipt — the
68
+ * hook refused it, nothing failed — so the transcript's only record that the
69
+ * tool never ran is this sentence. `file-evidence-replay.ts` reads it back to
70
+ * keep a skipped `write` from being replayed as a body the file now holds,
71
+ * and it can only do that while one function owns both the writing and the
72
+ * recognising.
73
+ */
74
+ export function skippedToolResultText(toolName: string, reason: string): string {
75
+ return `Tool ${toolName} skipped by plugin: ${reason}`
76
+ }
77
+
78
+ /** Whether `content` is the receipt {@link skippedToolResultText} writes for `toolName`. */
79
+ export function isSkippedToolResult(toolName: string, content: string): boolean {
80
+ return content.startsWith(skippedToolResultText(toolName, ''))
81
+ }
@@ -1,3 +1,4 @@
1
+ import { createRuntimeContextMessage } from '../../types/message/index.js'
1
2
  import type { Message } from '../../types/message/index.js'
2
3
  import { stableStringify } from './tool-grants.js'
3
4
 
@@ -128,11 +129,24 @@ export class RepeatCallTracker {
128
129
  }
129
130
 
130
131
  /**
131
- * Rides the notice out on the last `tool_result` of the batch.
132
+ * Rides the notice out on the last `tool_result` of the batch, same slot
133
+ * steering uses: a `tool_use` block must be answered by a `tool_result` with
134
+ * the same id, so a user message wedged between them is rejected by the
135
+ * provider outright.
132
136
  *
133
- * The same and only legal slot steering uses: a `tool_use` block must be
134
- * answered by a `tool_result` with the same id, so a user message wedged
135
- * between them is rejected by the provider outright.
137
+ * That slot only exists when the trailing result's content is plain text. A
138
+ * result answered with structured content (an image, a document, an MCP
139
+ * block) has a shape the model reads positionally, and appending a string to
140
+ * it is either dropped or corrupts the block — this used to mean the notice
141
+ * was simply dropped, on the theory that an advisory costs nothing to lose.
142
+ * It costs more than a refusal would: `RepeatCallTracker.record` already
143
+ * marked the threshold as announced the moment it fired, so a notice lost
144
+ * here never comes back, unlike steering, which can requeue and wait for a
145
+ * later plain-text result. The fallback instead rides out as its own
146
+ * `runtime-context` message placed AFTER the complete tool-result batch —
147
+ * never between a `tool_use` and its `tool_result`, so provider-required
148
+ * adjacency still holds — carrying that provenance so it is never mistaken
149
+ * for operator input (see `isOperatorUserMessage` in `steering.ts`).
136
150
  */
137
151
  export function attachRepeatNotice(
138
152
  messages: readonly Message[],
@@ -140,6 +154,8 @@ export function attachRepeatNotice(
140
154
  ): readonly Message[] {
141
155
  if (notices.length === 0) return messages
142
156
 
157
+ const noticeText = notices.map((n) => n.text).join('\n')
158
+
143
159
  let lastToolIndex = -1
144
160
  for (let index = messages.length - 1; index >= 0; index--) {
145
161
  if (messages[index]?.role === 'tool') {
@@ -147,19 +163,13 @@ export function attachRepeatNotice(
147
163
  break
148
164
  }
149
165
  }
150
- if (lastToolIndex === -1) return messages
151
-
152
- const target = messages[lastToolIndex] as Message
153
- // Text only, for the reason `attachSteering` gives: a result answered
154
- // with structured content has a shape the model reads positionally, and
155
- // appending a string to it is either dropped or corrupts the block. The
156
- // notice is advisory, so dropping it costs nothing a refusal would.
157
- if (typeof target.content !== 'string') return messages
158
-
159
- const next = [...messages]
160
- next[lastToolIndex] = {
161
- ...target,
162
- content: `${target.content}\n\n${notices.map((n) => n.text).join('\n')}`,
166
+ const target = lastToolIndex === -1 ? undefined : (messages[lastToolIndex] as Message)
167
+
168
+ if (target && typeof target.content === 'string') {
169
+ const next = [...messages]
170
+ next[lastToolIndex] = { ...target, content: `${target.content}\n\n${noticeText}` }
171
+ return next
163
172
  }
164
- return next
173
+
174
+ return [...messages, createRuntimeContextMessage(noticeText, 'repeat-call')]
165
175
  }
@@ -100,6 +100,16 @@ export function attachNotice(
100
100
  messages: readonly Message[],
101
101
  channel: SteeringChannel | undefined,
102
102
  format: (text: string) => string,
103
+ /**
104
+ * Told only when the text actually landed on a result the model will read.
105
+ *
106
+ * Not on the two paths above it: a batch with no tool result leaves the
107
+ * notice queued, and a result whose content is not a string puts it back.
108
+ * Draining is therefore not delivery, which matters to the caller that
109
+ * keeps its own record of what the text accounts for — `AwaitedJobs`
110
+ * drops an exit's entry here, and a premature drop would strand the exit.
111
+ */
112
+ onDelivered?: () => void,
103
113
  ): readonly Message[] {
104
114
  if (!channel?.pending) return messages
105
115
  let lastToolIndex = -1
@@ -119,6 +129,7 @@ export function attachNotice(
119
129
  }
120
130
  const next = [...messages]
121
131
  next[lastToolIndex] = { ...target, content: target.content + format(text) }
132
+ onDelivered?.()
122
133
  return next
123
134
  }
124
135
 
@@ -38,6 +38,8 @@ export interface ToolingBootstrapConfig {
38
38
  backgroundJobs?: BackgroundJobRegistry
39
39
  /** See `QueryParams.backgroundJobOwner`. */
40
40
  backgroundJobOwner?: string
41
+ /** Where `wait_for_job` records wait-intent; see the executor's own field. */
42
+ onJobAwaited?: (id: string) => void
41
43
  /** Where the `skill` tool reads from. */
42
44
  skills?: SkillRegistryRef
43
45
  /** How this run reaches the web. */
@@ -85,6 +87,7 @@ export class ToolingBootstrap {
85
87
  pluginManager: config.pluginManager,
86
88
  ...(config.backgroundJobs ? { backgroundJobs: config.backgroundJobs } : {}),
87
89
  ...(config.backgroundJobOwner ? { backgroundJobOwner: config.backgroundJobOwner } : {}),
90
+ ...(config.onJobAwaited ? { onJobAwaited: config.onJobAwaited } : {}),
88
91
  ...(config.skills ? { skills: config.skills } : {}),
89
92
  ...(config.web ? { web: config.web } : {}),
90
93
  ...(config.toolTimeoutMs !== undefined ? { toolTimeoutMs: config.toolTimeoutMs } : {}),
@@ -26,6 +26,23 @@ import { type Logger, resolveLogger } from '../utils/logger.js'
26
26
  */
27
27
  const UNOWNED_BUFFER_LIMIT = 32
28
28
 
29
+ /**
30
+ * How many owned tasks {@link CompletionInbox.describeOwnedWork} can name at
31
+ * once.
32
+ *
33
+ * Unrelated to {@link UNOWNED_BUFFER_LIMIT} — that one bounds a memory leak;
34
+ * this one bounds a request payload. `describeOwnedWork`'s output is admitted
35
+ * into the request through `appendWorkContext`'s 8,000-character / ~2,000-token
36
+ * gate (`runtime/query/iteration/index.ts`), and 16 tasks' worth of scheduler
37
+ * state fits that with room to spare, including the overflow note below.
38
+ *
39
+ * Running tasks fill these slots first — see {@link CompletionInbox.runningOwned}
40
+ * — so a task still going is never bumped out by launch count alone the way a
41
+ * single FIFO over every owned task used to bump it. Only once every running
42
+ * task has a slot do the most recently SETTLED tasks take what is left.
43
+ */
44
+ const OWNED_WORK_DISPLAY_LIMIT = 16
45
+
29
46
  /**
30
47
  * Completions that finished with nobody left to hear them.
31
48
  *
@@ -77,7 +94,30 @@ export class CompletionInbox {
77
94
  * takes one, and a host that owns a gateway naturally reuses it.
78
95
  */
79
96
  private readonly ours = new Set<TaskId>()
80
- private readonly recentOwned: TaskId[] = []
97
+ /**
98
+ * Owned tasks not yet observed to have settled, in launch order.
99
+ *
100
+ * A `Set`'s iteration order is insertion order, so the most recently
101
+ * launched entry is always last — {@link describeOwnedWork} reads that
102
+ * order back to front. An id leaves this set the moment its completion is
103
+ * observed, by whichever of the three paths sees it first: the gateway's
104
+ * broadcast, an announcement recovered from {@link unowned}, or a launch
105
+ * that finds the task already terminal. It is never evicted for any other
106
+ * reason, so a long-running task stays in here — and therefore visible —
107
+ * for exactly as long as it is actually running, regardless of how many
108
+ * other tasks this run launches meanwhile.
109
+ */
110
+ private readonly runningOwned = new Set<TaskId>()
111
+ /**
112
+ * Owned tasks observed to have settled, oldest first, bounded to
113
+ * {@link OWNED_WORK_DISPLAY_LIMIT}.
114
+ *
115
+ * This bound is display-driven, not a memory concern the way
116
+ * {@link UNOWNED_BUFFER_LIMIT} is: a settled task's own result already
117
+ * reached the model once (inline or as a notification), so dropping it
118
+ * from here loses a convenience, not the only record of it.
119
+ */
120
+ private readonly settledOwned: TaskId[] = []
81
121
  /**
82
122
  * Announcements that arrived before anyone said whose task it was.
83
123
  *
@@ -130,6 +170,7 @@ export class CompletionInbox {
130
170
  // finished its wait faster than the listener ran — is already
131
171
  // delivered. Nothing to queue.
132
172
  this.outstanding.delete(handle.taskId)
173
+ this.markSettled(handle.taskId)
133
174
  if (this.claimed.has(handle.taskId)) return
134
175
  this.unheard.set(handle.taskId, handle)
135
176
  for (const wake of this.arrivals) wake()
@@ -193,25 +234,49 @@ export class CompletionInbox {
193
234
  launched(taskId: TaskId): void {
194
235
  if (this.ours.has(taskId)) return
195
236
  this.ours.add(taskId)
196
- this.recentOwned.push(taskId)
197
- if (this.recentOwned.length > 16) this.recentOwned.shift()
198
237
 
199
- if (this.claimed.has(taskId) || this.unheard.has(taskId)) return
238
+ if (this.claimed.has(taskId) || this.unheard.has(taskId)) {
239
+ this.markSettled(taskId)
240
+ return
241
+ }
200
242
 
201
243
  const parked = this.unowned.get(taskId)
202
244
  if (parked) {
203
245
  this.unowned.delete(taskId)
204
246
  this.unheard.set(taskId, parked)
247
+ this.markSettled(taskId)
205
248
  for (const wake of [...this.arrivals]) wake()
206
249
  return
207
250
  }
208
251
 
209
252
  const settled = this.gateway?.getTask(taskId)
210
- if (!settled || !isTerminalAgentTaskState(settled.state)) return
253
+ if (!settled || !isTerminalAgentTaskState(settled.state)) {
254
+ // Not observed terminal by any of the checks above: still running,
255
+ // as far as this inbox knows.
256
+ this.runningOwned.add(taskId)
257
+ return
258
+ }
211
259
  this.unheard.set(taskId, settled)
260
+ this.markSettled(taskId)
212
261
  for (const wake of [...this.arrivals]) wake()
213
262
  }
214
263
 
264
+ /**
265
+ * Move an owned task from {@link runningOwned} to {@link settledOwned}.
266
+ *
267
+ * Idempotent: `claim` and the completion listener can both observe the
268
+ * same settlement (the race either can win), and re-marking a task that
269
+ * is already in {@link settledOwned} moves it to the most-recently-settled
270
+ * end rather than duplicating it there.
271
+ */
272
+ private markSettled(taskId: TaskId): void {
273
+ this.runningOwned.delete(taskId)
274
+ const at = this.settledOwned.indexOf(taskId)
275
+ if (at !== -1) this.settledOwned.splice(at, 1)
276
+ this.settledOwned.push(taskId)
277
+ if (this.settledOwned.length > OWNED_WORK_DISPLAY_LIMIT) this.settledOwned.shift()
278
+ }
279
+
215
280
  /**
216
281
  * Say that a task was launched with nothing waiting on it.
217
282
  *
@@ -274,6 +339,13 @@ export class CompletionInbox {
274
339
  this.claimed.add(taskId)
275
340
  this.unheard.delete(taskId)
276
341
  this.outstanding.delete(taskId)
342
+ // A tool only claims a result it is already holding, so a claim is
343
+ // itself proof of settlement — independent of whether the gateway's
344
+ // own broadcast has reached this inbox yet. Without this, a task
345
+ // claimed ahead of its announcement (see the race above) stayed in
346
+ // {@link runningOwned} until that announcement arrived, understating
347
+ // its own "still running" count in the meantime.
348
+ this.markSettled(taskId)
277
349
  }
278
350
 
279
351
  /** Whether anything is waiting to be told. */
@@ -281,10 +353,29 @@ export class CompletionInbox {
281
353
  return this.unheard.size > 0
282
354
  }
283
355
 
284
- /** Bounded, non-consuming scheduler observations. Delivery never establishes user-facing completion. */
356
+ /**
357
+ * Bounded, non-consuming scheduler observations. Delivery never
358
+ * establishes user-facing completion.
359
+ *
360
+ * Running tasks — most recently launched first, matching the order the
361
+ * rest of this projection has always used — fill the {@link
362
+ * OWNED_WORK_DISPLAY_LIMIT} slots first, so a task still going is named
363
+ * here for as long as it keeps running, no matter how many other tasks
364
+ * this run has since launched. The most recently SETTLED tasks fill
365
+ * whatever slots running tasks leave over. A settled task bumped out
366
+ * entirely is not reported missing — its own result already reached the
367
+ * model once — but a RUNNING task that does not fit is: the preamble
368
+ * says exactly how many, so the model never mistakes the cap for the
369
+ * task having ended.
370
+ */
285
371
  describeOwnedWork(): string | undefined {
286
372
  if (this.ours.size === 0) return
287
- const ids = this.recentOwned
373
+ const running = [...this.runningOwned].reverse()
374
+ const shownRunning = running.slice(0, OWNED_WORK_DISPLAY_LIMIT)
375
+ const stillRunningNotShown = running.length - shownRunning.length
376
+ const settled = [...this.settledOwned].reverse()
377
+ const shownSettled = settled.slice(0, OWNED_WORK_DISPLAY_LIMIT - shownRunning.length)
378
+ const ids = [...shownRunning, ...shownSettled]
288
379
  const tasks = ids.map((taskId) => {
289
380
  let handle = this.unheard.get(taskId)
290
381
  try {
@@ -300,7 +391,11 @@ export class CompletionInbox {
300
391
  resultDelivery: this.claimed.has(taskId) ? 'delivered-to-history' : 'not-delivered',
301
392
  }
302
393
  })
303
- return `Owned delegated work (current scheduler observations). Result delivery is not proof that you answered the original request. Answer the latest operator message and complete the requested synthesis of available results unless the operator cancelled or changed that request. Reuse delivered results when sufficient; do not relaunch completed work. If a delivered result is no longer visible, retrieve it by task ID. A terminal state with a non-end_turn stop reason is not successful task completion. Do not repeat a synthesis already given.\n${JSON.stringify({ tasks, omitted: this.ours.size - ids.length })}`
394
+ const overflow =
395
+ stillRunningNotShown > 0
396
+ ? ` ${shownRunning.length} running tasks are shown below, and ${stillRunningNotShown} more still running.`
397
+ : ''
398
+ return `Owned delegated work (current scheduler observations).${overflow} Result delivery is not proof that you answered the original request. Answer the latest operator message and complete the requested synthesis of available results unless the operator cancelled or changed that request. Reuse delivered results when sufficient; do not relaunch completed work. If a delivered result is no longer visible, retrieve it by task ID. A terminal state with a non-end_turn stop reason is not successful task completion. Do not repeat a synthesis already given.\n${JSON.stringify({ tasks, omitted: this.ours.size - ids.length })}`
304
399
  }
305
400
 
306
401
  /**
@@ -391,7 +486,8 @@ export class CompletionInbox {
391
486
  this.unheard.clear()
392
487
  this.outstanding.clear()
393
488
  this.ours.clear()
394
- this.recentOwned.length = 0
489
+ this.runningOwned.clear()
490
+ this.settledOwned.length = 0
395
491
  this.claimed.clear()
396
492
  this.unowned.clear()
397
493
  // Release anyone still waiting. A closed inbox would otherwise hold
@@ -5,6 +5,7 @@ import { SANDBOX_KILL_GRACE_MS } from '../../constants/sandbox/index.js'
5
5
  import { DANGEROUS_PATTERNS } from '../../constants/tools/index.js'
6
6
  import { killTree } from '../../process/kill-tree.js'
7
7
  import { subscribeToAbort } from '../../utils/abort.js'
8
+ import { readPositiveIntEnv } from '../../utils/env.js'
8
9
  import { defineTool } from '../defineTool.js'
9
10
  import { scrubInheritedEnv } from '../env-scrub.js'
10
11
 
@@ -57,13 +58,13 @@ const inputSchema = z.object({
57
58
  z.number().positive().max(MAX_BASH_TIMEOUT_MS).default(DEFAULT_BASH_TIMEOUT_MS),
58
59
  )
59
60
  .describe(
60
- `Command timeout in milliseconds. Default: ${DEFAULT_BASH_TIMEOUT_MS}, maximum: ${MAX_BASH_TIMEOUT_MS}. For work that legitimately runs longer than the maximum, set run_in_background and poll with the \`job\` tool, rather than holding the turn open.`,
61
+ `Command timeout in milliseconds. Default: ${DEFAULT_BASH_TIMEOUT_MS}, maximum: ${MAX_BASH_TIMEOUT_MS}. For work that legitimately runs longer than the maximum, set run_in_background and await it with \`wait_for_job\`, rather than holding the turn open.`,
61
62
  ),
62
63
  run_in_background: z
63
64
  .boolean()
64
65
  .optional()
65
66
  .describe(
66
- 'Start the command as a background job and return its id immediately, instead of waiting. The turn is not held open; read its output with the `job` tool. Use for watchers, dev servers and long builds. Do NOT write `cmd &` yourself — under the sandbox the shell that backgrounds it exits immediately and takes the job with it.',
67
+ 'Start the command as a background job and return its id immediately, instead of waiting. The turn is not held open; await its completion with `wait_for_job` — one call, no waiting turns. Use `job` with action "read" only for incremental output while it keeps running, or to pick up after a `wait_for_job` call times out. Use for watchers, dev servers and long builds. Do NOT write `cmd &` yourself — under the sandbox the shell that backgrounds it exits immediately and takes the job with it.',
67
68
  ),
68
69
  })
69
70
 
@@ -343,7 +344,7 @@ export const BashTool = defineTool({
343
344
  })
344
345
  return {
345
346
  success: true,
346
- output: `Started background job ${job.id}. Read its output with the \`job\` tool: {"action":"read","id":"${job.id}"}.`,
347
+ output: `Started background job ${job.id}. Await its completion with wait_for_job: {"id":"${job.id}"}. Use job with action "read" only for output while it keeps running.`,
347
348
  data: { jobId: job.id, background: true },
348
349
  }
349
350
  } catch (err) {
@@ -539,10 +540,3 @@ function describeWithheldEnv(dropped: readonly string[]): string {
539
540
  const names = rest > 0 ? `${shown}, and ${rest} more` : shown
540
541
  return `NOTE: ${dropped.length} credential-shaped environment variable(s) were withheld from this command and are unset rather than empty: ${names}. A host that means this command to have one passes it explicitly.`
541
542
  }
542
-
543
- function readPositiveIntEnv(key: string, fallback: number): number {
544
- const value = process.env[key]?.trim()
545
- if (!value) return fallback
546
- const parsed = Number(value)
547
- return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : fallback
548
- }