@gajae-code/agent-core 0.11.11 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,10 +2,20 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.1] - 2026-07-29
6
+ - Agent session configuration can carry an explicit first-event stream timeout while preserving provider defaults when the setting is absent.
7
+
8
+ ### Fixed
9
+
10
+ - The `invalid_prompt` circuit breaker no longer replays the rejected turn on its repaired resend. The streaming path commits the failed assistant message to the context before the breaker runs, so the one repaired resend re-sent that errored turn as if the model had spoken it — re-triggering `Request blocked (code=invalid_prompt)` and leaving a second assistant tail that no continuation can resume from. The breaker now repairs and resends only the history that preceded the rejection.
11
+ - Compaction pruning now protects the newest two user/`bashExecution` turns, uses conservative read supersession, preserves bounded error-first diagnostics, and exposes reversible artifact-backed originals with exact savings accounting.
12
+
5
13
  ## [0.11.11] - 2026-07-26
6
14
 
7
15
  ### Fixed
8
16
 
17
+ - Managed runs now release their logical-run ownership before terminal observers are notified, so terminal overflow recovery cannot leave a stale owner behind.
18
+ - The OpenAI remote-compaction endpoint is now resolved from trusted environment sources only. `OPENAI_BASE_URL` was read through the merged view that includes the caller's `cwd/.env`, so a repository could redirect compaction requests that carry the OpenAI credential; it now uses the non-project resolver, leaving shell and user-level configuration unchanged.
9
19
  - Repeated malformed tool calls now get one tool-free recovery response, preventing argument-validation loops from ending without an answer while leaving ordinary execution-error retries unchanged. The recovery turn commits its assistant to the durable context, forces `toolChoice: "none"` alongside an empty tool list without consuming a queued tool choice, and never executes a tool call it did not advertise. Its recovery prompt is request-only, so append-only tool prefixes stay stable and the durable message log is unchanged.
10
20
  - Argument-validation loops now reach a deterministic terminal state. If a model keeps emitting only malformed tool calls after the one-shot recovery turn, the run stops with an explanatory error instead of calling the provider indefinitely. The bound counts consecutive all-malformed turns rather than repeated argument signatures, so a model rotating invalid argument shapes is bounded too; any healthy tool turn resets it.
11
21
 
@@ -4,7 +4,7 @@
4
4
  import { type AssistantMessage, type AssistantMessageEvent, type CursorExecHandlers, type CursorToolResultHandler, type Effort, type ImageContent, type Message, type Model, type ProviderSessionState, type ServiceTier, type SimpleStreamOptions, type ThinkingBudgets, type ToolChoice } from "@gajae-code/ai";
5
5
  import type { AppendOnlyContextManager } from "./append-only-context";
6
6
  import type { HarmonyAuditEvent } from "./harmony-leak";
7
- import type { AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, ManagedLogicalRunId, RunTerminalRequest, StreamFn, ToolCallContext } from "./types";
7
+ import type { AgentEvent, AgentLoopConfig, AgentMessage, AgentState, AgentTool, AgentToolContext, ManagedLogicalRunId, RunResourceLedger, RunTerminalRequest, StreamFn, ToolCallContext } from "./types";
8
8
  /**
9
9
  * Whether persisted history ends at a point where a new model turn can resume.
10
10
  * Assistant-ended histories require an in-memory queued message and are handled
@@ -127,6 +127,8 @@ export interface AgentOptions {
127
127
  requestMaxRetries?: number;
128
128
  /** Provider stream replay retry budget. Counts retries, not the initial attempt. */
129
129
  streamMaxRetries?: number;
130
+ /** Explicit first-event stream watchdog override in milliseconds. Set to 0 to disable. */
131
+ streamFirstEventTimeoutMs?: number;
130
132
  /**
131
133
  * Provides tool execution context, resolved per tool call.
132
134
  * Use for late-bound UI or session state access.
@@ -191,6 +193,7 @@ export type AgentQueueSnapshot = {
191
193
  export declare class Agent {
192
194
  #private;
193
195
  get intentTracing(): boolean;
196
+ readonly resourceLedger: RunResourceLedger;
194
197
  streamFn: StreamFn;
195
198
  getApiKey?: (provider: string) => Promise<string | undefined> | string | undefined;
196
199
  getAuthCredentialType?: (provider: string) => "api_key" | "oauth" | undefined;
@@ -314,6 +317,8 @@ export declare class Agent {
314
317
  set requestMaxRetries(value: number | undefined);
315
318
  get streamMaxRetries(): number | undefined;
316
319
  set streamMaxRetries(value: number | undefined);
320
+ get streamFirstEventTimeoutMs(): number | undefined;
321
+ set streamFirstEventTimeoutMs(value: number | undefined);
317
322
  get state(): AgentState;
318
323
  get contextRevision(): number;
319
324
  get appendOnlyContext(): AppendOnlyContextManager | undefined;
@@ -415,6 +420,8 @@ export declare class Agent {
415
420
  waitForIdle(): Promise<void>;
416
421
  /** The active per-attempt run identifier. */
417
422
  get activeRunId(): number | undefined;
423
+ /** Stable resource ownership identifier for the active prompt run. */
424
+ get activeResourceRunId(): string | undefined;
418
425
  /**
419
426
  * Stable identifier for the active managed logical run, shared by every retry
420
427
  * attempt. Pass this value to requestRunTerminal(); never retain activeRunId
@@ -39,6 +39,8 @@ export interface RemoteCompactionResponse {
39
39
  shortSummary?: string;
40
40
  }
41
41
  export declare function shouldUseOpenAiRemoteCompaction(model: Model): boolean;
42
+ /** Test seam: the compaction endpoint as resolved from trusted env. */
43
+ export declare function resolveOpenAiCompactEndpointForTest(model: Model, authCredentialType?: "api_key" | "oauth"): string;
42
44
  export declare function getPreservedOpenAiRemoteCompactionData(preserveData: Record<string, unknown> | undefined): OpenAiRemoteCompactionPreserveData | undefined;
43
45
  export declare function withOpenAiRemoteCompactionPreserveData(preserveData: Record<string, unknown> | undefined, remoteCompaction: OpenAiRemoteCompactionPreserveData | undefined): Record<string, unknown> | undefined;
44
46
  export declare function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number;
@@ -15,6 +15,8 @@ export interface PruneConfig {
15
15
  minimumSavings: number;
16
16
  /** Tool names that should never be pruned. */
17
17
  protectedTools: string[];
18
+ /** Number of newest user turns whose tool outputs must remain intact. Defaults to 2. */
19
+ protectRecentTurns?: number;
18
20
  /**
19
21
  * Tools in `protectedTools` whose protection is waived once the result is
20
22
  * superseded (a later result for the same target, or a later successful
@@ -24,9 +26,18 @@ export interface PruneConfig {
24
26
  staleOverridableTools?: string[];
25
27
  }
26
28
  export declare const DEFAULT_PRUNE_CONFIG: PruneConfig;
29
+ export interface PrunedOriginal {
30
+ entryId: string;
31
+ toolName?: string;
32
+ originalText: string;
33
+ tokens: number;
34
+ /** Whether originalText captures all-text result content without omission. */
35
+ complete?: boolean;
36
+ }
27
37
  export interface PruneResult {
28
38
  prunedCount: number;
29
39
  tokensSaved: number;
40
+ originals: PrunedOriginal[];
30
41
  /**
31
42
  * The mutated message entries. Callers whose entry source returns
32
43
  * materialized copies (not live references) must write these back into
@@ -45,9 +56,11 @@ export interface AssistantArgumentPruneResult {
45
56
  }
46
57
  export declare function pruneAssistantToolArguments(entries: SessionEntry[], config?: PruneConfig): AssistantArgumentPruneResult;
47
58
  /**
48
- * Estimate the token savings {@link pruneToolOutputs} would achieve, without
49
- * mutating any entry. Returns 0 savings when below the configured minimum so the
50
- * caller sees the same gate the real prune enforces.
59
+ * Estimate the conservative final token savings {@link pruneToolOutputs} would
60
+ * achieve, without mutating entries or invoking the artifact-reference planner.
61
+ * When `artifactRefMaxChars` is present, the estimate budgets that full length
62
+ * for every complete candidate so the real artifact-backed prune cannot save
63
+ * less than the estimate.
51
64
  */
52
65
  export declare function estimateToolOutputPruneSavings(entries: SessionEntry[], config?: PruneConfig, options?: PruneToolOutputsOptions): {
53
66
  prunableCount: number;
@@ -69,5 +82,18 @@ export declare function shouldRunMaintenancePrune(args: {
69
82
  export interface PruneToolOutputsOptions {
70
83
  /** Lower the usual minimum only when the caller is already over its compaction threshold. */
71
84
  relaxedMinimum?: number;
85
+ /**
86
+ * Conservative maximum ASCII length of every planned artifact reference.
87
+ * Required when `artifactRef` is provided so estimation and final admission
88
+ * use the same worst-case notice size.
89
+ */
90
+ artifactRefMaxChars?: number;
91
+ /**
92
+ * Plan a numeric `artifact://<id>` reference for a candidate's original
93
+ * text. The callback may reserve an in-memory identifier, but MUST NOT publish
94
+ * files or mutate session entries; publish only the originals returned by a
95
+ * successful {@link pruneToolOutputs} result.
96
+ */
97
+ artifactRef?: (candidate: PrunedOriginal) => string | undefined;
72
98
  }
73
99
  export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig, options?: PruneToolOutputsOptions): PruneResult;
@@ -6,6 +6,7 @@ export * from "./harmony-leak";
6
6
  export * from "./image-placeholder-guard";
7
7
  export * from "./proxy";
8
8
  export * from "./run-collector";
9
+ export * from "./run-resource-ledger";
9
10
  export * from "./telemetry";
10
11
  export * from "./thinking";
11
12
  export * from "./types";
@@ -0,0 +1,2 @@
1
+ import type { RunResourceLedger } from "./types";
2
+ export declare function createRunResourceLedger(): RunResourceLedger;
@@ -7,6 +7,33 @@ import type { AgentTelemetryConfig } from "./telemetry";
7
7
  export type StreamFn = (...args: Parameters<typeof streamSimple>) => AssistantMessageEventStream | Promise<AssistantMessageEventStream>;
8
8
  /** Stable identifier for a managed logical run, shared by all of its retry attempts. */
9
9
  export type ManagedLogicalRunId = number;
10
+ /** A resource owned by a prompt run until its promise settles. */
11
+ export type RunResourceKind = "provider_factory" | "provider_iterator" | "tool" | "post_prompt";
12
+ export interface RunResourceEntry {
13
+ id: string;
14
+ kind: RunResourceKind;
15
+ label: string;
16
+ registeredAt: number;
17
+ }
18
+ export type RunSettlementProof = {
19
+ status: "settled";
20
+ } | {
21
+ status: "unfenced";
22
+ pending: RunResourceEntry[];
23
+ };
24
+ export interface RunResourceLedger {
25
+ /** Reserve a run handle before publishing its `agent_start` event. */
26
+ open(resourceRunId: string): void;
27
+ track(resourceRunId: string, kind: RunResourceKind, label: string, settled: PromiseLike<unknown>): void;
28
+ pending(resourceRunId: string): RunResourceEntry[];
29
+ /** Seal a run after terminal event publication; only sealed empty runs settle. */
30
+ seal(resourceRunId: string): void;
31
+ waitForSettlement(resourceRunId: string, options: {
32
+ graceMs: number;
33
+ }): Promise<RunSettlementProof>;
34
+ /** Terminally detach a run; its bounded tombstone remains unfenced forever. */
35
+ quarantine(resourceRunId: string): RunResourceEntry[];
36
+ }
10
37
  /** Terminal completion requested for a logical run. */
11
38
  export interface RunTerminalRequest {
12
39
  stopReason: "cancelled" | "error" | "exhausted";
@@ -302,6 +329,13 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
302
329
  * capture, cost estimator, agent identity).
303
330
  */
304
331
  telemetry?: AgentTelemetryConfig;
332
+ /**
333
+ * Optional prompt-run resource ownership ledger. Provider and scheduler-level tool
334
+ * work is tracked until its owned lifecycle promise settles.
335
+ */
336
+ resourceLedger?: RunResourceLedger;
337
+ /** Stable resource ownership identifier for this prompt run. */
338
+ resourceRunId?: string;
305
339
  }
306
340
  /**
307
341
  * Batch/sequencing metadata for the tool call currently being processed.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/agent-core",
4
- "version": "0.11.11",
4
+ "version": "0.12.1",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -32,9 +32,9 @@
32
32
  "fmt": "biome format --write ."
33
33
  },
34
34
  "dependencies": {
35
- "@gajae-code/ai": "0.11.11",
36
- "@gajae-code/natives": "0.11.11",
37
- "@gajae-code/utils": "0.11.11",
35
+ "@gajae-code/ai": "0.12.1",
36
+ "@gajae-code/natives": "0.12.1",
37
+ "@gajae-code/utils": "0.12.1",
38
38
  "@opentelemetry/api": "^1.9.0"
39
39
  },
40
40
  "devDependencies": {
package/src/agent-loop.ts CHANGED
@@ -305,6 +305,7 @@ export function agentLoop(
305
305
  ? new ManagedAttemptTransaction(stream, config.onAssistantMessageEvent, config.model)
306
306
  : undefined;
307
307
  const attemptStream = transaction ?? stream;
308
+ openResourceRun(config);
308
309
  if (!config.fallbackManaged || emitManagedAgentStart) stream.push({ type: "agent_start" });
309
310
  attemptStream.push({ type: "turn_start" });
310
311
  for (const prompt of prompts) {
@@ -315,6 +316,7 @@ export function agentLoop(
315
316
  try {
316
317
  await runLoop(currentContext, newMessages, config, signal, stream, streamFn, transaction);
317
318
  } catch (err) {
319
+ if (config.resourceLedger && config.resourceRunId) config.resourceLedger.seal(config.resourceRunId);
318
320
  stream.fail(err);
319
321
  }
320
322
  })();
@@ -354,12 +356,14 @@ export function agentLoopContinue(
354
356
  ? new ManagedAttemptTransaction(stream, config.onAssistantMessageEvent, config.model)
355
357
  : undefined;
356
358
  const attemptStream = transaction ?? stream;
359
+ openResourceRun(config);
357
360
  if (!config.fallbackManaged || emitManagedAgentStart) stream.push({ type: "agent_start" });
358
361
  attemptStream.push({ type: "turn_start" });
359
362
 
360
363
  try {
361
364
  await runLoop(currentContext, newMessages, config, signal, stream, streamFn, transaction);
362
365
  } catch (err) {
366
+ if (config.resourceLedger && config.resourceRunId) config.resourceLedger.seal(config.resourceRunId);
363
367
  stream.fail(err);
364
368
  }
365
369
  })();
@@ -374,6 +378,21 @@ function createAgentStream(): EventStream<AgentEvent, AgentMessage[]> {
374
378
  );
375
379
  }
376
380
 
381
+ function openResourceRun(config: AgentLoopConfig): void {
382
+ if (config.resourceLedger && config.resourceRunId) config.resourceLedger.open(config.resourceRunId);
383
+ }
384
+
385
+ function publishAgentEnd(
386
+ stream: EventStream<AgentEvent, AgentMessage[]>,
387
+ config: AgentLoopConfig,
388
+ event: Extract<AgentEvent, { type: "agent_end" }>,
389
+ ): void {
390
+ stream.push(event);
391
+ if (event.stopReason !== "maintenance" && config.resourceLedger && config.resourceRunId) {
392
+ config.resourceLedger.seal(config.resourceRunId);
393
+ }
394
+ }
395
+
377
396
  /**
378
397
  * Hard work budget for one degraded snapshot: every visited node AND every
379
398
  * enumerated own key is debited against this budget before it is processed
@@ -1475,7 +1494,19 @@ async function runLoopBody(
1475
1494
  isInvalidPromptError(message)
1476
1495
  ) {
1477
1496
  invalidPromptRepairAttempted = true;
1478
- if (repairInvalidPromptHistory(currentContext.messages)) {
1497
+ // The rejected turn was already committed to the context by the
1498
+ // streaming path. Repair (and resend) only the history that
1499
+ // preceded it: replaying an errored assistant turn re-poisons the
1500
+ // request and leaves a second assistant tail behind, which no
1501
+ // continuation can resume from.
1502
+ const rejectedIndex = currentContext.messages.length - 1;
1503
+ const rejectedCommitted =
1504
+ rejectedIndex >= 0 && currentContext.messages[rejectedIndex]?.role === "assistant";
1505
+ const retained = rejectedCommitted
1506
+ ? currentContext.messages.slice(0, rejectedIndex)
1507
+ : currentContext.messages;
1508
+ if (repairInvalidPromptHistory(retained)) {
1509
+ if (rejectedCommitted) currentContext.messages.splice(rejectedIndex, 1);
1479
1510
  continue;
1480
1511
  }
1481
1512
  }
@@ -1558,7 +1589,7 @@ async function runLoopBody(
1558
1589
  });
1559
1590
  }
1560
1591
  stream.push({ type: "turn_end", message, toolResults });
1561
- stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1592
+ publishAgentEnd(stream, config, buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1562
1593
  stream.end(newMessages);
1563
1594
  return;
1564
1595
  }
@@ -1638,7 +1669,7 @@ async function runLoopBody(
1638
1669
  pendingMessages = (await config.getSteeringMessages?.()) || [];
1639
1670
  if (pendingMessages.length > 0) continue;
1640
1671
  if (config.shouldPause?.()) {
1641
- stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count, "paused"));
1672
+ publishAgentEnd(stream, config, buildAgentEndEvent(newMessages, telemetry, stepCounter.count, "paused"));
1642
1673
  stream.end(newMessages);
1643
1674
  return;
1644
1675
  }
@@ -1659,7 +1690,7 @@ async function runLoopBody(
1659
1690
  message.errorMessage = message.errorMessage
1660
1691
  ? `${message.errorMessage} | ${breakerMessage}`
1661
1692
  : breakerMessage;
1662
- stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1693
+ publishAgentEnd(stream, config, buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1663
1694
  stream.end(newMessages);
1664
1695
  return;
1665
1696
  }
@@ -1668,7 +1699,7 @@ async function runLoopBody(
1668
1699
  // Agent would stop here. Check for follow-up messages.
1669
1700
  await config.onBeforeYield?.();
1670
1701
  if (config.shouldPause?.()) {
1671
- stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count, "paused"));
1702
+ publishAgentEnd(stream, config, buildAgentEndEvent(newMessages, telemetry, stepCounter.count, "paused"));
1672
1703
  stream.end(newMessages);
1673
1704
  return;
1674
1705
  }
@@ -1683,7 +1714,7 @@ async function runLoopBody(
1683
1714
  break;
1684
1715
  }
1685
1716
 
1686
- stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1717
+ publishAgentEnd(stream, config, buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
1687
1718
  stream.end(newMessages);
1688
1719
  }
1689
1720
 
@@ -1831,24 +1862,64 @@ async function streamAssistantResponse(
1831
1862
  try {
1832
1863
  return await runInActiveSpan(chatSpan, async () => {
1833
1864
  const fallbackAttempt = config.fallbackManaged ? config.nextFallbackAttempt?.(config.model) : undefined;
1834
- const response = await streamFunction(config.model, llmContext, {
1835
- ...config,
1836
- fallbackAttempt,
1837
- apiKey: resolvedApiKey,
1838
- authCredentialType,
1839
- metadata: resolvedMetadata,
1840
- sessionId: config.providerSessionId ?? config.sessionId,
1841
- toolChoice: effectiveToolChoice,
1842
- reasoning: effectiveReasoning,
1843
- temperature: effectiveTemperature,
1844
- signal: requestSignal,
1845
- onResponse: captureOnResponse,
1865
+ const responsePromise = Promise.resolve().then(() =>
1866
+ streamFunction(config.model, llmContext, {
1867
+ ...config,
1868
+ fallbackAttempt,
1869
+ apiKey: resolvedApiKey,
1870
+ authCredentialType,
1871
+ metadata: resolvedMetadata,
1872
+ sessionId: config.providerSessionId ?? config.sessionId,
1873
+ toolChoice: effectiveToolChoice,
1874
+ reasoning: effectiveReasoning,
1875
+ temperature: effectiveTemperature,
1876
+ signal: requestSignal,
1877
+ onResponse: captureOnResponse,
1878
+ }),
1879
+ );
1880
+ const { promise: iteratorSettled, resolve: settleIterator } = Promise.withResolvers<void>();
1881
+ let responseResultPromise: Promise<AssistantMessage> | undefined;
1882
+ let responseForResult: { result(): Promise<AssistantMessage> } | undefined;
1883
+ const getResponseResult = (): Promise<AssistantMessage> =>
1884
+ (responseResultPromise ??= Promise.resolve().then(() => responseForResult!.result()));
1885
+ const providerLifecycle = responsePromise.then(async response => {
1886
+ responseForResult = response;
1887
+ await iteratorSettled;
1888
+ await Promise.allSettled([getResponseResult()]);
1846
1889
  });
1890
+ if (config.resourceLedger && config.resourceRunId) {
1891
+ // One ownership spans factory creation, iterator close, and trailing result.
1892
+ config.resourceLedger.track(
1893
+ config.resourceRunId,
1894
+ "provider_factory",
1895
+ `${config.model.provider}/${config.model.id}`,
1896
+ providerLifecycle,
1897
+ );
1898
+ }
1899
+ const response = await responsePromise;
1900
+ responseForResult = response;
1847
1901
 
1848
1902
  let partialMessage: AssistantMessage | null = null;
1849
1903
  let addedPartial = false;
1850
1904
 
1851
1905
  const responseIterator = response[Symbol.asyncIterator]();
1906
+ let iteratorClosed = false;
1907
+ const closeIterator = (): void => {
1908
+ if (iteratorClosed) return;
1909
+ iteratorClosed = true;
1910
+
1911
+ void Promise.resolve()
1912
+ .then(() => responseIterator.return?.())
1913
+ .then(
1914
+ () => settleIterator(),
1915
+ () => settleIterator(),
1916
+ );
1917
+ };
1918
+ const finishResponse = async (): Promise<AssistantMessage> => {
1919
+ closeIterator();
1920
+ await iteratorSettled;
1921
+ return getResponseResult();
1922
+ };
1852
1923
 
1853
1924
  // Set up a single abort race: register the abort listener once for the whole
1854
1925
  // stream and reuse the same race promise for every iterator.next() instead of
@@ -1857,6 +1928,7 @@ async function streamAssistantResponse(
1857
1928
  let detachAbortListener: (() => void) | undefined;
1858
1929
  if (requestSignal) {
1859
1930
  if (requestSignal.aborted) {
1931
+ closeIterator();
1860
1932
  const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream);
1861
1933
  await finishChat(aborted);
1862
1934
  return aborted;
@@ -1874,7 +1946,7 @@ async function streamAssistantResponse(
1874
1946
  if (abortRacePromise) {
1875
1947
  const result = await Promise.race([responseIterator.next(), abortRacePromise]);
1876
1948
  if (result === ABORTED) {
1877
- responseIterator.return?.()?.catch(() => {});
1949
+ closeIterator();
1878
1950
  const aborted = emitAbortedAssistantMessage(partialMessage, addedPartial, context, config, stream);
1879
1951
  await finishChat(aborted);
1880
1952
  return aborted;
@@ -1888,7 +1960,11 @@ async function streamAssistantResponse(
1888
1960
  await finishChat(aborted);
1889
1961
  return aborted;
1890
1962
  }
1891
- if (next.done) break;
1963
+ if (next.done) {
1964
+ iteratorClosed = true;
1965
+ settleIterator();
1966
+ break;
1967
+ }
1892
1968
 
1893
1969
  const event = next.value;
1894
1970
 
@@ -1937,8 +2013,8 @@ async function streamAssistantResponse(
1937
2013
  case "done":
1938
2014
  case "error": {
1939
2015
  const finalMessage = config.fallbackManaged
1940
- ? managedAssistantShell(await response.result(), config.model)
1941
- : await response.result();
2016
+ ? managedAssistantShell(await finishResponse(), config.model)
2017
+ : await finishResponse();
1942
2018
  if (addedPartial) {
1943
2019
  context.messages[context.messages.length - 1] = finalMessage;
1944
2020
  } else {
@@ -1955,11 +2031,12 @@ async function streamAssistantResponse(
1955
2031
  }
1956
2032
  } finally {
1957
2033
  detachAbortListener?.();
2034
+ closeIterator();
1958
2035
  }
1959
2036
 
1960
2037
  const trailing = config.fallbackManaged
1961
- ? managedAssistantShell(await response.result(), config.model)
1962
- : await response.result();
2038
+ ? managedAssistantShell(await finishResponse(), config.model)
2039
+ : await finishResponse();
1963
2040
  await finishChat(trailing);
1964
2041
  return trailing;
1965
2042
  });
@@ -2142,10 +2219,8 @@ async function executeToolCalls(
2142
2219
  const runTool = async (record: (typeof records)[number], index: number): Promise<void> => {
2143
2220
  if (interruptState.triggered) {
2144
2221
  // Skip both span emission and the collector orphan record here. The
2145
- // tail sweep below (after `Promise.allSettled`) is the single path
2146
- // that handles "no result message was produced" — it calls
2147
- // `recordSkippedTool` and `emitToolResult` once per record, so any
2148
- // work we did here would double-count.
2222
+ // scheduler-task finalizer emits the skipped result and collector record;
2223
+ // the tail sweep below remains a defensive fallback for unexpected throws.
2149
2224
  record.skipped = true;
2150
2225
  return;
2151
2226
  }
@@ -2261,7 +2336,7 @@ async function executeToolCalls(
2261
2336
  toolCalls: toolCallInfos,
2262
2337
  })
2263
2338
  : undefined;
2264
- const rawResult = await tool.execute(
2339
+ const execution = tool.execute(
2265
2340
  toolCall.id,
2266
2341
  transformToolCallArguments ? transformToolCallArguments(effectiveArgs, toolCall.name) : effectiveArgs,
2267
2342
  tool.nonAbortable ? undefined : toolSignal,
@@ -2276,6 +2351,7 @@ async function executeToolCalls(
2276
2351
  },
2277
2352
  toolContext,
2278
2353
  );
2354
+ const rawResult = await execution;
2279
2355
  const coerced = coerceToolResult(rawResult);
2280
2356
  result = coerced.result;
2281
2357
  if (coerced.malformed || result.isError) isError = true;
@@ -2359,8 +2435,30 @@ async function executeToolCalls(
2359
2435
  const record = records[index];
2360
2436
  const concurrency = record.tool?.concurrency ?? "shared";
2361
2437
  const start = concurrency === "exclusive" ? Promise.all([lastExclusive, ...sharedTasks]) : lastExclusive;
2362
- const task = start.then(() => runTool(record, index));
2438
+ const task = start
2439
+ .then(() => runTool(record, index))
2440
+ .finally(() => {
2441
+ // Scheduler ownership includes dependency waits and the fallback skip
2442
+ // emission, not only tool.execute().
2443
+ if (!record.toolResultMessage) {
2444
+ record.skipped = true;
2445
+ recordSkippedTool(telemetry, {
2446
+ toolCallId: record.toolCall.id,
2447
+ toolName: record.toolCall.name,
2448
+ status: "skipped",
2449
+ });
2450
+ emitToolResult(record, createSkippedToolResult(), true);
2451
+ }
2452
+ });
2363
2453
  tasks.push(task);
2454
+ if (config.resourceLedger && config.resourceRunId) {
2455
+ config.resourceLedger.track(
2456
+ config.resourceRunId,
2457
+ "tool",
2458
+ `${record.toolCall.name}:${record.toolCall.id}`,
2459
+ task,
2460
+ );
2461
+ }
2364
2462
  if (concurrency === "exclusive") {
2365
2463
  lastExclusive = task;
2366
2464
  sharedTasks = [];