@k2b/cloud 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,6 +9,14 @@ export type AiActiveTurn = {
9
9
  status: "running" | "waiting_for_action";
10
10
  blocks: AiTurnBlock[];
11
11
  modelProfileId: string | null;
12
+ /** A model call failed transiently and waits for its retry. Live only: the next turn event or a snapshot ends it. */
13
+ providerRetry?: boolean;
14
+ /** Server start of the turn, known from a state snapshot. */
15
+ createdAt?: string;
16
+ /** Time already spent waiting for answered user actions, from the last state snapshot. */
17
+ actionWaitMs?: number;
18
+ /** Start of the user action the turn waited for at the last state snapshot. */
19
+ waitingSince?: string | null;
12
20
  };
13
21
 
14
22
  export type AiChatProjection = {
@@ -90,7 +98,7 @@ export const mergeActiveTurn = (previous: AiActiveTurn | null, incoming: AiActiv
90
98
  const pendingSteers = previous.blocks.filter((block) => block.kind === "steer_message" && block.status !== "consumed");
91
99
  const known = new Set(incoming.blocks.map((block) => block.id));
92
100
  const blocks = [...incoming.blocks, ...pendingSteers.filter((block) => !known.has(block.id))];
93
- return { ...incoming, blocks, status: deriveStatus(blocks) };
101
+ return { ...turnTiming(previous), ...incoming, blocks, status: deriveStatus(blocks) };
94
102
  };
95
103
 
96
104
  export const reconcileActiveTurnActions = (
@@ -112,9 +120,22 @@ export const activeTurnFromSnapshot = (snapshot: AiTurnSnapshot | null): AiActiv
112
120
  status: snapshot.status === "waiting_for_action" ? "waiting_for_action" : "running",
113
121
  blocks: normalizeCustomApprovalBlocks(snapshot.blocks),
114
122
  modelProfileId: snapshot.modelProfileId,
123
+ createdAt: snapshot.createdAt,
124
+ ...(snapshot.actionWaitMs !== undefined ? { actionWaitMs: snapshot.actionWaitMs } : {}),
125
+ ...(snapshot.waitingSince !== undefined ? { waitingSince: snapshot.waitingSince } : {}),
115
126
  };
116
127
  };
117
128
 
129
+ /** Snapshot timing survives later attempts of the same turn, which carry no timing of their own. */
130
+ const turnTiming = (turn: AiActiveTurn | null | undefined): Pick<AiActiveTurn, "createdAt" | "actionWaitMs" | "waitingSince"> =>
131
+ turn
132
+ ? {
133
+ ...(turn.createdAt !== undefined ? { createdAt: turn.createdAt } : {}),
134
+ ...(turn.actionWaitMs !== undefined ? { actionWaitMs: turn.actionWaitMs } : {}),
135
+ ...(turn.waitingSince !== undefined ? { waitingSince: turn.waitingSince } : {}),
136
+ }
137
+ : {};
138
+
118
139
  /**
119
140
  * Fold one stream event into the projection. Pure and total: the client and any
120
141
  * test converge on the same state for the same ordered event sequence.
@@ -125,6 +146,7 @@ export const activeTurnFromSnapshot = (snapshot: AiTurnSnapshot | null): AiActiv
125
146
  * stale attempts are ignored. Older senders without a baseline reset to only
126
147
  * locally pending steering blocks.
127
148
  * - block events apply only when strictly newer than the active turn's cursor.
149
+ * - `provider_retry` marks the active turn until its next event.
128
150
  * - `turn_finished` folds the turn's persisted messages in and clears the active turn.
129
151
  */
130
152
  export const reduceProjection = (state: AiChatProjection, event: AiStreamEvent): AiChatProjection => {
@@ -146,6 +168,8 @@ export const reduceProjection = (state: AiChatProjection, event: AiStreamEvent):
146
168
  return reduceWireEvent(state, event);
147
169
  };
148
170
 
171
+ const withoutProviderRetry = ({ providerRetry: _, ...turn }: AiActiveTurn): AiActiveTurn => turn;
172
+
149
173
  export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): AiChatProjection => {
150
174
  const active = state.activeTurn;
151
175
 
@@ -163,6 +187,7 @@ export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): Ai
163
187
  status: deriveStatus(blocks),
164
188
  blocks,
165
189
  modelProfileId: event.modelProfileId,
190
+ ...turnTiming(active?.turnId === event.turnId ? active : null),
166
191
  },
167
192
  };
168
193
  }
@@ -172,20 +197,26 @@ export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): Ai
172
197
  return { ...state, messages: mergeMessages(state.messages, event.messages ?? []), activeTurn: null };
173
198
  }
174
199
 
200
+ if (!active || active.turnId !== event.turnId || !isNewerWireEvent(event, active)) return state;
201
+
202
+ if (event.type === "provider_retry") {
203
+ return { ...state, activeTurn: { ...active, seq: event.seq, attempt: event.attempt, providerRetry: true } };
204
+ }
205
+
175
206
  if (event.type === "message_saved") {
176
- if (!active || active.turnId !== event.turnId || !isNewerWireEvent(event, active)) return state;
177
207
  return {
178
208
  ...state,
179
209
  messages: mergeMessages(state.messages, [event.message]),
180
- activeTurn: { ...active, seq: event.seq, attempt: event.attempt },
210
+ activeTurn: { ...withoutProviderRetry(active), seq: event.seq, attempt: event.attempt },
181
211
  };
182
212
  }
183
213
 
184
214
  // block_set / block_delta
185
- if (!active || active.turnId !== event.turnId) return state;
186
- if (!isNewerWireEvent(event, active)) return state;
187
215
  const blocks = applyWireEventToBlocks(active.blocks, event);
188
- return { ...state, activeTurn: { ...active, attempt: event.attempt, seq: event.seq, blocks, status: deriveStatus(blocks) } };
216
+ return {
217
+ ...state,
218
+ activeTurn: { ...withoutProviderRetry(active), attempt: event.attempt, seq: event.seq, blocks, status: deriveStatus(blocks) },
219
+ };
189
220
  };
190
221
 
191
222
  /** Assistant/tool messages of the active turn are represented by live blocks; hide them. */
@@ -40,6 +40,7 @@ import {
40
40
  streamBlockId,
41
41
  toolBlockId,
42
42
  } from "./protocol";
43
+ import { retryTransientProviderErrors } from "./provider-retry";
43
44
  import { assistantQuotaProvider, inferenceProvider } from "./quota-provider";
44
45
  import { collectConversationResourceObservations } from "./resource-refs";
45
46
  import { AiRunTimeout } from "./run-timeout";
@@ -249,6 +250,8 @@ export type ExecutorConfig = {
249
250
  validateTurn?: typeof validateAiTurnRequest;
250
251
  /** Runs after the durable turn state and final wire event are flushed. */
251
252
  onTurnFinalized?: (event: AiTurnFinalizedEvent) => Promise<void>;
253
+ /** Waits before retrying a transient provider failure without Retry-After; tests shorten them. */
254
+ providerRetryDelaysMs?: readonly number[];
252
255
  };
253
256
 
254
257
  type ResolvedModel = Awaited<ReturnType<typeof resolveAiModel>>;
@@ -392,6 +395,10 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
392
395
  const setRejectedCallIds = (items: Set<string>) => {
393
396
  rejectedCallIds = items;
394
397
  };
398
+ let approvedCallIds = new Set<string>();
399
+ const setApprovedCallIds = (items: Set<string>) => {
400
+ approvedCallIds = items;
401
+ };
395
402
  /** nessi stream block ids (turn-scoped) that belong to tool_call blocks — their deltas are raw args JSON. */
396
403
  const toolStreamIds = new Set<string>();
397
404
  /** kind per open Cloud stream block id, for delta create-if-missing. */
@@ -414,6 +421,7 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
414
421
  approval: approval && !allowRememberedApprovals ? { ...approval, allowAlways: false } : approval,
415
422
  frontendMode: patch.frontendMode ?? existing?.frontendMode,
416
423
  presentation: patch.presentation ?? existing?.presentation ?? presentations.get(rawName) ?? presentations.get(name),
424
+ ...(approvedCallIds.has(callId) || approvedCallIds.has(displayCallId) || existing?.approved ? { approved: true } : {}),
417
425
  };
418
426
  toolBlocks.set(callId, block);
419
427
  return { type: "block_set", block };
@@ -518,6 +526,7 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
518
526
  setApprovalPolicies,
519
527
  setApprovalReviews,
520
528
  setRejectedCallIds,
529
+ setApprovedCallIds,
521
530
  };
522
531
  };
523
532
 
@@ -829,6 +838,7 @@ export class AiTurnExecutor {
829
838
  pipeline.setApprovalPolicies(prepared.approvalPolicies);
830
839
  const toolPresentations = new Map<string, AiToolPresentation>();
831
840
  const rejectedToolCallIds = new Set<string>();
841
+ const approvedToolCallIds = new Set<string>();
832
842
  pipeline.setPresentations(toolPresentations);
833
843
  pipeline.setApprovalReviews(capabilityActionReviews);
834
844
  let turnInput = config.input;
@@ -868,17 +878,19 @@ export class AiTurnExecutor {
868
878
  turnInput,
869
879
  toolPresentations,
870
880
  rejectedToolCallIds,
881
+ approvedToolCallIds,
871
882
  });
872
883
 
873
884
  const { loopMessages, pendingRecords, resolvedRecords, turnSteers } =
874
885
  attemptState ?? (await loadChatAttemptState(conversationId, turnId));
875
886
  for (const action of resolvedRecords) {
876
- if (action.resolvedEvent?.type !== "approval_response" || action.resolvedEvent.approved) continue;
877
- rejectedToolCallIds.add(
878
- action.kind === "custom_approval" ? (customApprovalParentCallId(action.callId) ?? action.callId) : action.callId,
879
- );
887
+ if (action.resolvedEvent?.type !== "approval_response") continue;
888
+ // A decision on a custom approval belongs to the call that asked for it.
889
+ const callId = action.kind === "custom_approval" ? (customApprovalParentCallId(action.callId) ?? action.callId) : action.callId;
890
+ (action.resolvedEvent.approved ? approvedToolCallIds : rejectedToolCallIds).add(callId);
880
891
  }
881
892
  pipeline.setRejectedCallIds(rejectedToolCallIds);
893
+ pipeline.setApprovedCallIds(approvedToolCallIds);
882
894
  const assistantMessages = loopMessages.filter((message) => message.message.role !== "user");
883
895
  const isFresh = assistantMessages.length === 0 && resolvedRecords.length === 0 && !skipResolvedActions;
884
896
 
@@ -1064,8 +1076,19 @@ export class AiTurnExecutor {
1064
1076
  });
1065
1077
  const priorToolRounds = toolRoundState(loopMessages);
1066
1078
  const quotaSubject = accessSubjectForActor(material.actor);
1079
+ const deadline = claim.turn.deadline ? Date.parse(claim.turn.deadline) : null;
1067
1080
  const toolRoundPolicy = applyToolRoundPolicy({
1068
- provider: assistantQuotaProvider(resolved.provider, config, quotaSubject, resolved.profile, turnId, conversationId),
1081
+ provider: retryTransientProviderErrors(
1082
+ assistantQuotaProvider(resolved.provider, config, quotaSubject, resolved.profile, turnId, conversationId),
1083
+ {
1084
+ deadline,
1085
+ delaysMs: this.config.providerRetryDelaysMs,
1086
+ onRetry: async ({ retry, delayMs, issue }) => {
1087
+ log.warn("AI provider call retried", { conversationId, turnId, retry, delayMs, kind: issue.kind, message: issue.message });
1088
+ await pipeline.emitProviderRetry();
1089
+ },
1090
+ },
1091
+ ),
1069
1092
  tools,
1070
1093
  maxToolRounds: resolved.profile.maxToolRounds,
1071
1094
  issuedToolRounds: priorToolRounds.issued,
@@ -1667,6 +1690,10 @@ class StreamPipeline {
1667
1690
  this.mapper.setRejectedCallIds(callIds);
1668
1691
  }
1669
1692
 
1693
+ setApprovedCallIds(callIds: Set<string>): void {
1694
+ this.mapper.setApprovedCallIds(callIds);
1695
+ }
1696
+
1670
1697
  async emitBaseline(): Promise<void> {
1671
1698
  for (const block of this.blocks) {
1672
1699
  const seq = this.nextSeq();
@@ -1765,6 +1792,12 @@ class StreamPipeline {
1765
1792
  await this.publish(event);
1766
1793
  }
1767
1794
 
1795
+ /** Transient: says a model call is waiting to be retried. Snapshots never carry it. */
1796
+ async emitProviderRetry(): Promise<void> {
1797
+ const seq = this.nextSeq();
1798
+ await this.publish(this.envelope({ type: "provider_retry" as const, seq }));
1799
+ }
1800
+
1768
1801
  async emitTurnFinished(status: "completed" | "failed" | "aborted", error: string | null): Promise<void> {
1769
1802
  const seq = this.nextSeq();
1770
1803
  await this.publish(this.envelope({ type: "turn_finished" as const, seq, status, error }) as AiWireEvent);
@@ -39,6 +39,8 @@ export type AiTurnBlock =
39
39
  isError?: boolean;
40
40
  /** Present while status is awaiting_approval. */
41
41
  approval?: { message?: string; review?: CapabilityActionReview; allowAlways: boolean };
42
+ /** The user approved this call in the chat; the decided approval stays visible as a receipt. */
43
+ approved?: boolean;
42
44
  /** Present for frontend tools. */
43
45
  frontendMode?: AiFrontendToolMode;
44
46
  /** Saved Cloud-owned display snapshot for capability calls. */
@@ -67,6 +69,8 @@ export type AiWireEvent =
67
69
  | (AiWireEventBase & { type: "message_saved"; message: AiStoredMessage })
68
70
  | (AiWireEventBase & { type: "block_set"; block: AiTurnBlock })
69
71
  | (AiWireEventBase & { type: "block_delta"; blockId: string; blockKind: "text" | "thinking"; delta: string })
72
+ /** A model call failed transiently and waits for its retry. Transient: the next event of the turn ends it. */
73
+ | (AiWireEventBase & { type: "provider_retry" })
70
74
  | (AiWireEventBase & {
71
75
  type: "turn_finished";
72
76
  status: AiTurnFinishedStatus;
@@ -85,6 +89,10 @@ export type AiTurnSnapshot = {
85
89
  blocks: AiTurnBlock[];
86
90
  modelProfileId: string | null;
87
91
  createdAt: string;
92
+ /** Time the turn already waited for answered user actions such as approvals. Absent from older servers. */
93
+ actionWaitMs?: number;
94
+ /** Start of the user action the turn waits for now, or null while it works. Absent from older servers. */
95
+ waitingSince?: string | null;
88
96
  };
89
97
 
90
98
  /** Full projection seed sent as the first SSE event on every (re)connect. */
@@ -151,7 +159,12 @@ export const reconcileResolvedTurnActions = (
151
159
  const event = resolvedByCallId.get(block.callId);
152
160
  if (!event || event.callId !== block.callId) return block;
153
161
  if (block.status === "awaiting_approval" && event.type === "approval_response") {
154
- return { ...block, status: event.approved ? "running" : "rejected", approval: undefined };
162
+ return {
163
+ ...block,
164
+ status: event.approved ? "running" : "rejected",
165
+ approval: undefined,
166
+ ...(event.approved ? { approved: true } : {}),
167
+ };
155
168
  }
156
169
  if (block.status === "awaiting_client" && event.type === "tool_result") {
157
170
  return { ...block, status: "completed", result: event.result, isError: false };
@@ -172,7 +185,7 @@ export const buildBlocksFromMessages = (
172
185
  meta?: {
173
186
  steerId?: string;
174
187
  toolPresentations?: Record<string, AiToolPresentation>;
175
- toolOutcomes?: Record<string, "rejected">;
188
+ toolOutcomes?: Record<string, "rejected" | "approved">;
176
189
  } | null;
177
190
  }[],
178
191
  ): AiTurnBlock[] => {
@@ -196,6 +209,8 @@ export const buildBlocksFromMessages = (
196
209
  args: block.args,
197
210
  status: "running",
198
211
  presentation: meta?.toolPresentations?.[block.id],
212
+ // A turn that ended before an approved call returned records the approval on the call's message.
213
+ ...(meta?.toolOutcomes?.[block.id] === "approved" ? { approved: true } : {}),
199
214
  });
200
215
  }
201
216
  });
@@ -212,12 +227,13 @@ export const buildBlocksFromMessages = (
212
227
  const at = toolIndex.get(message.callId);
213
228
  const existing = at !== undefined ? blocks[at] : undefined;
214
229
  if (existing?.kind === "tool") {
215
- const rejected = meta?.toolOutcomes?.[message.callId] === "rejected";
230
+ const outcome = meta?.toolOutcomes?.[message.callId];
216
231
  blocks[at!] = {
217
232
  ...existing,
218
- status: rejected ? "rejected" : message.isError ? "failed" : "completed",
233
+ status: outcome === "rejected" ? "rejected" : message.isError ? "failed" : "completed",
219
234
  result: message.result,
220
235
  isError: message.isError,
236
+ ...(outcome === "approved" ? { approved: true } : {}),
221
237
  };
222
238
  }
223
239
  }
@@ -1,27 +1,63 @@
1
1
  import { AsyncLocalStorage } from "node:async_hooks";
2
2
 
3
3
  /**
4
- * Dates the provider response headers and the first body byte of the inference
5
- * call that is currently streaming, without touching the wire. nessi adapters
6
- * call the global `fetch`; the wrapper is installed once, on first use, and
7
- * only observes requests made inside `runWithProviderFetchMarks`.
4
+ * Observes the provider request of the inference call that is currently
5
+ * running, without touching the wire: when its headers and first body byte
6
+ * arrived, whether the provider certainly did not process it, and how long a
7
+ * rejecting provider asked the caller to wait. nessi adapters call the global
8
+ * `fetch`; the wrapper is installed once, on first use, and only observes
9
+ * requests made inside `runWithProviderFetchMarks`. Scopes nest, so an outer
10
+ * retry policy and the inner per-call accounting each see the same request.
8
11
  *
9
12
  * Nothing here logs: frames may carry reasoning text and headers carry keys.
10
13
  */
11
- export type ProviderFetchMarks = { headersAt?: number; firstByteAt?: number };
14
+ export type ProviderFetchMarks = {
15
+ headersAt?: number;
16
+ firstByteAt?: number;
17
+ /** The request never reached the provider, or the provider answered with a status that says it did not process it. */
18
+ refused?: boolean;
19
+ retryAfterMs?: number;
20
+ };
12
21
 
13
- const marks = new AsyncLocalStorage<ProviderFetchMarks>();
22
+ const scopes = new AsyncLocalStorage<readonly ProviderFetchMarks[]>();
14
23
  let installed = false;
15
24
 
16
25
  export const runWithProviderFetchMarks = <T>(store: ProviderFetchMarks, run: () => Promise<T>): Promise<T> => {
17
26
  install();
18
- return marks.run(store, run);
27
+ return scopes.run([...(scopes.getStore() ?? []), store], run);
28
+ };
29
+
30
+ /** `retry-after-ms` (OpenAI) wins over the standard `retry-after` seconds or HTTP date. */
31
+ export const retryAfterMs = (headers: Headers, now = Date.now()): number | undefined => {
32
+ const milliseconds = Number(headers.get("retry-after-ms")?.trim() || Number.NaN);
33
+ if (Number.isFinite(milliseconds) && milliseconds >= 0) return milliseconds;
34
+ const value = headers.get("retry-after")?.trim();
35
+ if (!value) return undefined;
36
+ const seconds = Number(value);
37
+ if (Number.isFinite(seconds)) return seconds >= 0 ? seconds * 1_000 : undefined;
38
+ const date = Date.parse(value);
39
+ return Number.isFinite(date) ? Math.max(0, date - now) : undefined;
19
40
  };
20
41
 
21
- const firstByteObserver = (store: ProviderFetchMarks) =>
42
+ /**
43
+ * A client error, 503, or Anthropic's overloaded 529 says the provider did not
44
+ * process the request. Other server errors, such as a gateway's 502 or 504, can
45
+ * follow processing upstream.
46
+ */
47
+ const refusedStatus = (status: number) => (status >= 400 && status < 500) || status === 503 || status === 529;
48
+
49
+ /** Bun's codes for a request that never left: no connection, or no address for the host. */
50
+ const unsentCodes = new Set(["ConnectionRefused", "ENOTFOUND"]);
51
+ const unsent = (error: unknown) =>
52
+ typeof error === "object" && error !== null && "code" in error && typeof error.code === "string" && unsentCodes.has(error.code);
53
+
54
+ const firstByteObserver = (stores: readonly ProviderFetchMarks[]) =>
22
55
  new TransformStream<Uint8Array, Uint8Array>({
23
56
  transform(chunk, controller) {
24
- if (chunk.byteLength > 0 && store.firstByteAt === undefined) store.firstByteAt = Date.now();
57
+ if (chunk.byteLength > 0)
58
+ for (const store of stores) {
59
+ store.firstByteAt ??= Date.now();
60
+ }
25
61
  controller.enqueue(chunk);
26
62
  },
27
63
  });
@@ -31,13 +67,29 @@ const install = (): void => {
31
67
  installed = true;
32
68
  const realFetch = globalThis.fetch;
33
69
  const instrumented = async (input: RequestInfo | URL, init?: RequestInit): Promise<Response> => {
34
- const store = marks.getStore();
35
- // Only the first request of a call is the provider request; nessi retries create a new call.
36
- if (store === undefined || store.headersAt !== undefined) return realFetch(input, init);
37
- const response = await realFetch(input, init);
38
- store.headersAt = Date.now();
70
+ // Only the first request of a scope is its provider request; a retry opens a new scope.
71
+ const stores = (scopes.getStore() ?? []).filter((store) => store.headersAt === undefined);
72
+ if (stores.length === 0) return realFetch(input, init);
73
+ let response: Response;
74
+ try {
75
+ response = await realFetch(input, init);
76
+ } catch (error) {
77
+ // A reset, an abort, or a timeout can follow delivery, so only an unsent request counts as refused.
78
+ if (unsent(error))
79
+ for (const store of stores) {
80
+ store.refused = true;
81
+ }
82
+ throw error;
83
+ }
84
+ const headersAt = Date.now();
85
+ const wait = response.ok ? undefined : retryAfterMs(response.headers, headersAt);
86
+ for (const store of stores) {
87
+ store.headersAt = headersAt;
88
+ store.refused = refusedStatus(response.status);
89
+ if (wait !== undefined) store.retryAfterMs = wait;
90
+ }
39
91
  if (!response.body) return response;
40
- return new Response(response.body.pipeThrough(firstByteObserver(store)), {
92
+ return new Response(response.body.pipeThrough(firstByteObserver(stores)), {
41
93
  status: response.status,
42
94
  statusText: response.statusText,
43
95
  headers: response.headers,
@@ -0,0 +1,105 @@
1
+ import { setTimeout as delay } from "node:timers/promises";
2
+ import type { NessiIssue, Provider, ProviderIssue, StreamEvent, TimeoutIssue } from "@k2b/nessi/ai";
3
+ import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
4
+
5
+ /** Waits before the first and the second retry when the provider names none. nessi itself never retries. */
6
+ export const AI_PROVIDER_RETRY_DELAYS_MS: readonly number[] = [1_000, 4_000];
7
+
8
+ /**
9
+ * The longest Retry-After that is honored, the bound the OpenAI and Anthropic
10
+ * SDKs apply. A provider that asks for a longer wait is unavailable for this
11
+ * turn, so the call fails now with the provider's message.
12
+ */
13
+ const AI_PROVIDER_MAX_RETRY_AFTER_MS = 60_000;
14
+
15
+ export type AiProviderRetry = {
16
+ /** 1 for the first retry. */
17
+ retry: number;
18
+ delayMs: number;
19
+ issue: ProviderIssue | TimeoutIssue;
20
+ };
21
+
22
+ const transientIssue = (issue: NessiIssue): issue is ProviderIssue | TimeoutIssue =>
23
+ (issue.kind === "provider_error" && issue.retryable && !issue.contextOverflow) ||
24
+ (issue.kind === "timeout" && issue.retryable && issue.scope !== "tool");
25
+
26
+ const isBlockEvent = (event: StreamEvent) => event.type === "block_start" || event.type === "block_delta" || event.type === "block_end";
27
+
28
+ /**
29
+ * Repeats a model call that failed transiently (429, 5xx, a lost connection, a
30
+ * provider timeout) before it produced any output. Each attempt passes through
31
+ * `provider` again, so quota admission and accounting see every request.
32
+ *
33
+ * Once a block event was emitted the attempt is final: streamed text cannot be
34
+ * taken back, so a failure mid-stream still ends the call. Context overflow is
35
+ * left to compaction. Waits follow the provider's Retry-After, otherwise
36
+ * `delaysMs`, end before `deadline`, and stop on the request's abort signal.
37
+ */
38
+ export function retryTransientProviderErrors(
39
+ provider: Provider,
40
+ options: {
41
+ /** Epoch milliseconds by which a wait must end; null when the turn has no run time limit. */
42
+ deadline: number | null;
43
+ delaysMs?: readonly number[];
44
+ /** Announces a wait before it starts. */
45
+ onRetry?: (retry: AiProviderRetry) => Promise<void>;
46
+ },
47
+ ): Provider {
48
+ const delays = options.delaysMs ?? AI_PROVIDER_RETRY_DELAYS_MS;
49
+ const waitFor = (retry: number, marks: ProviderFetchMarks, signal: AbortSignal | undefined): number | null => {
50
+ if (retry >= delays.length || signal?.aborted) return null;
51
+ const delayMs = marks.retryAfterMs ?? delays[retry]!;
52
+ if (delayMs > AI_PROVIDER_MAX_RETRY_AFTER_MS) return null;
53
+ if (options.deadline !== null && Date.now() + delayMs >= options.deadline) return null;
54
+ return delayMs;
55
+ };
56
+ return {
57
+ name: provider.name,
58
+ family: provider.family,
59
+ model: provider.model,
60
+ contextWindow: provider.contextWindow,
61
+ capabilities: provider.capabilities,
62
+ complete: (request) => provider.complete(request),
63
+ stream: async function* (request) {
64
+ for (let retry = 0; ; retry += 1) {
65
+ const marks: ProviderFetchMarks = {};
66
+ const events = provider.stream(request)[Symbol.asyncIterator]();
67
+ const next = () => runWithProviderFetchMarks(marks, () => events.next());
68
+ // Events before the first block (usage, issues) are held so that a retried attempt leaves no trace.
69
+ const held: StreamEvent[] = [];
70
+ let committed = false;
71
+ let pending: AiProviderRetry | null = null;
72
+ try {
73
+ for (let step = await next(); !step.done; step = await next()) {
74
+ const event = step.value;
75
+ if (!committed && event.type === "issue" && transientIssue(event.issue)) {
76
+ const delayMs = waitFor(retry, marks, request.signal);
77
+ if (delayMs !== null) {
78
+ pending = { retry: retry + 1, delayMs, issue: event.issue };
79
+ break;
80
+ }
81
+ }
82
+ if (!committed && !isBlockEvent(event)) {
83
+ held.push(event);
84
+ continue;
85
+ }
86
+ if (!committed) {
87
+ committed = true;
88
+ yield* held.splice(0);
89
+ }
90
+ yield event;
91
+ }
92
+ } finally {
93
+ // Settles the failed attempt's accounting before the wait starts.
94
+ await events.return?.();
95
+ }
96
+ if (!pending) {
97
+ yield* held;
98
+ return;
99
+ }
100
+ await options.onRetry?.(pending);
101
+ await delay(pending.delayMs, undefined, { signal: request.signal });
102
+ }
103
+ },
104
+ };
105
+ }
@@ -5,7 +5,7 @@ import type { AccessSubject } from "../server/services/access";
5
5
  import { logger } from "../services/logging";
6
6
  import { isAssistantChatTurn } from "./assistant-models";
7
7
  import { AiBackgroundAdmissionError, type AiCallContext, type AiCallDetails, beginAiCall, finishAiCall } from "./inference-calls";
8
- import { runWithProviderFetchMarks } from "./provider-fetch";
8
+ import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
9
9
  import type { AiModelProfile } from "./types";
10
10
 
11
11
  const log = logger("ai:quotas");
@@ -93,7 +93,7 @@ export function inferenceProvider(
93
93
  let status: "ok" | "failed" = "failed";
94
94
  let error: string | undefined;
95
95
  const requestStartedAt = Date.now();
96
- const marks: { headersAt?: number; firstByteAt?: number } = {};
96
+ const marks: ProviderFetchMarks = {};
97
97
  try {
98
98
  const result = await runWithProviderFetchMarks(marks, () =>
99
99
  provider.complete({ ...request, maxOutputTokens: call.maxOutputTokens }),
@@ -112,7 +112,9 @@ export function inferenceProvider(
112
112
  throw thrown;
113
113
  } finally {
114
114
  call.stop();
115
- if (!usage && status === "failed") usage = { input: call.inputTokens, output: 0, estimated: true };
115
+ // A request the provider refused or never received cost nothing; any other failure may have been processed.
116
+ if (!usage && status === "failed")
117
+ usage = marks.refused ? { input: 0, output: 0 } : { input: call.inputTokens, output: 0, estimated: true };
116
118
  const cancelled = request.signal?.aborted === true;
117
119
  await finish(call.id, usage, cancelled && status === "failed" ? "aborted" : status, {
118
120
  error: cancelled ? null : error,
@@ -133,7 +135,7 @@ export function inferenceProvider(
133
135
  const outputBlocks = new Map<string, number>();
134
136
  let generated = false;
135
137
  const requestStartedAt = Date.now();
136
- const marks: { headersAt?: number; firstByteAt?: number; firstBlockAt?: number } = {};
138
+ const marks: ProviderFetchMarks & { firstBlockAt?: number } = {};
137
139
  // The wrapped adapter reads lazily, so the request only leaves once the first pull runs inside the marked scope.
138
140
  const events = provider.stream({ ...request, maxOutputTokens: call.maxOutputTokens })[Symbol.asyncIterator]();
139
141
  const next = () => runWithProviderFetchMarks(marks, () => events.next());
@@ -158,7 +160,14 @@ export function inferenceProvider(
158
160
  outputBlocks.set(event.blockId, size);
159
161
  }
160
162
  if (event.type === "block_start" || event.type === "block_delta" || event.type === "block_end") generated = true;
161
- if (event.type === "issue" && event.issue.kind === "provider_error" && event.issue.contextOverflow && !generated && !usage)
163
+ // A provider that refused the request, or never received it, did not process it.
164
+ if (
165
+ event.type === "issue" &&
166
+ event.issue.kind === "provider_error" &&
167
+ (event.issue.contextOverflow || marks.refused) &&
168
+ !generated &&
169
+ !usage
170
+ )
162
171
  usage = { input: 0, output: 0 };
163
172
  if (event.type === "usage") {
164
173
  if (event.finishReason === "aborted" || event.finishReason === "interrupted" || event.finishReason === "error") failed = true;