@k2b/cloud 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/_internal/capabilities.ts +12 -0
- package/src/_internal/registry.ts +1 -0
- package/src/access/GroupCoverage.tsx +175 -0
- package/src/access/PermissionEditor.tsx +119 -99
- package/src/access/messages.ts +30 -0
- package/src/ai/approval-routes.ts +5 -5
- package/src/ai/capabilities.ts +34 -12
- package/src/ai/chat/blocks.tsx +86 -179
- package/src/ai/chat/builtin-tools.tsx +57 -30
- package/src/ai/chat/live-turn.browser-harness.tsx +37 -0
- package/src/ai/chat/message-actions.tsx +6 -2
- package/src/ai/chat/message-utils.ts +17 -14
- package/src/ai/chat/messages.ts +324 -2
- package/src/ai/chat/presentation.tsx +257 -103
- package/src/ai/chat/tool-groups.ts +55 -35
- package/src/ai/chat/turn-layout.ts +141 -0
- package/src/ai/chat/turn-view.tsx +609 -0
- package/src/ai/client/projection.ts +37 -6
- package/src/ai/executor.ts +38 -5
- package/src/ai/protocol.ts +20 -4
- package/src/ai/provider-fetch.ts +67 -15
- package/src/ai/provider-retry.ts +105 -0
- package/src/ai/quota-provider.ts +14 -5
- package/src/ai/store.ts +94 -3
- package/src/ai/stream.ts +2 -0
- package/src/ai/timeline.ts +9 -11
- package/src/ai/turn-timing.ts +31 -3
- package/src/ai/types.ts +12 -2
- package/src/contracts/registry.ts +2 -0
- package/src/shared/app-presentation.ts +10 -2
- package/src/styles/effects.css +69 -0
|
@@ -9,6 +9,14 @@ export type AiActiveTurn = {
|
|
|
9
9
|
status: "running" | "waiting_for_action";
|
|
10
10
|
blocks: AiTurnBlock[];
|
|
11
11
|
modelProfileId: string | null;
|
|
12
|
+
/** A model call failed transiently and waits for its retry. Live only: the next turn event or a snapshot ends it. */
|
|
13
|
+
providerRetry?: boolean;
|
|
14
|
+
/** Server start of the turn, known from a state snapshot. */
|
|
15
|
+
createdAt?: string;
|
|
16
|
+
/** Time already spent waiting for answered user actions, from the last state snapshot. */
|
|
17
|
+
actionWaitMs?: number;
|
|
18
|
+
/** Start of the user action the turn waited for at the last state snapshot. */
|
|
19
|
+
waitingSince?: string | null;
|
|
12
20
|
};
|
|
13
21
|
|
|
14
22
|
export type AiChatProjection = {
|
|
@@ -90,7 +98,7 @@ export const mergeActiveTurn = (previous: AiActiveTurn | null, incoming: AiActiv
|
|
|
90
98
|
const pendingSteers = previous.blocks.filter((block) => block.kind === "steer_message" && block.status !== "consumed");
|
|
91
99
|
const known = new Set(incoming.blocks.map((block) => block.id));
|
|
92
100
|
const blocks = [...incoming.blocks, ...pendingSteers.filter((block) => !known.has(block.id))];
|
|
93
|
-
return { ...incoming, blocks, status: deriveStatus(blocks) };
|
|
101
|
+
return { ...turnTiming(previous), ...incoming, blocks, status: deriveStatus(blocks) };
|
|
94
102
|
};
|
|
95
103
|
|
|
96
104
|
export const reconcileActiveTurnActions = (
|
|
@@ -112,9 +120,22 @@ export const activeTurnFromSnapshot = (snapshot: AiTurnSnapshot | null): AiActiv
|
|
|
112
120
|
status: snapshot.status === "waiting_for_action" ? "waiting_for_action" : "running",
|
|
113
121
|
blocks: normalizeCustomApprovalBlocks(snapshot.blocks),
|
|
114
122
|
modelProfileId: snapshot.modelProfileId,
|
|
123
|
+
createdAt: snapshot.createdAt,
|
|
124
|
+
...(snapshot.actionWaitMs !== undefined ? { actionWaitMs: snapshot.actionWaitMs } : {}),
|
|
125
|
+
...(snapshot.waitingSince !== undefined ? { waitingSince: snapshot.waitingSince } : {}),
|
|
115
126
|
};
|
|
116
127
|
};
|
|
117
128
|
|
|
129
|
+
/** Snapshot timing survives later attempts of the same turn, which carry no timing of their own. */
|
|
130
|
+
const turnTiming = (turn: AiActiveTurn | null | undefined): Pick<AiActiveTurn, "createdAt" | "actionWaitMs" | "waitingSince"> =>
|
|
131
|
+
turn
|
|
132
|
+
? {
|
|
133
|
+
...(turn.createdAt !== undefined ? { createdAt: turn.createdAt } : {}),
|
|
134
|
+
...(turn.actionWaitMs !== undefined ? { actionWaitMs: turn.actionWaitMs } : {}),
|
|
135
|
+
...(turn.waitingSince !== undefined ? { waitingSince: turn.waitingSince } : {}),
|
|
136
|
+
}
|
|
137
|
+
: {};
|
|
138
|
+
|
|
118
139
|
/**
|
|
119
140
|
* Fold one stream event into the projection. Pure and total: the client and any
|
|
120
141
|
* test converge on the same state for the same ordered event sequence.
|
|
@@ -125,6 +146,7 @@ export const activeTurnFromSnapshot = (snapshot: AiTurnSnapshot | null): AiActiv
|
|
|
125
146
|
* stale attempts are ignored. Older senders without a baseline reset to only
|
|
126
147
|
* locally pending steering blocks.
|
|
127
148
|
* - block events apply only when strictly newer than the active turn's cursor.
|
|
149
|
+
* - `provider_retry` marks the active turn until its next event.
|
|
128
150
|
* - `turn_finished` folds the turn's persisted messages in and clears the active turn.
|
|
129
151
|
*/
|
|
130
152
|
export const reduceProjection = (state: AiChatProjection, event: AiStreamEvent): AiChatProjection => {
|
|
@@ -146,6 +168,8 @@ export const reduceProjection = (state: AiChatProjection, event: AiStreamEvent):
|
|
|
146
168
|
return reduceWireEvent(state, event);
|
|
147
169
|
};
|
|
148
170
|
|
|
171
|
+
const withoutProviderRetry = ({ providerRetry: _, ...turn }: AiActiveTurn): AiActiveTurn => turn;
|
|
172
|
+
|
|
149
173
|
export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): AiChatProjection => {
|
|
150
174
|
const active = state.activeTurn;
|
|
151
175
|
|
|
@@ -163,6 +187,7 @@ export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): Ai
|
|
|
163
187
|
status: deriveStatus(blocks),
|
|
164
188
|
blocks,
|
|
165
189
|
modelProfileId: event.modelProfileId,
|
|
190
|
+
...turnTiming(active?.turnId === event.turnId ? active : null),
|
|
166
191
|
},
|
|
167
192
|
};
|
|
168
193
|
}
|
|
@@ -172,20 +197,26 @@ export const reduceWireEvent = (state: AiChatProjection, event: AiWireEvent): Ai
|
|
|
172
197
|
return { ...state, messages: mergeMessages(state.messages, event.messages ?? []), activeTurn: null };
|
|
173
198
|
}
|
|
174
199
|
|
|
200
|
+
if (!active || active.turnId !== event.turnId || !isNewerWireEvent(event, active)) return state;
|
|
201
|
+
|
|
202
|
+
if (event.type === "provider_retry") {
|
|
203
|
+
return { ...state, activeTurn: { ...active, seq: event.seq, attempt: event.attempt, providerRetry: true } };
|
|
204
|
+
}
|
|
205
|
+
|
|
175
206
|
if (event.type === "message_saved") {
|
|
176
|
-
if (!active || active.turnId !== event.turnId || !isNewerWireEvent(event, active)) return state;
|
|
177
207
|
return {
|
|
178
208
|
...state,
|
|
179
209
|
messages: mergeMessages(state.messages, [event.message]),
|
|
180
|
-
activeTurn: { ...active, seq: event.seq, attempt: event.attempt },
|
|
210
|
+
activeTurn: { ...withoutProviderRetry(active), seq: event.seq, attempt: event.attempt },
|
|
181
211
|
};
|
|
182
212
|
}
|
|
183
213
|
|
|
184
214
|
// block_set / block_delta
|
|
185
|
-
if (!active || active.turnId !== event.turnId) return state;
|
|
186
|
-
if (!isNewerWireEvent(event, active)) return state;
|
|
187
215
|
const blocks = applyWireEventToBlocks(active.blocks, event);
|
|
188
|
-
return {
|
|
216
|
+
return {
|
|
217
|
+
...state,
|
|
218
|
+
activeTurn: { ...withoutProviderRetry(active), attempt: event.attempt, seq: event.seq, blocks, status: deriveStatus(blocks) },
|
|
219
|
+
};
|
|
189
220
|
};
|
|
190
221
|
|
|
191
222
|
/** Assistant/tool messages of the active turn are represented by live blocks; hide them. */
|
package/src/ai/executor.ts
CHANGED
|
@@ -40,6 +40,7 @@ import {
|
|
|
40
40
|
streamBlockId,
|
|
41
41
|
toolBlockId,
|
|
42
42
|
} from "./protocol";
|
|
43
|
+
import { retryTransientProviderErrors } from "./provider-retry";
|
|
43
44
|
import { assistantQuotaProvider, inferenceProvider } from "./quota-provider";
|
|
44
45
|
import { collectConversationResourceObservations } from "./resource-refs";
|
|
45
46
|
import { AiRunTimeout } from "./run-timeout";
|
|
@@ -249,6 +250,8 @@ export type ExecutorConfig = {
|
|
|
249
250
|
validateTurn?: typeof validateAiTurnRequest;
|
|
250
251
|
/** Runs after the durable turn state and final wire event are flushed. */
|
|
251
252
|
onTurnFinalized?: (event: AiTurnFinalizedEvent) => Promise<void>;
|
|
253
|
+
/** Waits before retrying a transient provider failure without Retry-After; tests shorten them. */
|
|
254
|
+
providerRetryDelaysMs?: readonly number[];
|
|
252
255
|
};
|
|
253
256
|
|
|
254
257
|
type ResolvedModel = Awaited<ReturnType<typeof resolveAiModel>>;
|
|
@@ -392,6 +395,10 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
|
|
|
392
395
|
const setRejectedCallIds = (items: Set<string>) => {
|
|
393
396
|
rejectedCallIds = items;
|
|
394
397
|
};
|
|
398
|
+
let approvedCallIds = new Set<string>();
|
|
399
|
+
const setApprovedCallIds = (items: Set<string>) => {
|
|
400
|
+
approvedCallIds = items;
|
|
401
|
+
};
|
|
395
402
|
/** nessi stream block ids (turn-scoped) that belong to tool_call blocks — their deltas are raw args JSON. */
|
|
396
403
|
const toolStreamIds = new Set<string>();
|
|
397
404
|
/** kind per open Cloud stream block id, for delta create-if-missing. */
|
|
@@ -414,6 +421,7 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
|
|
|
414
421
|
approval: approval && !allowRememberedApprovals ? { ...approval, allowAlways: false } : approval,
|
|
415
422
|
frontendMode: patch.frontendMode ?? existing?.frontendMode,
|
|
416
423
|
presentation: patch.presentation ?? existing?.presentation ?? presentations.get(rawName) ?? presentations.get(name),
|
|
424
|
+
...(approvedCallIds.has(callId) || approvedCallIds.has(displayCallId) || existing?.approved ? { approved: true } : {}),
|
|
417
425
|
};
|
|
418
426
|
toolBlocks.set(callId, block);
|
|
419
427
|
return { type: "block_set", block };
|
|
@@ -518,6 +526,7 @@ const createEventMapper = (attempt: number, seedBlocks: AiTurnBlock[], allowReme
|
|
|
518
526
|
setApprovalPolicies,
|
|
519
527
|
setApprovalReviews,
|
|
520
528
|
setRejectedCallIds,
|
|
529
|
+
setApprovedCallIds,
|
|
521
530
|
};
|
|
522
531
|
};
|
|
523
532
|
|
|
@@ -829,6 +838,7 @@ export class AiTurnExecutor {
|
|
|
829
838
|
pipeline.setApprovalPolicies(prepared.approvalPolicies);
|
|
830
839
|
const toolPresentations = new Map<string, AiToolPresentation>();
|
|
831
840
|
const rejectedToolCallIds = new Set<string>();
|
|
841
|
+
const approvedToolCallIds = new Set<string>();
|
|
832
842
|
pipeline.setPresentations(toolPresentations);
|
|
833
843
|
pipeline.setApprovalReviews(capabilityActionReviews);
|
|
834
844
|
let turnInput = config.input;
|
|
@@ -868,17 +878,19 @@ export class AiTurnExecutor {
|
|
|
868
878
|
turnInput,
|
|
869
879
|
toolPresentations,
|
|
870
880
|
rejectedToolCallIds,
|
|
881
|
+
approvedToolCallIds,
|
|
871
882
|
});
|
|
872
883
|
|
|
873
884
|
const { loopMessages, pendingRecords, resolvedRecords, turnSteers } =
|
|
874
885
|
attemptState ?? (await loadChatAttemptState(conversationId, turnId));
|
|
875
886
|
for (const action of resolvedRecords) {
|
|
876
|
-
if (action.resolvedEvent?.type !== "approval_response"
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
);
|
|
887
|
+
if (action.resolvedEvent?.type !== "approval_response") continue;
|
|
888
|
+
// A decision on a custom approval belongs to the call that asked for it.
|
|
889
|
+
const callId = action.kind === "custom_approval" ? (customApprovalParentCallId(action.callId) ?? action.callId) : action.callId;
|
|
890
|
+
(action.resolvedEvent.approved ? approvedToolCallIds : rejectedToolCallIds).add(callId);
|
|
880
891
|
}
|
|
881
892
|
pipeline.setRejectedCallIds(rejectedToolCallIds);
|
|
893
|
+
pipeline.setApprovedCallIds(approvedToolCallIds);
|
|
882
894
|
const assistantMessages = loopMessages.filter((message) => message.message.role !== "user");
|
|
883
895
|
const isFresh = assistantMessages.length === 0 && resolvedRecords.length === 0 && !skipResolvedActions;
|
|
884
896
|
|
|
@@ -1064,8 +1076,19 @@ export class AiTurnExecutor {
|
|
|
1064
1076
|
});
|
|
1065
1077
|
const priorToolRounds = toolRoundState(loopMessages);
|
|
1066
1078
|
const quotaSubject = accessSubjectForActor(material.actor);
|
|
1079
|
+
const deadline = claim.turn.deadline ? Date.parse(claim.turn.deadline) : null;
|
|
1067
1080
|
const toolRoundPolicy = applyToolRoundPolicy({
|
|
1068
|
-
provider:
|
|
1081
|
+
provider: retryTransientProviderErrors(
|
|
1082
|
+
assistantQuotaProvider(resolved.provider, config, quotaSubject, resolved.profile, turnId, conversationId),
|
|
1083
|
+
{
|
|
1084
|
+
deadline,
|
|
1085
|
+
delaysMs: this.config.providerRetryDelaysMs,
|
|
1086
|
+
onRetry: async ({ retry, delayMs, issue }) => {
|
|
1087
|
+
log.warn("AI provider call retried", { conversationId, turnId, retry, delayMs, kind: issue.kind, message: issue.message });
|
|
1088
|
+
await pipeline.emitProviderRetry();
|
|
1089
|
+
},
|
|
1090
|
+
},
|
|
1091
|
+
),
|
|
1069
1092
|
tools,
|
|
1070
1093
|
maxToolRounds: resolved.profile.maxToolRounds,
|
|
1071
1094
|
issuedToolRounds: priorToolRounds.issued,
|
|
@@ -1667,6 +1690,10 @@ class StreamPipeline {
|
|
|
1667
1690
|
this.mapper.setRejectedCallIds(callIds);
|
|
1668
1691
|
}
|
|
1669
1692
|
|
|
1693
|
+
setApprovedCallIds(callIds: Set<string>): void {
|
|
1694
|
+
this.mapper.setApprovedCallIds(callIds);
|
|
1695
|
+
}
|
|
1696
|
+
|
|
1670
1697
|
async emitBaseline(): Promise<void> {
|
|
1671
1698
|
for (const block of this.blocks) {
|
|
1672
1699
|
const seq = this.nextSeq();
|
|
@@ -1765,6 +1792,12 @@ class StreamPipeline {
|
|
|
1765
1792
|
await this.publish(event);
|
|
1766
1793
|
}
|
|
1767
1794
|
|
|
1795
|
+
/** Transient: says a model call is waiting to be retried. Snapshots never carry it. */
|
|
1796
|
+
async emitProviderRetry(): Promise<void> {
|
|
1797
|
+
const seq = this.nextSeq();
|
|
1798
|
+
await this.publish(this.envelope({ type: "provider_retry" as const, seq }));
|
|
1799
|
+
}
|
|
1800
|
+
|
|
1768
1801
|
async emitTurnFinished(status: "completed" | "failed" | "aborted", error: string | null): Promise<void> {
|
|
1769
1802
|
const seq = this.nextSeq();
|
|
1770
1803
|
await this.publish(this.envelope({ type: "turn_finished" as const, seq, status, error }) as AiWireEvent);
|
package/src/ai/protocol.ts
CHANGED
|
@@ -39,6 +39,8 @@ export type AiTurnBlock =
|
|
|
39
39
|
isError?: boolean;
|
|
40
40
|
/** Present while status is awaiting_approval. */
|
|
41
41
|
approval?: { message?: string; review?: CapabilityActionReview; allowAlways: boolean };
|
|
42
|
+
/** The user approved this call in the chat; the decided approval stays visible as a receipt. */
|
|
43
|
+
approved?: boolean;
|
|
42
44
|
/** Present for frontend tools. */
|
|
43
45
|
frontendMode?: AiFrontendToolMode;
|
|
44
46
|
/** Saved Cloud-owned display snapshot for capability calls. */
|
|
@@ -67,6 +69,8 @@ export type AiWireEvent =
|
|
|
67
69
|
| (AiWireEventBase & { type: "message_saved"; message: AiStoredMessage })
|
|
68
70
|
| (AiWireEventBase & { type: "block_set"; block: AiTurnBlock })
|
|
69
71
|
| (AiWireEventBase & { type: "block_delta"; blockId: string; blockKind: "text" | "thinking"; delta: string })
|
|
72
|
+
/** A model call failed transiently and waits for its retry. Transient: the next event of the turn ends it. */
|
|
73
|
+
| (AiWireEventBase & { type: "provider_retry" })
|
|
70
74
|
| (AiWireEventBase & {
|
|
71
75
|
type: "turn_finished";
|
|
72
76
|
status: AiTurnFinishedStatus;
|
|
@@ -85,6 +89,10 @@ export type AiTurnSnapshot = {
|
|
|
85
89
|
blocks: AiTurnBlock[];
|
|
86
90
|
modelProfileId: string | null;
|
|
87
91
|
createdAt: string;
|
|
92
|
+
/** Time the turn already waited for answered user actions such as approvals. Absent from older servers. */
|
|
93
|
+
actionWaitMs?: number;
|
|
94
|
+
/** Start of the user action the turn waits for now, or null while it works. Absent from older servers. */
|
|
95
|
+
waitingSince?: string | null;
|
|
88
96
|
};
|
|
89
97
|
|
|
90
98
|
/** Full projection seed sent as the first SSE event on every (re)connect. */
|
|
@@ -151,7 +159,12 @@ export const reconcileResolvedTurnActions = (
|
|
|
151
159
|
const event = resolvedByCallId.get(block.callId);
|
|
152
160
|
if (!event || event.callId !== block.callId) return block;
|
|
153
161
|
if (block.status === "awaiting_approval" && event.type === "approval_response") {
|
|
154
|
-
return {
|
|
162
|
+
return {
|
|
163
|
+
...block,
|
|
164
|
+
status: event.approved ? "running" : "rejected",
|
|
165
|
+
approval: undefined,
|
|
166
|
+
...(event.approved ? { approved: true } : {}),
|
|
167
|
+
};
|
|
155
168
|
}
|
|
156
169
|
if (block.status === "awaiting_client" && event.type === "tool_result") {
|
|
157
170
|
return { ...block, status: "completed", result: event.result, isError: false };
|
|
@@ -172,7 +185,7 @@ export const buildBlocksFromMessages = (
|
|
|
172
185
|
meta?: {
|
|
173
186
|
steerId?: string;
|
|
174
187
|
toolPresentations?: Record<string, AiToolPresentation>;
|
|
175
|
-
toolOutcomes?: Record<string, "rejected">;
|
|
188
|
+
toolOutcomes?: Record<string, "rejected" | "approved">;
|
|
176
189
|
} | null;
|
|
177
190
|
}[],
|
|
178
191
|
): AiTurnBlock[] => {
|
|
@@ -196,6 +209,8 @@ export const buildBlocksFromMessages = (
|
|
|
196
209
|
args: block.args,
|
|
197
210
|
status: "running",
|
|
198
211
|
presentation: meta?.toolPresentations?.[block.id],
|
|
212
|
+
// A turn that ended before an approved call returned records the approval on the call's message.
|
|
213
|
+
...(meta?.toolOutcomes?.[block.id] === "approved" ? { approved: true } : {}),
|
|
199
214
|
});
|
|
200
215
|
}
|
|
201
216
|
});
|
|
@@ -212,12 +227,13 @@ export const buildBlocksFromMessages = (
|
|
|
212
227
|
const at = toolIndex.get(message.callId);
|
|
213
228
|
const existing = at !== undefined ? blocks[at] : undefined;
|
|
214
229
|
if (existing?.kind === "tool") {
|
|
215
|
-
const
|
|
230
|
+
const outcome = meta?.toolOutcomes?.[message.callId];
|
|
216
231
|
blocks[at!] = {
|
|
217
232
|
...existing,
|
|
218
|
-
status: rejected ? "rejected" : message.isError ? "failed" : "completed",
|
|
233
|
+
status: outcome === "rejected" ? "rejected" : message.isError ? "failed" : "completed",
|
|
219
234
|
result: message.result,
|
|
220
235
|
isError: message.isError,
|
|
236
|
+
...(outcome === "approved" ? { approved: true } : {}),
|
|
221
237
|
};
|
|
222
238
|
}
|
|
223
239
|
}
|
package/src/ai/provider-fetch.ts
CHANGED
|
@@ -1,27 +1,63 @@
|
|
|
1
1
|
import { AsyncLocalStorage } from "node:async_hooks";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
4
|
+
* Observes the provider request of the inference call that is currently
|
|
5
|
+
* running, without touching the wire: when its headers and first body byte
|
|
6
|
+
* arrived, whether the provider certainly did not process it, and how long a
|
|
7
|
+
* rejecting provider asked the caller to wait. nessi adapters call the global
|
|
8
|
+
* `fetch`; the wrapper is installed once, on first use, and only observes
|
|
9
|
+
* requests made inside `runWithProviderFetchMarks`. Scopes nest, so an outer
|
|
10
|
+
* retry policy and the inner per-call accounting each see the same request.
|
|
8
11
|
*
|
|
9
12
|
* Nothing here logs: frames may carry reasoning text and headers carry keys.
|
|
10
13
|
*/
|
|
11
|
-
export type ProviderFetchMarks = {
|
|
14
|
+
export type ProviderFetchMarks = {
|
|
15
|
+
headersAt?: number;
|
|
16
|
+
firstByteAt?: number;
|
|
17
|
+
/** The request never reached the provider, or the provider answered with a status that says it did not process it. */
|
|
18
|
+
refused?: boolean;
|
|
19
|
+
retryAfterMs?: number;
|
|
20
|
+
};
|
|
12
21
|
|
|
13
|
-
const
|
|
22
|
+
const scopes = new AsyncLocalStorage<readonly ProviderFetchMarks[]>();
|
|
14
23
|
let installed = false;
|
|
15
24
|
|
|
16
25
|
export const runWithProviderFetchMarks = <T>(store: ProviderFetchMarks, run: () => Promise<T>): Promise<T> => {
|
|
17
26
|
install();
|
|
18
|
-
return
|
|
27
|
+
return scopes.run([...(scopes.getStore() ?? []), store], run);
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
/** `retry-after-ms` (OpenAI) wins over the standard `retry-after` seconds or HTTP date. */
|
|
31
|
+
export const retryAfterMs = (headers: Headers, now = Date.now()): number | undefined => {
|
|
32
|
+
const milliseconds = Number(headers.get("retry-after-ms")?.trim() || Number.NaN);
|
|
33
|
+
if (Number.isFinite(milliseconds) && milliseconds >= 0) return milliseconds;
|
|
34
|
+
const value = headers.get("retry-after")?.trim();
|
|
35
|
+
if (!value) return undefined;
|
|
36
|
+
const seconds = Number(value);
|
|
37
|
+
if (Number.isFinite(seconds)) return seconds >= 0 ? seconds * 1_000 : undefined;
|
|
38
|
+
const date = Date.parse(value);
|
|
39
|
+
return Number.isFinite(date) ? Math.max(0, date - now) : undefined;
|
|
19
40
|
};
|
|
20
41
|
|
|
21
|
-
|
|
42
|
+
/**
|
|
43
|
+
* A client error, 503, or Anthropic's overloaded 529 says the provider did not
|
|
44
|
+
* process the request. Other server errors, such as a gateway's 502 or 504, can
|
|
45
|
+
* follow processing upstream.
|
|
46
|
+
*/
|
|
47
|
+
const refusedStatus = (status: number) => (status >= 400 && status < 500) || status === 503 || status === 529;
|
|
48
|
+
|
|
49
|
+
/** Bun's codes for a request that never left: no connection, or no address for the host. */
|
|
50
|
+
const unsentCodes = new Set(["ConnectionRefused", "ENOTFOUND"]);
|
|
51
|
+
const unsent = (error: unknown) =>
|
|
52
|
+
typeof error === "object" && error !== null && "code" in error && typeof error.code === "string" && unsentCodes.has(error.code);
|
|
53
|
+
|
|
54
|
+
const firstByteObserver = (stores: readonly ProviderFetchMarks[]) =>
|
|
22
55
|
new TransformStream<Uint8Array, Uint8Array>({
|
|
23
56
|
transform(chunk, controller) {
|
|
24
|
-
if (chunk.byteLength > 0
|
|
57
|
+
if (chunk.byteLength > 0)
|
|
58
|
+
for (const store of stores) {
|
|
59
|
+
store.firstByteAt ??= Date.now();
|
|
60
|
+
}
|
|
25
61
|
controller.enqueue(chunk);
|
|
26
62
|
},
|
|
27
63
|
});
|
|
@@ -31,13 +67,29 @@ const install = (): void => {
|
|
|
31
67
|
installed = true;
|
|
32
68
|
const realFetch = globalThis.fetch;
|
|
33
69
|
const instrumented = async (input: RequestInfo | URL, init?: RequestInit): Promise<Response> => {
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
if (
|
|
37
|
-
|
|
38
|
-
|
|
70
|
+
// Only the first request of a scope is its provider request; a retry opens a new scope.
|
|
71
|
+
const stores = (scopes.getStore() ?? []).filter((store) => store.headersAt === undefined);
|
|
72
|
+
if (stores.length === 0) return realFetch(input, init);
|
|
73
|
+
let response: Response;
|
|
74
|
+
try {
|
|
75
|
+
response = await realFetch(input, init);
|
|
76
|
+
} catch (error) {
|
|
77
|
+
// A reset, an abort, or a timeout can follow delivery, so only an unsent request counts as refused.
|
|
78
|
+
if (unsent(error))
|
|
79
|
+
for (const store of stores) {
|
|
80
|
+
store.refused = true;
|
|
81
|
+
}
|
|
82
|
+
throw error;
|
|
83
|
+
}
|
|
84
|
+
const headersAt = Date.now();
|
|
85
|
+
const wait = response.ok ? undefined : retryAfterMs(response.headers, headersAt);
|
|
86
|
+
for (const store of stores) {
|
|
87
|
+
store.headersAt = headersAt;
|
|
88
|
+
store.refused = refusedStatus(response.status);
|
|
89
|
+
if (wait !== undefined) store.retryAfterMs = wait;
|
|
90
|
+
}
|
|
39
91
|
if (!response.body) return response;
|
|
40
|
-
return new Response(response.body.pipeThrough(firstByteObserver(
|
|
92
|
+
return new Response(response.body.pipeThrough(firstByteObserver(stores)), {
|
|
41
93
|
status: response.status,
|
|
42
94
|
statusText: response.statusText,
|
|
43
95
|
headers: response.headers,
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { setTimeout as delay } from "node:timers/promises";
|
|
2
|
+
import type { NessiIssue, Provider, ProviderIssue, StreamEvent, TimeoutIssue } from "@k2b/nessi/ai";
|
|
3
|
+
import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
|
|
4
|
+
|
|
5
|
+
/** Waits before the first and the second retry when the provider names none. nessi itself never retries. */
|
|
6
|
+
export const AI_PROVIDER_RETRY_DELAYS_MS: readonly number[] = [1_000, 4_000];
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* The longest Retry-After that is honored, the bound the OpenAI and Anthropic
|
|
10
|
+
* SDKs apply. A provider that asks for a longer wait is unavailable for this
|
|
11
|
+
* turn, so the call fails now with the provider's message.
|
|
12
|
+
*/
|
|
13
|
+
const AI_PROVIDER_MAX_RETRY_AFTER_MS = 60_000;
|
|
14
|
+
|
|
15
|
+
export type AiProviderRetry = {
|
|
16
|
+
/** 1 for the first retry. */
|
|
17
|
+
retry: number;
|
|
18
|
+
delayMs: number;
|
|
19
|
+
issue: ProviderIssue | TimeoutIssue;
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
const transientIssue = (issue: NessiIssue): issue is ProviderIssue | TimeoutIssue =>
|
|
23
|
+
(issue.kind === "provider_error" && issue.retryable && !issue.contextOverflow) ||
|
|
24
|
+
(issue.kind === "timeout" && issue.retryable && issue.scope !== "tool");
|
|
25
|
+
|
|
26
|
+
const isBlockEvent = (event: StreamEvent) => event.type === "block_start" || event.type === "block_delta" || event.type === "block_end";
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Repeats a model call that failed transiently (429, 5xx, a lost connection, a
|
|
30
|
+
* provider timeout) before it produced any output. Each attempt passes through
|
|
31
|
+
* `provider` again, so quota admission and accounting see every request.
|
|
32
|
+
*
|
|
33
|
+
* Once a block event was emitted the attempt is final: streamed text cannot be
|
|
34
|
+
* taken back, so a failure mid-stream still ends the call. Context overflow is
|
|
35
|
+
* left to compaction. Waits follow the provider's Retry-After, otherwise
|
|
36
|
+
* `delaysMs`, end before `deadline`, and stop on the request's abort signal.
|
|
37
|
+
*/
|
|
38
|
+
export function retryTransientProviderErrors(
|
|
39
|
+
provider: Provider,
|
|
40
|
+
options: {
|
|
41
|
+
/** Epoch milliseconds by which a wait must end; null when the turn has no run time limit. */
|
|
42
|
+
deadline: number | null;
|
|
43
|
+
delaysMs?: readonly number[];
|
|
44
|
+
/** Announces a wait before it starts. */
|
|
45
|
+
onRetry?: (retry: AiProviderRetry) => Promise<void>;
|
|
46
|
+
},
|
|
47
|
+
): Provider {
|
|
48
|
+
const delays = options.delaysMs ?? AI_PROVIDER_RETRY_DELAYS_MS;
|
|
49
|
+
const waitFor = (retry: number, marks: ProviderFetchMarks, signal: AbortSignal | undefined): number | null => {
|
|
50
|
+
if (retry >= delays.length || signal?.aborted) return null;
|
|
51
|
+
const delayMs = marks.retryAfterMs ?? delays[retry]!;
|
|
52
|
+
if (delayMs > AI_PROVIDER_MAX_RETRY_AFTER_MS) return null;
|
|
53
|
+
if (options.deadline !== null && Date.now() + delayMs >= options.deadline) return null;
|
|
54
|
+
return delayMs;
|
|
55
|
+
};
|
|
56
|
+
return {
|
|
57
|
+
name: provider.name,
|
|
58
|
+
family: provider.family,
|
|
59
|
+
model: provider.model,
|
|
60
|
+
contextWindow: provider.contextWindow,
|
|
61
|
+
capabilities: provider.capabilities,
|
|
62
|
+
complete: (request) => provider.complete(request),
|
|
63
|
+
stream: async function* (request) {
|
|
64
|
+
for (let retry = 0; ; retry += 1) {
|
|
65
|
+
const marks: ProviderFetchMarks = {};
|
|
66
|
+
const events = provider.stream(request)[Symbol.asyncIterator]();
|
|
67
|
+
const next = () => runWithProviderFetchMarks(marks, () => events.next());
|
|
68
|
+
// Events before the first block (usage, issues) are held so that a retried attempt leaves no trace.
|
|
69
|
+
const held: StreamEvent[] = [];
|
|
70
|
+
let committed = false;
|
|
71
|
+
let pending: AiProviderRetry | null = null;
|
|
72
|
+
try {
|
|
73
|
+
for (let step = await next(); !step.done; step = await next()) {
|
|
74
|
+
const event = step.value;
|
|
75
|
+
if (!committed && event.type === "issue" && transientIssue(event.issue)) {
|
|
76
|
+
const delayMs = waitFor(retry, marks, request.signal);
|
|
77
|
+
if (delayMs !== null) {
|
|
78
|
+
pending = { retry: retry + 1, delayMs, issue: event.issue };
|
|
79
|
+
break;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
if (!committed && !isBlockEvent(event)) {
|
|
83
|
+
held.push(event);
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
86
|
+
if (!committed) {
|
|
87
|
+
committed = true;
|
|
88
|
+
yield* held.splice(0);
|
|
89
|
+
}
|
|
90
|
+
yield event;
|
|
91
|
+
}
|
|
92
|
+
} finally {
|
|
93
|
+
// Settles the failed attempt's accounting before the wait starts.
|
|
94
|
+
await events.return?.();
|
|
95
|
+
}
|
|
96
|
+
if (!pending) {
|
|
97
|
+
yield* held;
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
await options.onRetry?.(pending);
|
|
101
|
+
await delay(pending.delayMs, undefined, { signal: request.signal });
|
|
102
|
+
}
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
}
|
package/src/ai/quota-provider.ts
CHANGED
|
@@ -5,7 +5,7 @@ import type { AccessSubject } from "../server/services/access";
|
|
|
5
5
|
import { logger } from "../services/logging";
|
|
6
6
|
import { isAssistantChatTurn } from "./assistant-models";
|
|
7
7
|
import { AiBackgroundAdmissionError, type AiCallContext, type AiCallDetails, beginAiCall, finishAiCall } from "./inference-calls";
|
|
8
|
-
import { runWithProviderFetchMarks } from "./provider-fetch";
|
|
8
|
+
import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
|
|
9
9
|
import type { AiModelProfile } from "./types";
|
|
10
10
|
|
|
11
11
|
const log = logger("ai:quotas");
|
|
@@ -93,7 +93,7 @@ export function inferenceProvider(
|
|
|
93
93
|
let status: "ok" | "failed" = "failed";
|
|
94
94
|
let error: string | undefined;
|
|
95
95
|
const requestStartedAt = Date.now();
|
|
96
|
-
const marks:
|
|
96
|
+
const marks: ProviderFetchMarks = {};
|
|
97
97
|
try {
|
|
98
98
|
const result = await runWithProviderFetchMarks(marks, () =>
|
|
99
99
|
provider.complete({ ...request, maxOutputTokens: call.maxOutputTokens }),
|
|
@@ -112,7 +112,9 @@ export function inferenceProvider(
|
|
|
112
112
|
throw thrown;
|
|
113
113
|
} finally {
|
|
114
114
|
call.stop();
|
|
115
|
-
|
|
115
|
+
// A request the provider refused or never received cost nothing; any other failure may have been processed.
|
|
116
|
+
if (!usage && status === "failed")
|
|
117
|
+
usage = marks.refused ? { input: 0, output: 0 } : { input: call.inputTokens, output: 0, estimated: true };
|
|
116
118
|
const cancelled = request.signal?.aborted === true;
|
|
117
119
|
await finish(call.id, usage, cancelled && status === "failed" ? "aborted" : status, {
|
|
118
120
|
error: cancelled ? null : error,
|
|
@@ -133,7 +135,7 @@ export function inferenceProvider(
|
|
|
133
135
|
const outputBlocks = new Map<string, number>();
|
|
134
136
|
let generated = false;
|
|
135
137
|
const requestStartedAt = Date.now();
|
|
136
|
-
const marks:
|
|
138
|
+
const marks: ProviderFetchMarks & { firstBlockAt?: number } = {};
|
|
137
139
|
// The wrapped adapter reads lazily, so the request only leaves once the first pull runs inside the marked scope.
|
|
138
140
|
const events = provider.stream({ ...request, maxOutputTokens: call.maxOutputTokens })[Symbol.asyncIterator]();
|
|
139
141
|
const next = () => runWithProviderFetchMarks(marks, () => events.next());
|
|
@@ -158,7 +160,14 @@ export function inferenceProvider(
|
|
|
158
160
|
outputBlocks.set(event.blockId, size);
|
|
159
161
|
}
|
|
160
162
|
if (event.type === "block_start" || event.type === "block_delta" || event.type === "block_end") generated = true;
|
|
161
|
-
|
|
163
|
+
// A provider that refused the request, or never received it, did not process it.
|
|
164
|
+
if (
|
|
165
|
+
event.type === "issue" &&
|
|
166
|
+
event.issue.kind === "provider_error" &&
|
|
167
|
+
(event.issue.contextOverflow || marks.refused) &&
|
|
168
|
+
!generated &&
|
|
169
|
+
!usage
|
|
170
|
+
)
|
|
162
171
|
usage = { input: 0, output: 0 };
|
|
163
172
|
if (event.type === "usage") {
|
|
164
173
|
if (event.finishReason === "aborted" || event.finishReason === "interrupted" || event.finishReason === "error") failed = true;
|