@skydiveai/pi-server 0.1.0-beta.805 → 0.1.250
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.mts +285 -79
- package/dist/index.mjs +314 -68
- package/package.json +2 -9
package/dist/index.mjs
CHANGED
|
@@ -259,7 +259,7 @@ function parseShellEnv(request) {
|
|
|
259
259
|
const env = {};
|
|
260
260
|
for (const [k, v] of Object.entries(parsed)) if (typeof v === "string") env[k] = v;
|
|
261
261
|
return Object.keys(env).length > 0 ? env : null;
|
|
262
|
-
} catch {
|
|
262
|
+
} catch (_err) {
|
|
263
263
|
return null;
|
|
264
264
|
}
|
|
265
265
|
}
|
|
@@ -283,12 +283,12 @@ function createSSEResponse(handler) {
|
|
|
283
283
|
write(chunk) {
|
|
284
284
|
try {
|
|
285
285
|
controller.enqueue(encoder.encode(chunk));
|
|
286
|
-
} catch {}
|
|
286
|
+
} catch (_err) {}
|
|
287
287
|
},
|
|
288
288
|
close() {
|
|
289
289
|
try {
|
|
290
290
|
controller.close();
|
|
291
|
-
} catch {}
|
|
291
|
+
} catch (_err) {}
|
|
292
292
|
}
|
|
293
293
|
};
|
|
294
294
|
const keepalive = setInterval(() => {
|
|
@@ -298,13 +298,13 @@ function createSSEResponse(handler) {
|
|
|
298
298
|
clearInterval(keepalive);
|
|
299
299
|
try {
|
|
300
300
|
controller.close();
|
|
301
|
-
} catch {}
|
|
301
|
+
} catch (_err) {}
|
|
302
302
|
}, (err) => {
|
|
303
303
|
logger.error({ err }, "SSE handler crashed — stream closed without response");
|
|
304
304
|
clearInterval(keepalive);
|
|
305
305
|
try {
|
|
306
306
|
controller.close();
|
|
307
|
-
} catch {}
|
|
307
|
+
} catch (_err) {}
|
|
308
308
|
});
|
|
309
309
|
return new Response(stream, {
|
|
310
310
|
status: 200,
|
|
@@ -373,6 +373,57 @@ async function runConversation({ session, prompt, images, log, postPrompt }) {
|
|
|
373
373
|
});
|
|
374
374
|
}
|
|
375
375
|
/**
|
|
376
|
+
* Track the agent session's built-in model-call auto-retry so it leaves a
|
|
377
|
+
* trace.
|
|
378
|
+
*
|
|
379
|
+
* `AgentSession` already restarts a failed assistant turn in place (via
|
|
380
|
+
* `agent.continue()`, so no prompt is replayed and no tool re-executes) for the
|
|
381
|
+
* transient provider/transport failures pi classifies as retryable -- dropped
|
|
382
|
+
* streams, `terminated`, 5xx, overloaded, rate limits. It is on by default,
|
|
383
|
+
* with its own budget and backoff, and it emits `auto_retry_start` /
|
|
384
|
+
* `auto_retry_end` around each attempt.
|
|
385
|
+
*
|
|
386
|
+
* Nothing consumed those events, so a retry left no trace anywhere: a call that
|
|
387
|
+
* succeeded first try and one that burned the whole budget before failing
|
|
388
|
+
* produced the same terminal error, and the fleet-wide retry rate was
|
|
389
|
+
* unmeasurable. That gap is why a 2026-08-16 investigation into three runs lost
|
|
390
|
+
* to `provider_error: terminated` could not tell whether the budget had run out
|
|
391
|
+
* (ANY-7101).
|
|
392
|
+
*
|
|
393
|
+
* Exposed as a handler rather than its own `session.subscribe` call so each
|
|
394
|
+
* protocol feeds it from the single subscription it already owns -- one
|
|
395
|
+
* subscriber, explicit ordering.
|
|
396
|
+
*
|
|
397
|
+
* `attempts()` reports what has been spent so far, so a terminal error can
|
|
398
|
+
* carry the count to the worker, where it lands in a log group we can query
|
|
399
|
+
* fleet-wide (the sandbox's own logs are not).
|
|
400
|
+
*/
|
|
401
|
+
function createAutoRetryObserver(log) {
|
|
402
|
+
let attempts = 0;
|
|
403
|
+
return {
|
|
404
|
+
observe(event) {
|
|
405
|
+
if (event.type === "auto_retry_start") {
|
|
406
|
+
attempts = typeof event.attempt === "number" ? event.attempt : attempts + 1;
|
|
407
|
+
log.warn({
|
|
408
|
+
event: "model_call_auto_retry",
|
|
409
|
+
attempt: event.attempt,
|
|
410
|
+
max_attempts: event.maxAttempts,
|
|
411
|
+
delay_ms: event.delayMs,
|
|
412
|
+
error_message: event.errorMessage
|
|
413
|
+
}, "retrying failed model call in place");
|
|
414
|
+
return;
|
|
415
|
+
}
|
|
416
|
+
if (event.type === "auto_retry_end") log.warn({
|
|
417
|
+
event: "model_call_auto_retry_end",
|
|
418
|
+
attempt: event.attempt,
|
|
419
|
+
success: event.success,
|
|
420
|
+
final_error: event.finalError
|
|
421
|
+
}, event.success ? "model call recovered after retry" : "model call retries exhausted");
|
|
422
|
+
},
|
|
423
|
+
attempts: () => attempts
|
|
424
|
+
};
|
|
425
|
+
}
|
|
426
|
+
/**
|
|
376
427
|
* Hard-stop the in-flight turn for a session. Shared by every protocol's
|
|
377
428
|
* `/:id/abort` route: a cancel signals the stop explicitly instead of relying
|
|
378
429
|
* on a dropped connection. `session.abort()` interrupts the turn and resolves
|
|
@@ -419,7 +470,10 @@ async function extractParts(parts, log) {
|
|
|
419
470
|
}
|
|
420
471
|
if (part.url != null && part.mediaType?.startsWith("image/")) {
|
|
421
472
|
try {
|
|
422
|
-
images.push(await fetchImageAsBase64({
|
|
473
|
+
images.push(await fetchImageAsBase64({
|
|
474
|
+
url: part.url,
|
|
475
|
+
userAgent: null
|
|
476
|
+
}));
|
|
423
477
|
} catch (err) {
|
|
424
478
|
log.warn({
|
|
425
479
|
event: "a2a_image_fetch_failed",
|
|
@@ -687,6 +741,7 @@ const thinkingLevelMapSchema = z.object({
|
|
|
687
741
|
*/
|
|
688
742
|
const openaiCompletionsCompatSchema = z.object({
|
|
689
743
|
supportsReasoningEffort: z.boolean().optional(),
|
|
744
|
+
supportsStore: z.boolean().optional(),
|
|
690
745
|
requiresThinkingAsText: z.boolean().optional(),
|
|
691
746
|
thinkingFormat: z.enum([
|
|
692
747
|
"openai",
|
|
@@ -726,15 +781,26 @@ const baseSpecShape = {
|
|
|
726
781
|
*/
|
|
727
782
|
apiKey: z.string().min(1).optional()
|
|
728
783
|
};
|
|
729
|
-
const modelSpecSchema = z.discriminatedUnion("api", [
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
784
|
+
const modelSpecSchema = z.discriminatedUnion("api", [
|
|
785
|
+
z.object({
|
|
786
|
+
...baseSpecShape,
|
|
787
|
+
api: z.literal("openai-completions"),
|
|
788
|
+
compat: openaiCompletionsCompatSchema.optional()
|
|
789
|
+
}),
|
|
790
|
+
z.object({
|
|
791
|
+
...baseSpecShape,
|
|
792
|
+
api: z.literal("anthropic-messages"),
|
|
793
|
+
compat: anthropicMessagesCompatSchema.optional()
|
|
794
|
+
}),
|
|
795
|
+
z.object({
|
|
796
|
+
...baseSpecShape,
|
|
797
|
+
api: z.literal("openai-responses")
|
|
798
|
+
}),
|
|
799
|
+
z.object({
|
|
800
|
+
...baseSpecShape,
|
|
801
|
+
api: z.literal("openai-codex-responses")
|
|
802
|
+
}).omit({ apiKey: true })
|
|
803
|
+
]);
|
|
738
804
|
/**
|
|
739
805
|
* Parse a request-body `x_model`. Null when absent or invalid — never a
|
|
740
806
|
* request rejection, so a malformed spec degrades to the legacy body-model
|
|
@@ -781,6 +847,14 @@ function buildModelFromSpec(spec) {
|
|
|
781
847
|
api: spec.api,
|
|
782
848
|
...spec.compat ? { compat: spec.compat } : {}
|
|
783
849
|
};
|
|
850
|
+
case "openai-responses": return {
|
|
851
|
+
...base,
|
|
852
|
+
api: spec.api
|
|
853
|
+
};
|
|
854
|
+
case "openai-codex-responses": return {
|
|
855
|
+
...base,
|
|
856
|
+
api: spec.api
|
|
857
|
+
};
|
|
784
858
|
default: throw new Error(`Unhandled x_model api: ${String(spec)}`);
|
|
785
859
|
}
|
|
786
860
|
}
|
|
@@ -802,7 +876,7 @@ function resolveRequestModel({ modelInput, modelSpecInput, defaultModel, log })
|
|
|
802
876
|
* always wins.
|
|
803
877
|
*/
|
|
804
878
|
function withSpecApiKey(keys, modelSpec) {
|
|
805
|
-
if (!modelSpec
|
|
879
|
+
if (!modelSpec || !("apiKey" in modelSpec) || !modelSpec.apiKey || keys[modelSpec.provider]) return keys;
|
|
806
880
|
return {
|
|
807
881
|
...keys,
|
|
808
882
|
[modelSpec.provider]: modelSpec.apiKey
|
|
@@ -865,6 +939,18 @@ function providerId(err, message) {
|
|
|
865
939
|
*/
|
|
866
940
|
const CONTEXT_OVERFLOW_MESSAGE = /prompt is too long|context[_ ]length[_ ]exceeded|maximum context length|exceeds maximum token limit|input token count|too many tokens|exceeds the maximum/i;
|
|
867
941
|
/**
|
|
942
|
+
* A content-filter / model-safety termination message from any provider.
|
|
943
|
+
* Used by BOTH the classifyCode message fallback and isModelProviderError's
|
|
944
|
+
* gate (a shape recognized there must pass this gate or it never gets
|
|
945
|
+
* classified), so it lives in one place rather than two copies kept "in
|
|
946
|
+
* sync" by comment. Verified shapes (see classifyCode):
|
|
947
|
+
* - "Provider finish_reason: content_filter" (Anthropic/gateway 200 stream)
|
|
948
|
+
* - "This request triggered restrictions on violative cyber content ..."
|
|
949
|
+
* - "This request triggered cyber-related safeguards. ..."
|
|
950
|
+
* - "This content was flagged for possible cybersecurity risk. ..."
|
|
951
|
+
*/
|
|
952
|
+
const CONTENT_FILTER_MESSAGE = /finish_reason:\s*content_filter|content[_ ]filter|content[_ ]management[_ ]policy|violative cyber content|cyber[- ]related safeguards|cybersecurity risk|cyber verification program/i;
|
|
953
|
+
/**
|
|
868
954
|
* Map a provider error to a stable code. Prefers the provider's machine code
|
|
869
955
|
* (OpenAI `error.code`, OpenRouter `error.metadata.error_type`) over message
|
|
870
956
|
* text, then falls back to status + message regex for providers that don't
|
|
@@ -877,7 +963,7 @@ function classifyCode(status, message, code) {
|
|
|
877
963
|
if (status === 429 || code === "rate_limit_exceeded" || /rate[_ ]limit|too many requests|resource[_ ]exhausted/i.test(message)) return "rate_limited";
|
|
878
964
|
if (status === 529 || code === "provider_overloaded" || /overloaded/i.test(message)) return "provider_overloaded";
|
|
879
965
|
if (status === 503 || status === 504 || code === "provider_unavailable" || /no healthy upstream|upstream request timeout|stream timeout|service unavailable|unavailable|gateway/i.test(message)) return "provider_unavailable";
|
|
880
|
-
if (code === "content_filter" ||
|
|
966
|
+
if (code === "content_filter" || CONTENT_FILTER_MESSAGE.test(message)) return "content_filter";
|
|
881
967
|
return "provider_error";
|
|
882
968
|
}
|
|
883
969
|
/**
|
|
@@ -892,7 +978,7 @@ function isModelProviderError(err) {
|
|
|
892
978
|
if (numericStatus(e) !== null) return true;
|
|
893
979
|
if (e.error && typeof e.error === "object") return true;
|
|
894
980
|
const msg = messageText(err);
|
|
895
|
-
return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits
|
|
981
|
+
return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits/i.test(msg) || CONTENT_FILTER_MESSAGE.test(msg);
|
|
896
982
|
}
|
|
897
983
|
/** Build the structured, forwardable error from a thrown model-provider error. */
|
|
898
984
|
function toModelProviderError(err) {
|
|
@@ -948,7 +1034,10 @@ async function extractContent$1(content, log) {
|
|
|
948
1034
|
data: src.data
|
|
949
1035
|
});
|
|
950
1036
|
else if (src?.type === "url" && src.url) try {
|
|
951
|
-
images.push(await fetchImageAsBase64({
|
|
1037
|
+
images.push(await fetchImageAsBase64({
|
|
1038
|
+
url: src.url,
|
|
1039
|
+
userAgent: null
|
|
1040
|
+
}));
|
|
952
1041
|
} catch (err) {
|
|
953
1042
|
log.warn({
|
|
954
1043
|
event: "image_fetch_failed",
|
|
@@ -1037,7 +1126,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1037
1126
|
let parsed;
|
|
1038
1127
|
try {
|
|
1039
1128
|
parsed = JSON.parse(await request.text());
|
|
1040
|
-
} catch {
|
|
1129
|
+
} catch (_err) {
|
|
1041
1130
|
return jsonError(400, "invalid json");
|
|
1042
1131
|
}
|
|
1043
1132
|
const { messages, stream = false, model: modelInput, system, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -1060,7 +1149,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1060
1149
|
const baseSystemPrompt = extractSystemPrompt(system);
|
|
1061
1150
|
const { sessionId } = parseSessionId(request);
|
|
1062
1151
|
const shellEnv = parseShellEnv(request);
|
|
1063
|
-
const { session
|
|
1152
|
+
const { session } = await ctx.createSession({
|
|
1064
1153
|
cwd: ctx.cwd,
|
|
1065
1154
|
sessionId,
|
|
1066
1155
|
perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
|
|
@@ -1083,7 +1172,6 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1083
1172
|
if (stream) {
|
|
1084
1173
|
const response = runStream$2({
|
|
1085
1174
|
session,
|
|
1086
|
-
sm,
|
|
1087
1175
|
prompt,
|
|
1088
1176
|
images,
|
|
1089
1177
|
id,
|
|
@@ -1099,7 +1187,6 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1099
1187
|
}
|
|
1100
1188
|
const response = await runBlocking$2({
|
|
1101
1189
|
session,
|
|
1102
|
-
sm,
|
|
1103
1190
|
prompt,
|
|
1104
1191
|
images,
|
|
1105
1192
|
id,
|
|
@@ -1113,7 +1200,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1113
1200
|
if (traceId) response.headers.set("x-trace-id", traceId);
|
|
1114
1201
|
return response;
|
|
1115
1202
|
}
|
|
1116
|
-
function runStream$2({ session,
|
|
1203
|
+
function runStream$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
|
|
1117
1204
|
return createSSEResponse(async (writer) => {
|
|
1118
1205
|
let contentBlockIndex = 0;
|
|
1119
1206
|
let textBlockOpen = false;
|
|
@@ -1159,12 +1246,15 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1159
1246
|
textBlockOpen = false;
|
|
1160
1247
|
};
|
|
1161
1248
|
let capturedModelError = null;
|
|
1249
|
+
let sawModelErrorStop = false;
|
|
1250
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
1162
1251
|
const emitProviderError = (providerError) => {
|
|
1163
1252
|
log.warn({
|
|
1164
1253
|
event: "model_provider_error",
|
|
1165
1254
|
code: providerError.code,
|
|
1166
1255
|
upstream_status: providerError.upstreamStatus,
|
|
1167
|
-
provider: providerError.provider
|
|
1256
|
+
provider: providerError.provider,
|
|
1257
|
+
retry_attempts: autoRetry.attempts()
|
|
1168
1258
|
}, "forwarding model-provider error to client");
|
|
1169
1259
|
sseEvent(writer, "error", {
|
|
1170
1260
|
type: "error",
|
|
@@ -1174,15 +1264,24 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1174
1264
|
x_model_provider_error: {
|
|
1175
1265
|
code: providerError.code,
|
|
1176
1266
|
provider: providerError.provider,
|
|
1177
|
-
upstream_status: providerError.upstreamStatus
|
|
1267
|
+
upstream_status: providerError.upstreamStatus,
|
|
1268
|
+
retry_attempts: autoRetry.attempts()
|
|
1178
1269
|
}
|
|
1179
1270
|
}
|
|
1180
1271
|
});
|
|
1181
1272
|
};
|
|
1182
1273
|
session.subscribe((event) => {
|
|
1183
1274
|
const ev = event;
|
|
1275
|
+
autoRetry.observe(ev);
|
|
1276
|
+
if (ev.type === "auto_retry_end" && ev.success === true) {
|
|
1277
|
+
capturedModelError = null;
|
|
1278
|
+
sawModelErrorStop = false;
|
|
1279
|
+
}
|
|
1184
1280
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
1185
|
-
if (endedMessage?.stopReason === "error" &&
|
|
1281
|
+
if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
|
|
1282
|
+
sawModelErrorStop = true;
|
|
1283
|
+
if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
|
|
1284
|
+
}
|
|
1186
1285
|
if (ev.type !== "message_update") return;
|
|
1187
1286
|
const inner = ev.assistantMessageEvent;
|
|
1188
1287
|
if (!inner) return;
|
|
@@ -1240,6 +1339,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1240
1339
|
postPrompt
|
|
1241
1340
|
});
|
|
1242
1341
|
if (capturedModelError !== null) emitProviderError(toModelProviderError(new Error(capturedModelError)));
|
|
1342
|
+
else if (sawModelErrorStop) emitProviderError(toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message")));
|
|
1243
1343
|
} catch (err) {
|
|
1244
1344
|
log.error({
|
|
1245
1345
|
err,
|
|
@@ -1267,7 +1367,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1267
1367
|
sseEvent(writer, "message_stop", { type: "message_stop" });
|
|
1268
1368
|
});
|
|
1269
1369
|
}
|
|
1270
|
-
async function runBlocking$2({ session,
|
|
1370
|
+
async function runBlocking$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
|
|
1271
1371
|
let text = "";
|
|
1272
1372
|
const toolUseBlocks = [];
|
|
1273
1373
|
const tcByContentIdx = /* @__PURE__ */ new Map();
|
|
@@ -1318,7 +1418,7 @@ async function runBlocking$2({ session, sm, prompt, images, id, sessionId, model
|
|
|
1318
1418
|
}
|
|
1319
1419
|
for (const block of toolUseBlocks) if (block.type === "tool_use" && typeof block.input === "string") try {
|
|
1320
1420
|
block.input = JSON.parse(block.input);
|
|
1321
|
-
} catch {
|
|
1421
|
+
} catch (_err) {
|
|
1322
1422
|
block.input = {};
|
|
1323
1423
|
}
|
|
1324
1424
|
const content = [];
|
|
@@ -1409,34 +1509,79 @@ const nestedProxyEnvelopeSchema = z.object({
|
|
|
1409
1509
|
blockReason: z.enum(BILLING_BLOCK_REASONS),
|
|
1410
1510
|
message: z.string()
|
|
1411
1511
|
}).strict();
|
|
1512
|
+
/**
|
|
1513
|
+
* pi-ai's openai-compatible providers (openrouter, xai, groq, deepseek, …) can
|
|
1514
|
+
* NOT fold a proxy's non-2xx body into `error.message`, so `formatProviderError`
|
|
1515
|
+
* (@earendil-works/pi-ai utils/error-body) composes the display string as
|
|
1516
|
+
* `"<status>: <body>"` or, with a provider label, `"<prefix> (<status>): <body>"`.
|
|
1517
|
+
* The billing gate's 402 body therefore reaches us wrapped, e.g.
|
|
1518
|
+
* `"402: {\"error\":{...},\"blockReason\":\"insufficient_balance\",\"message\":...}"`
|
|
1519
|
+
* or `"OpenRouter (402): {...}"`. Neither the bare `JSON.parse` nor the `"402 "`
|
|
1520
|
+
* (space) strip below recognizes that, so a real billing block from an
|
|
1521
|
+
* openrouter-routed model degrades to the generic model-provider error. Peel a
|
|
1522
|
+
* single leading `"<status>: "` / `"<prefix> (<status>): "` wrapper off the
|
|
1523
|
+
* front so the recovered body flows through the existing shape checks. Returns
|
|
1524
|
+
* the message unchanged when no wrapper is present.
|
|
1525
|
+
*/
|
|
1526
|
+
function unwrapOpenAICompatStatusPrefix(message) {
|
|
1527
|
+
const withPrefix = message.match(/^.+ \(\d{3}\): ([\s\S]+)$/);
|
|
1528
|
+
if (withPrefix?.[1] !== void 0) return withPrefix[1];
|
|
1529
|
+
const bare = message.match(/^\d{3}: ([\s\S]+)$/);
|
|
1530
|
+
if (bare?.[1] !== void 0) return bare[1];
|
|
1531
|
+
return message;
|
|
1532
|
+
}
|
|
1412
1533
|
function parsePrefixedSignal(message) {
|
|
1413
1534
|
const normalized = message.startsWith("402 ") ? message.slice(4) : message;
|
|
1414
1535
|
if (!normalized.startsWith(BILLING_BLOCKED_PROVIDER_SIGNAL_PREFIX)) return null;
|
|
1415
1536
|
try {
|
|
1416
1537
|
const parsed = payloadSchema.safeParse(JSON.parse(normalized.slice(26)));
|
|
1417
1538
|
return parsed.success ? parsed.data : null;
|
|
1418
|
-
} catch {
|
|
1539
|
+
} catch (_err) {
|
|
1419
1540
|
return null;
|
|
1420
1541
|
}
|
|
1421
1542
|
}
|
|
1422
1543
|
/**
|
|
1544
|
+
* The sandbox proxy's other 402 envelope (verified prod 2026-09-20): the
|
|
1545
|
+
* billing gate's body surfaces with the type/code at TOP level and the
|
|
1546
|
+
* versioned signal nested inside `message` —
|
|
1547
|
+
* {"type":"billing_blocked","code":"billing_blocked","message":
|
|
1548
|
+
* "ANYONE_BILLING_BLOCKED_V1:{\"blockReason\":\"insufficient_balance\",…}"}
|
|
1549
|
+
* — with no blockReason/error wrapper for nestedProxyEnvelopeSchema, so it
|
|
1550
|
+
* used to fall through and degrade to a generic model-provider error that
|
|
1551
|
+
* read as "The model stopped responding" on an out-of-credit org. The outer
|
|
1552
|
+
* shape is validated with a schema instead of ad-hoc casting, and only the
|
|
1553
|
+
* top-level `message` is peeled; a signal nested inside `error.message` is
|
|
1554
|
+
* handled below through nestedProxyEnvelopeSchema's full integrity checks
|
|
1555
|
+
* (validating type, code, blockReason, and message agreement).
|
|
1556
|
+
*/
|
|
1557
|
+
const topLevelMessageEnvelopeSchema = z.object({ message: z.string().optional() }).passthrough();
|
|
1558
|
+
/**
|
|
1423
1559
|
* Decode only the versioned platform envelope. Anthropic and OpenAI prefix the
|
|
1424
1560
|
* nested error message with HTTP 402. Google's SDK instead preserves the full
|
|
1425
1561
|
* proxy response as JSON, so that outer shape is validated separately.
|
|
1426
1562
|
*/
|
|
1427
1563
|
function parseBillingBlockedProviderSignal(message) {
|
|
1428
|
-
const
|
|
1564
|
+
const unwrapped = unwrapOpenAICompatStatusPrefix(message);
|
|
1565
|
+
const direct = parsePrefixedSignal(unwrapped);
|
|
1429
1566
|
if (direct) return direct;
|
|
1567
|
+
let body;
|
|
1430
1568
|
try {
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
const nested = parsePrefixedSignal(envelope.data.error.message);
|
|
1434
|
-
const topLevel = parsePrefixedSignal(envelope.data.message);
|
|
1435
|
-
if (!nested || nested.blockReason !== envelope.data.blockReason || envelope.data.message !== nested.message && (!topLevel || topLevel.blockReason !== nested.blockReason || topLevel.message !== nested.message)) return null;
|
|
1436
|
-
return nested;
|
|
1437
|
-
} catch {
|
|
1569
|
+
body = JSON.parse(unwrapped);
|
|
1570
|
+
} catch (_err) {
|
|
1438
1571
|
return null;
|
|
1439
1572
|
}
|
|
1573
|
+
if (body === null || typeof body !== "object") return null;
|
|
1574
|
+
const nested = topLevelMessageEnvelopeSchema.safeParse(body);
|
|
1575
|
+
if (nested.success && nested.data.message) {
|
|
1576
|
+
const signal = parsePrefixedSignal(nested.data.message);
|
|
1577
|
+
if (signal) return signal;
|
|
1578
|
+
}
|
|
1579
|
+
const envelope = nestedProxyEnvelopeSchema.safeParse(body);
|
|
1580
|
+
if (!envelope.success) return null;
|
|
1581
|
+
const nestedError = parsePrefixedSignal(envelope.data.error.message);
|
|
1582
|
+
const topLevel = parsePrefixedSignal(envelope.data.message);
|
|
1583
|
+
if (!nestedError || nestedError.blockReason !== envelope.data.blockReason || envelope.data.message !== nestedError.message && (!topLevel || topLevel.blockReason !== nestedError.blockReason || topLevel.message !== nestedError.message)) return null;
|
|
1584
|
+
return nestedError;
|
|
1440
1585
|
}
|
|
1441
1586
|
//#endregion
|
|
1442
1587
|
//#region src/protocols/chat-completions.ts
|
|
@@ -1623,7 +1768,7 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
|
|
|
1623
1768
|
let args = {};
|
|
1624
1769
|
try {
|
|
1625
1770
|
args = tc.function.arguments ? JSON.parse(tc.function.arguments) : {};
|
|
1626
|
-
} catch {
|
|
1771
|
+
} catch (_err) {
|
|
1627
1772
|
args = { _raw: tc.function.arguments };
|
|
1628
1773
|
}
|
|
1629
1774
|
contentArr.push({
|
|
@@ -1665,10 +1810,33 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
|
|
|
1665
1810
|
}
|
|
1666
1811
|
}
|
|
1667
1812
|
}
|
|
1813
|
+
/**
|
|
1814
|
+
* Each pi/session message carries the real wall-clock `timestamp` the
|
|
1815
|
+
* provider stamped when it produced that message (see @earendil-works/pi-ai
|
|
1816
|
+
* Message). Forward it verbatim as `x_created_at` (epoch ms) on the OAI
|
|
1817
|
+
* trailer message so the consumer (anyone-messaging completeRun) can stamp
|
|
1818
|
+
* each persisted segment at its true emit time and interleave a mid-run steer
|
|
1819
|
+
* by createdAt — instead of positionally reconstructing per-turn times, which
|
|
1820
|
+
* drifts on parallel tool calls and provider-split replies.
|
|
1821
|
+
*/
|
|
1668
1822
|
function messageTimestampMs(m) {
|
|
1669
1823
|
const ts = m?.timestamp;
|
|
1670
1824
|
return typeof ts === "number" && Number.isFinite(ts) ? ts : null;
|
|
1671
1825
|
}
|
|
1826
|
+
/**
|
|
1827
|
+
* Drop assistant messages recorded as failed model calls (stopReason 'error')
|
|
1828
|
+
* from the session-trailer slice. pi keeps an aborted attempt in the session
|
|
1829
|
+
* when auto-retry re-runs the model call in place, so a recovered turn carries
|
|
1830
|
+
* both the errored attempt and the successful retry. The trailer is only built
|
|
1831
|
+
* on the clean path -- a turn that exhausts its retries fails and never emits
|
|
1832
|
+
* one -- so any stopReason 'error' assistant message in the slice is an
|
|
1833
|
+
* aborted attempt the retry replaced, not a message downstream should join.
|
|
1834
|
+
* A user abort ('aborted') is a cancel, not a retried provider failure, and is
|
|
1835
|
+
* kept, matching the terminal-error capture above.
|
|
1836
|
+
*/
|
|
1837
|
+
function dropAbortedModelAttempts(messages) {
|
|
1838
|
+
return messages.filter((m) => !(m?.role === "assistant" && m.stopReason === "error" && m.errorMessage !== "aborted"));
|
|
1839
|
+
}
|
|
1672
1840
|
function piMessagesToOpenAI(messages) {
|
|
1673
1841
|
const out = [];
|
|
1674
1842
|
for (const m of messages) {
|
|
@@ -1726,7 +1894,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1726
1894
|
let parsed;
|
|
1727
1895
|
try {
|
|
1728
1896
|
parsed = JSON.parse(await request.text());
|
|
1729
|
-
} catch {
|
|
1897
|
+
} catch (_err) {
|
|
1730
1898
|
return jsonError(400, "invalid json");
|
|
1731
1899
|
}
|
|
1732
1900
|
const { messages, stream = false, model: modelInput, thinkingLevel: thinkingInput, context_window: contextWindowInput, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -1738,7 +1906,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1738
1906
|
}
|
|
1739
1907
|
if (lastUserIdx < 0) return jsonError(400, "no user message in history");
|
|
1740
1908
|
const lastUser = messages[lastUserIdx];
|
|
1741
|
-
const reqUA = request.headers.get("user-agent") ??
|
|
1909
|
+
const reqUA = request.headers.get("user-agent") ?? null;
|
|
1742
1910
|
const { text: prompt, images } = await extractContent(lastUser.content, {
|
|
1743
1911
|
userAgent: reqUA,
|
|
1744
1912
|
log
|
|
@@ -1769,7 +1937,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1769
1937
|
const sessionId = parsedSession.source === "header" ? parsedSession.sessionId : `chatcmpl-${parsedSession.sessionId}`;
|
|
1770
1938
|
const shellEnv = parseShellEnv(request);
|
|
1771
1939
|
const sessionSetupStart = performance.now();
|
|
1772
|
-
const [{ session, sessionManager: sm }
|
|
1940
|
+
const [{ session, sessionManager: sm }] = await Promise.all([ctx.createSession({
|
|
1773
1941
|
cwd: ctx.cwd,
|
|
1774
1942
|
sessionId,
|
|
1775
1943
|
perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
|
|
@@ -1856,6 +2024,8 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1856
2024
|
let usage = null;
|
|
1857
2025
|
let lastFollowUpCount = 0;
|
|
1858
2026
|
let capturedModelError = null;
|
|
2027
|
+
let sawModelErrorStop = false;
|
|
2028
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
1859
2029
|
const sendChunk = (delta, finishReason) => {
|
|
1860
2030
|
if (firstChunkAt === null) {
|
|
1861
2031
|
firstChunkAt = performance.now();
|
|
@@ -1880,14 +2050,22 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1880
2050
|
};
|
|
1881
2051
|
const ensureRole = () => {
|
|
1882
2052
|
if (roleEmitted) return;
|
|
1883
|
-
sendChunk({ role: "assistant" });
|
|
2053
|
+
sendChunk({ role: "assistant" }, null);
|
|
1884
2054
|
roleEmitted = true;
|
|
1885
2055
|
};
|
|
1886
2056
|
let _evCount = 0;
|
|
1887
2057
|
session.subscribe((event) => {
|
|
1888
2058
|
const ev = event;
|
|
2059
|
+
autoRetry.observe(ev);
|
|
2060
|
+
if (ev.type === "auto_retry_end" && ev.success === true) {
|
|
2061
|
+
capturedModelError = null;
|
|
2062
|
+
sawModelErrorStop = false;
|
|
2063
|
+
}
|
|
1889
2064
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
1890
|
-
if (endedMessage?.stopReason === "error" &&
|
|
2065
|
+
if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
|
|
2066
|
+
sawModelErrorStop = true;
|
|
2067
|
+
if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
|
|
2068
|
+
}
|
|
1891
2069
|
if (ev.type === "queue_update") {
|
|
1892
2070
|
const steering = Array.isArray(ev.steering) ? ev.steering.length : 0;
|
|
1893
2071
|
const followUp = Array.isArray(ev.followUp) ? ev.followUp.length : 0;
|
|
@@ -1903,7 +2081,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1903
2081
|
steering,
|
|
1904
2082
|
follow_up: followUp,
|
|
1905
2083
|
...consumedIds.length > 0 ? { consumed_steer_ids: consumedIds } : {}
|
|
1906
|
-
} });
|
|
2084
|
+
} }, null);
|
|
1907
2085
|
return;
|
|
1908
2086
|
}
|
|
1909
2087
|
_evCount++;
|
|
@@ -1926,7 +2104,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1926
2104
|
...ev.type === "message_start" || ev.type === "message_end" ? { messageJson: JSON.stringify(ev.message).slice(0, 500) } : {}
|
|
1927
2105
|
}, "harness session event");
|
|
1928
2106
|
if (ev.type === "tool_execution_start") {
|
|
1929
|
-
sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } });
|
|
2107
|
+
sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } }, null);
|
|
1930
2108
|
return;
|
|
1931
2109
|
}
|
|
1932
2110
|
if (ev.type === "tool_execution_end") {
|
|
@@ -1935,7 +2113,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1935
2113
|
tool_call_id: String(ev.toolCallId ?? ""),
|
|
1936
2114
|
content: text,
|
|
1937
2115
|
...ev.isError ? { is_error: true } : {}
|
|
1938
|
-
} });
|
|
2116
|
+
} }, null);
|
|
1939
2117
|
return;
|
|
1940
2118
|
}
|
|
1941
2119
|
if (ev.type === "agent_end") {
|
|
@@ -1947,11 +2125,11 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1947
2125
|
const t = inner?.type;
|
|
1948
2126
|
if (t === "text_delta") {
|
|
1949
2127
|
ensureRole();
|
|
1950
|
-
sendChunk({ content: inner.delta ?? "" });
|
|
2128
|
+
sendChunk({ content: inner.delta ?? "" }, null);
|
|
1951
2129
|
return;
|
|
1952
2130
|
}
|
|
1953
2131
|
if (t === "thinking_delta") {
|
|
1954
|
-
sendChunk({ x_thinking_delta: inner.delta ?? "" });
|
|
2132
|
+
sendChunk({ x_thinking_delta: inner.delta ?? "" }, null);
|
|
1955
2133
|
return;
|
|
1956
2134
|
}
|
|
1957
2135
|
if (t === "toolcall_start") {
|
|
@@ -1969,7 +2147,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1969
2147
|
name: tc?.name ?? "",
|
|
1970
2148
|
arguments: ""
|
|
1971
2149
|
}
|
|
1972
|
-
}] });
|
|
2150
|
+
}] }, null);
|
|
1973
2151
|
return;
|
|
1974
2152
|
}
|
|
1975
2153
|
if (t === "toolcall_delta") {
|
|
@@ -1978,7 +2156,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1978
2156
|
sendChunk({ tool_calls: [{
|
|
1979
2157
|
index: openaiIdx,
|
|
1980
2158
|
function: { arguments: inner.delta ?? "" }
|
|
1981
|
-
}] });
|
|
2159
|
+
}] }, null);
|
|
1982
2160
|
}
|
|
1983
2161
|
});
|
|
1984
2162
|
try {
|
|
@@ -2013,13 +2191,34 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
2013
2191
|
code: providerError.code,
|
|
2014
2192
|
upstream_status: providerError.upstreamStatus,
|
|
2015
2193
|
provider: providerError.provider,
|
|
2194
|
+
retry_attempts: autoRetry.attempts(),
|
|
2016
2195
|
source: "stop_reason_error"
|
|
2017
2196
|
}, "forwarding model-provider error to client");
|
|
2018
2197
|
emitModelProviderError({
|
|
2019
2198
|
writer,
|
|
2020
2199
|
sessionId,
|
|
2021
2200
|
created,
|
|
2022
|
-
providerError
|
|
2201
|
+
providerError,
|
|
2202
|
+
retryAttempts: autoRetry.attempts()
|
|
2203
|
+
});
|
|
2204
|
+
writer.write("data: [DONE]\n\n");
|
|
2205
|
+
} else if (sawModelErrorStop) {
|
|
2206
|
+
const providerError = toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message"));
|
|
2207
|
+
log.warn({
|
|
2208
|
+
event: "model_provider_error",
|
|
2209
|
+
chatcmpl_id: sessionId,
|
|
2210
|
+
code: providerError.code,
|
|
2211
|
+
upstream_status: providerError.upstreamStatus,
|
|
2212
|
+
provider: providerError.provider,
|
|
2213
|
+
retry_attempts: autoRetry.attempts(),
|
|
2214
|
+
source: "stop_reason_error_no_message"
|
|
2215
|
+
}, "forwarding model-provider error to client");
|
|
2216
|
+
emitModelProviderError({
|
|
2217
|
+
writer,
|
|
2218
|
+
sessionId,
|
|
2219
|
+
created,
|
|
2220
|
+
providerError,
|
|
2221
|
+
retryAttempts: autoRetry.attempts()
|
|
2023
2222
|
});
|
|
2024
2223
|
writer.write("data: [DONE]\n\n");
|
|
2025
2224
|
} else {
|
|
@@ -2058,13 +2257,15 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
2058
2257
|
chatcmpl_id: sessionId,
|
|
2059
2258
|
code: providerError.code,
|
|
2060
2259
|
upstream_status: providerError.upstreamStatus,
|
|
2061
|
-
provider: providerError.provider
|
|
2260
|
+
provider: providerError.provider,
|
|
2261
|
+
retry_attempts: autoRetry.attempts()
|
|
2062
2262
|
}, "forwarding model-provider error to client");
|
|
2063
2263
|
emitModelProviderError({
|
|
2064
2264
|
writer,
|
|
2065
2265
|
sessionId,
|
|
2066
2266
|
created,
|
|
2067
|
-
providerError
|
|
2267
|
+
providerError,
|
|
2268
|
+
retryAttempts: autoRetry.attempts()
|
|
2068
2269
|
});
|
|
2069
2270
|
} else sendChunk({ content: `\n[error: ${err?.message ?? err}]` }, "stop");
|
|
2070
2271
|
writer.write("data: [DONE]\n\n");
|
|
@@ -2081,7 +2282,8 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
2081
2282
|
});
|
|
2082
2283
|
}
|
|
2083
2284
|
function emitSessionMessagesTrailer({ writer, sm, baselineMessageCount, sessionId, created, usage }) {
|
|
2084
|
-
const
|
|
2285
|
+
const all = sm.buildSessionContext().messages;
|
|
2286
|
+
const sessionMessages = piMessagesToOpenAI(dropAbortedModelAttempts(all.slice(baselineMessageCount)));
|
|
2085
2287
|
if (sessionMessages.length === 0 && !usage) return;
|
|
2086
2288
|
const chunk = {
|
|
2087
2289
|
id: sessionId,
|
|
@@ -2105,7 +2307,7 @@ function emitSessionMessagesTrailer({ writer, sm, baselineMessageCount, sessionI
|
|
|
2105
2307
|
* cleanly closes the stream for any OpenAI-shaped reader that ignores the
|
|
2106
2308
|
* extension field.
|
|
2107
2309
|
*/
|
|
2108
|
-
function emitModelProviderError({ writer, sessionId, created, providerError }) {
|
|
2310
|
+
function emitModelProviderError({ writer, sessionId, created, providerError, retryAttempts }) {
|
|
2109
2311
|
const chunk = {
|
|
2110
2312
|
id: sessionId,
|
|
2111
2313
|
object: "chat.completion.chunk",
|
|
@@ -2120,7 +2322,8 @@ function emitModelProviderError({ writer, sessionId, created, providerError }) {
|
|
|
2120
2322
|
message: providerError.message,
|
|
2121
2323
|
provider: providerError.provider,
|
|
2122
2324
|
type: providerError.type,
|
|
2123
|
-
upstream_status: providerError.upstreamStatus
|
|
2325
|
+
upstream_status: providerError.upstreamStatus,
|
|
2326
|
+
retry_attempts: retryAttempts
|
|
2124
2327
|
}
|
|
2125
2328
|
};
|
|
2126
2329
|
writer.write(`data: ${JSON.stringify(chunk)}\n\n`);
|
|
@@ -2203,7 +2406,7 @@ async function runBlocking$1({ session, sm, prompt, images, sessionId, created,
|
|
|
2203
2406
|
};
|
|
2204
2407
|
if (toolCalls.length) message.tool_calls = toolCalls;
|
|
2205
2408
|
const all = sm.buildSessionContext().messages;
|
|
2206
|
-
const sessionMessages = piMessagesToOpenAI(all.slice(baselineMessageCount));
|
|
2409
|
+
const sessionMessages = piMessagesToOpenAI(dropAbortedModelAttempts(all.slice(baselineMessageCount)));
|
|
2207
2410
|
const response = {
|
|
2208
2411
|
id: sessionId,
|
|
2209
2412
|
object: "chat.completion",
|
|
@@ -2234,12 +2437,43 @@ const steerContentPartSchema = z.union([z.object({
|
|
|
2234
2437
|
type: z.literal("image_url"),
|
|
2235
2438
|
image_url: z.object({ url: z.string() })
|
|
2236
2439
|
})]);
|
|
2440
|
+
const steerSenderSchema = z.object({
|
|
2441
|
+
kind: z.enum([
|
|
2442
|
+
"user",
|
|
2443
|
+
"system",
|
|
2444
|
+
"agent",
|
|
2445
|
+
"external"
|
|
2446
|
+
]),
|
|
2447
|
+
displayName: z.string().nullable(),
|
|
2448
|
+
participantId: z.string().nullable(),
|
|
2449
|
+
attribute: z.boolean()
|
|
2450
|
+
});
|
|
2237
2451
|
const steerRequestSchema = z.object({
|
|
2238
2452
|
content: z.union([z.string().min(1), z.array(steerContentPartSchema).min(1)]).optional(),
|
|
2239
2453
|
mode: z.enum(["steer", "follow-up"]).default("steer"),
|
|
2454
|
+
sender: steerSenderSchema.optional(),
|
|
2240
2455
|
text: z.string().optional()
|
|
2241
2456
|
}).refine((v) => v.content !== void 0 || v.text !== void 0, { message: "content or text is required" });
|
|
2242
|
-
|
|
2457
|
+
/**
|
|
2458
|
+
* Render a steer as the user-role text the model reads.
|
|
2459
|
+
*
|
|
2460
|
+
* Typed senders let the harness own this frame instead of callers embedding a
|
|
2461
|
+
* prefix in the text itself — nothing downstream string-matches a marker.
|
|
2462
|
+
* Forms mirror the worker's history projection of the same rows
|
|
2463
|
+
* (messageToOai / to-oai-history) so the live steer and its re-read next turn
|
|
2464
|
+
* agree: system-authored rows project as `[platform] <text>`; attributed
|
|
2465
|
+
* humans as `Name: <text>`. Agent steers keep the explicit not-the-user agent
|
|
2466
|
+
* form rather than the human `Name:` attribution, per the impersonation fix
|
|
2467
|
+
* this carries forward. An attributed steer with no text (attachment-only)
|
|
2468
|
+
* still carries its attribution over `(attachments only)`.
|
|
2469
|
+
*/
|
|
2470
|
+
function frameSteerText({ text, sender }) {
|
|
2471
|
+
if (!sender?.attribute) return text;
|
|
2472
|
+
const body = text.length > 0 ? text : "(attachments only)";
|
|
2473
|
+
if (sender.kind === "system") return `[platform] ${body}`;
|
|
2474
|
+
if (sender.kind === "agent") return sender.displayName ? `[platform] steer from agent ${sender.displayName}: ${body}` : `[platform] steer from agent: ${body}`;
|
|
2475
|
+
return `${sender.displayName ?? "unnamed participant"}: ${body}`;
|
|
2476
|
+
}
|
|
2243
2477
|
async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
2244
2478
|
const { log } = requestLogger(request);
|
|
2245
2479
|
const session = registry.get(sessionId);
|
|
@@ -2247,17 +2481,21 @@ async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
|
2247
2481
|
let body;
|
|
2248
2482
|
try {
|
|
2249
2483
|
body = JSON.parse(await request.text());
|
|
2250
|
-
} catch {
|
|
2484
|
+
} catch (_err) {
|
|
2251
2485
|
return jsonError(400, "invalid json");
|
|
2252
2486
|
}
|
|
2253
2487
|
const parsed = steerRequestSchema.safeParse(body);
|
|
2254
2488
|
if (!parsed.success) return jsonError(400, parsed.error.message);
|
|
2255
2489
|
const { mode } = parsed.data;
|
|
2256
2490
|
const { text, images } = await extractContent(parsed.data.content ?? parsed.data.text ?? "", {
|
|
2257
|
-
userAgent: request.headers.get("user-agent") ??
|
|
2491
|
+
userAgent: request.headers.get("user-agent") ?? null,
|
|
2258
2492
|
log
|
|
2259
2493
|
});
|
|
2260
2494
|
if (!text && images.length === 0) return jsonError(400, "steer has no content");
|
|
2495
|
+
const framedText = frameSteerText({
|
|
2496
|
+
text,
|
|
2497
|
+
sender: parsed.data.sender ?? null
|
|
2498
|
+
});
|
|
2261
2499
|
const steerId = randomUUID();
|
|
2262
2500
|
let queues = pendingSteerIds.get(sessionId);
|
|
2263
2501
|
if (!queues) {
|
|
@@ -2269,10 +2507,10 @@ async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
|
2269
2507
|
}
|
|
2270
2508
|
if (mode === "follow-up") {
|
|
2271
2509
|
queues.followUp.push(steerId);
|
|
2272
|
-
await session.followUp(
|
|
2510
|
+
await session.followUp(framedText, images.length > 0 ? images : void 0);
|
|
2273
2511
|
} else {
|
|
2274
2512
|
queues.steer.push(steerId);
|
|
2275
|
-
await session.steer(
|
|
2513
|
+
await session.steer(framedText, images.length > 0 ? images : void 0);
|
|
2276
2514
|
}
|
|
2277
2515
|
log.info({
|
|
2278
2516
|
event: "steer_accepted",
|
|
@@ -2407,7 +2645,10 @@ async function extractImagesFromContent(content, log) {
|
|
|
2407
2645
|
for (const part of content) if (part?.type === "input_image") {
|
|
2408
2646
|
const url = part.image_url ?? part.url;
|
|
2409
2647
|
if (typeof url === "string" && /^https?:\/\//i.test(url)) try {
|
|
2410
|
-
images.push(await fetchImageAsBase64({
|
|
2648
|
+
images.push(await fetchImageAsBase64({
|
|
2649
|
+
url,
|
|
2650
|
+
userAgent: null
|
|
2651
|
+
}));
|
|
2411
2652
|
} catch (err) {
|
|
2412
2653
|
log.warn({
|
|
2413
2654
|
event: "image_fetch_failed",
|
|
@@ -2428,7 +2669,7 @@ async function handleResponses(request, ctx, registry) {
|
|
|
2428
2669
|
let parsed;
|
|
2429
2670
|
try {
|
|
2430
2671
|
parsed = JSON.parse(await request.text());
|
|
2431
|
-
} catch {
|
|
2672
|
+
} catch (_err) {
|
|
2432
2673
|
return jsonError(400, "invalid json");
|
|
2433
2674
|
}
|
|
2434
2675
|
const { input, stream = false, model: modelInput, instructions, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -2500,6 +2741,7 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2500
2741
|
const tcByContentIdx = /* @__PURE__ */ new Map();
|
|
2501
2742
|
let nextOutputIndex = 0;
|
|
2502
2743
|
let capturedModelError = null;
|
|
2744
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
2503
2745
|
const send = (event) => {
|
|
2504
2746
|
writer.write(`data: ${JSON.stringify(event)}\n\n`);
|
|
2505
2747
|
};
|
|
@@ -2508,7 +2750,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2508
2750
|
event: "model_provider_error",
|
|
2509
2751
|
code: providerError.code,
|
|
2510
2752
|
upstream_status: providerError.upstreamStatus,
|
|
2511
|
-
provider: providerError.provider
|
|
2753
|
+
provider: providerError.provider,
|
|
2754
|
+
retry_attempts: autoRetry.attempts()
|
|
2512
2755
|
}, "forwarding model-provider error to client");
|
|
2513
2756
|
send({
|
|
2514
2757
|
type: "response.failed",
|
|
@@ -2522,7 +2765,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2522
2765
|
x_model_provider_error: {
|
|
2523
2766
|
provider: providerError.provider,
|
|
2524
2767
|
type: providerError.type,
|
|
2525
|
-
upstream_status: providerError.upstreamStatus
|
|
2768
|
+
upstream_status: providerError.upstreamStatus,
|
|
2769
|
+
retry_attempts: autoRetry.attempts()
|
|
2526
2770
|
}
|
|
2527
2771
|
}
|
|
2528
2772
|
}
|
|
@@ -2558,8 +2802,10 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2558
2802
|
};
|
|
2559
2803
|
session.subscribe((event) => {
|
|
2560
2804
|
const ev = event;
|
|
2805
|
+
autoRetry.observe(ev);
|
|
2806
|
+
if (ev.type === "auto_retry_end" && ev.success === true) capturedModelError = null;
|
|
2561
2807
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
2562
|
-
if (endedMessage?.stopReason === "error"
|
|
2808
|
+
if (endedMessage?.stopReason === "error") capturedModelError = typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0 ? endedMessage.errorMessage : "model provider call failed without an error message";
|
|
2563
2809
|
if (ev.type !== "message_update") return;
|
|
2564
2810
|
const inner = ev.assistantMessageEvent;
|
|
2565
2811
|
if (!inner) return;
|