@skydiveai/pi-server 0.1.0 → 0.1.251-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.mts +307 -78
- package/dist/index.mjs +417 -59
- package/package.json +2 -9
package/dist/index.mjs
CHANGED
|
@@ -259,7 +259,7 @@ function parseShellEnv(request) {
|
|
|
259
259
|
const env = {};
|
|
260
260
|
for (const [k, v] of Object.entries(parsed)) if (typeof v === "string") env[k] = v;
|
|
261
261
|
return Object.keys(env).length > 0 ? env : null;
|
|
262
|
-
} catch {
|
|
262
|
+
} catch (_err) {
|
|
263
263
|
return null;
|
|
264
264
|
}
|
|
265
265
|
}
|
|
@@ -283,12 +283,12 @@ function createSSEResponse(handler) {
|
|
|
283
283
|
write(chunk) {
|
|
284
284
|
try {
|
|
285
285
|
controller.enqueue(encoder.encode(chunk));
|
|
286
|
-
} catch {}
|
|
286
|
+
} catch (_err) {}
|
|
287
287
|
},
|
|
288
288
|
close() {
|
|
289
289
|
try {
|
|
290
290
|
controller.close();
|
|
291
|
-
} catch {}
|
|
291
|
+
} catch (_err) {}
|
|
292
292
|
}
|
|
293
293
|
};
|
|
294
294
|
const keepalive = setInterval(() => {
|
|
@@ -298,13 +298,13 @@ function createSSEResponse(handler) {
|
|
|
298
298
|
clearInterval(keepalive);
|
|
299
299
|
try {
|
|
300
300
|
controller.close();
|
|
301
|
-
} catch {}
|
|
301
|
+
} catch (_err) {}
|
|
302
302
|
}, (err) => {
|
|
303
303
|
logger.error({ err }, "SSE handler crashed — stream closed without response");
|
|
304
304
|
clearInterval(keepalive);
|
|
305
305
|
try {
|
|
306
306
|
controller.close();
|
|
307
|
-
} catch {}
|
|
307
|
+
} catch (_err) {}
|
|
308
308
|
});
|
|
309
309
|
return new Response(stream, {
|
|
310
310
|
status: 200,
|
|
@@ -373,6 +373,57 @@ async function runConversation({ session, prompt, images, log, postPrompt }) {
|
|
|
373
373
|
});
|
|
374
374
|
}
|
|
375
375
|
/**
|
|
376
|
+
* Track the agent session's built-in model-call auto-retry so it leaves a
|
|
377
|
+
* trace.
|
|
378
|
+
*
|
|
379
|
+
* `AgentSession` already restarts a failed assistant turn in place (via
|
|
380
|
+
* `agent.continue()`, so no prompt is replayed and no tool re-executes) for the
|
|
381
|
+
* transient provider/transport failures pi classifies as retryable -- dropped
|
|
382
|
+
* streams, `terminated`, 5xx, overloaded, rate limits. It is on by default,
|
|
383
|
+
* with its own budget and backoff, and it emits `auto_retry_start` /
|
|
384
|
+
* `auto_retry_end` around each attempt.
|
|
385
|
+
*
|
|
386
|
+
* Nothing consumed those events, so a retry left no trace anywhere: a call that
|
|
387
|
+
* succeeded first try and one that burned the whole budget before failing
|
|
388
|
+
* produced the same terminal error, and the fleet-wide retry rate was
|
|
389
|
+
* unmeasurable. That gap is why a 2026-08-16 investigation into three runs lost
|
|
390
|
+
* to `provider_error: terminated` could not tell whether the budget had run out
|
|
391
|
+
* (ANY-7101).
|
|
392
|
+
*
|
|
393
|
+
* Exposed as a handler rather than its own `session.subscribe` call so each
|
|
394
|
+
* protocol feeds it from the single subscription it already owns -- one
|
|
395
|
+
* subscriber, explicit ordering.
|
|
396
|
+
*
|
|
397
|
+
* `attempts()` reports what has been spent so far, so a terminal error can
|
|
398
|
+
* carry the count to the worker, where it lands in a log group we can query
|
|
399
|
+
* fleet-wide (the sandbox's own logs are not).
|
|
400
|
+
*/
|
|
401
|
+
function createAutoRetryObserver(log) {
|
|
402
|
+
let attempts = 0;
|
|
403
|
+
return {
|
|
404
|
+
observe(event) {
|
|
405
|
+
if (event.type === "auto_retry_start") {
|
|
406
|
+
attempts = typeof event.attempt === "number" ? event.attempt : attempts + 1;
|
|
407
|
+
log.warn({
|
|
408
|
+
event: "model_call_auto_retry",
|
|
409
|
+
attempt: event.attempt,
|
|
410
|
+
max_attempts: event.maxAttempts,
|
|
411
|
+
delay_ms: event.delayMs,
|
|
412
|
+
error_message: event.errorMessage
|
|
413
|
+
}, "retrying failed model call in place");
|
|
414
|
+
return;
|
|
415
|
+
}
|
|
416
|
+
if (event.type === "auto_retry_end") log.warn({
|
|
417
|
+
event: "model_call_auto_retry_end",
|
|
418
|
+
attempt: event.attempt,
|
|
419
|
+
success: event.success,
|
|
420
|
+
final_error: event.finalError
|
|
421
|
+
}, event.success ? "model call recovered after retry" : "model call retries exhausted");
|
|
422
|
+
},
|
|
423
|
+
attempts: () => attempts
|
|
424
|
+
};
|
|
425
|
+
}
|
|
426
|
+
/**
|
|
376
427
|
* Hard-stop the in-flight turn for a session. Shared by every protocol's
|
|
377
428
|
* `/:id/abort` route: a cancel signals the stop explicitly instead of relying
|
|
378
429
|
* on a dropped connection. `session.abort()` interrupts the turn and resolves
|
|
@@ -419,7 +470,10 @@ async function extractParts(parts, log) {
|
|
|
419
470
|
}
|
|
420
471
|
if (part.url != null && part.mediaType?.startsWith("image/")) {
|
|
421
472
|
try {
|
|
422
|
-
images.push(await fetchImageAsBase64({
|
|
473
|
+
images.push(await fetchImageAsBase64({
|
|
474
|
+
url: part.url,
|
|
475
|
+
userAgent: null
|
|
476
|
+
}));
|
|
423
477
|
} catch (err) {
|
|
424
478
|
log.warn({
|
|
425
479
|
event: "a2a_image_fetch_failed",
|
|
@@ -687,6 +741,7 @@ const thinkingLevelMapSchema = z.object({
|
|
|
687
741
|
*/
|
|
688
742
|
const openaiCompletionsCompatSchema = z.object({
|
|
689
743
|
supportsReasoningEffort: z.boolean().optional(),
|
|
744
|
+
supportsStore: z.boolean().optional(),
|
|
690
745
|
requiresThinkingAsText: z.boolean().optional(),
|
|
691
746
|
thinkingFormat: z.enum([
|
|
692
747
|
"openai",
|
|
@@ -726,15 +781,26 @@ const baseSpecShape = {
|
|
|
726
781
|
*/
|
|
727
782
|
apiKey: z.string().min(1).optional()
|
|
728
783
|
};
|
|
729
|
-
const modelSpecSchema = z.discriminatedUnion("api", [
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
784
|
+
const modelSpecSchema = z.discriminatedUnion("api", [
|
|
785
|
+
z.object({
|
|
786
|
+
...baseSpecShape,
|
|
787
|
+
api: z.literal("openai-completions"),
|
|
788
|
+
compat: openaiCompletionsCompatSchema.optional()
|
|
789
|
+
}),
|
|
790
|
+
z.object({
|
|
791
|
+
...baseSpecShape,
|
|
792
|
+
api: z.literal("anthropic-messages"),
|
|
793
|
+
compat: anthropicMessagesCompatSchema.optional()
|
|
794
|
+
}),
|
|
795
|
+
z.object({
|
|
796
|
+
...baseSpecShape,
|
|
797
|
+
api: z.literal("openai-responses")
|
|
798
|
+
}),
|
|
799
|
+
z.object({
|
|
800
|
+
...baseSpecShape,
|
|
801
|
+
api: z.literal("openai-codex-responses")
|
|
802
|
+
}).omit({ apiKey: true })
|
|
803
|
+
]);
|
|
738
804
|
/**
|
|
739
805
|
* Parse a request-body `x_model`. Null when absent or invalid — never a
|
|
740
806
|
* request rejection, so a malformed spec degrades to the legacy body-model
|
|
@@ -781,6 +847,14 @@ function buildModelFromSpec(spec) {
|
|
|
781
847
|
api: spec.api,
|
|
782
848
|
...spec.compat ? { compat: spec.compat } : {}
|
|
783
849
|
};
|
|
850
|
+
case "openai-responses": return {
|
|
851
|
+
...base,
|
|
852
|
+
api: spec.api
|
|
853
|
+
};
|
|
854
|
+
case "openai-codex-responses": return {
|
|
855
|
+
...base,
|
|
856
|
+
api: spec.api
|
|
857
|
+
};
|
|
784
858
|
default: throw new Error(`Unhandled x_model api: ${String(spec)}`);
|
|
785
859
|
}
|
|
786
860
|
}
|
|
@@ -802,7 +876,7 @@ function resolveRequestModel({ modelInput, modelSpecInput, defaultModel, log })
|
|
|
802
876
|
* always wins.
|
|
803
877
|
*/
|
|
804
878
|
function withSpecApiKey(keys, modelSpec) {
|
|
805
|
-
if (!modelSpec
|
|
879
|
+
if (!modelSpec || !("apiKey" in modelSpec) || !modelSpec.apiKey || keys[modelSpec.provider]) return keys;
|
|
806
880
|
return {
|
|
807
881
|
...keys,
|
|
808
882
|
[modelSpec.provider]: modelSpec.apiKey
|
|
@@ -865,6 +939,18 @@ function providerId(err, message) {
|
|
|
865
939
|
*/
|
|
866
940
|
const CONTEXT_OVERFLOW_MESSAGE = /prompt is too long|context[_ ]length[_ ]exceeded|maximum context length|exceeds maximum token limit|input token count|too many tokens|exceeds the maximum/i;
|
|
867
941
|
/**
|
|
942
|
+
* A content-filter / model-safety termination message from any provider.
|
|
943
|
+
* Used by BOTH the classifyCode message fallback and isModelProviderError's
|
|
944
|
+
* gate (a shape recognized there must pass this gate or it never gets
|
|
945
|
+
* classified), so it lives in one place rather than two copies kept "in
|
|
946
|
+
* sync" by comment. Verified shapes (see classifyCode):
|
|
947
|
+
* - "Provider finish_reason: content_filter" (Anthropic/gateway 200 stream)
|
|
948
|
+
* - "This request triggered restrictions on violative cyber content ..."
|
|
949
|
+
* - "This request triggered cyber-related safeguards. ..."
|
|
950
|
+
* - "This content was flagged for possible cybersecurity risk. ..."
|
|
951
|
+
*/
|
|
952
|
+
const CONTENT_FILTER_MESSAGE = /finish_reason:\s*content_filter|content[_ ]filter|content[_ ]management[_ ]policy|violative cyber content|cyber[- ]related safeguards|cybersecurity risk|cyber verification program/i;
|
|
953
|
+
/**
|
|
868
954
|
* Map a provider error to a stable code. Prefers the provider's machine code
|
|
869
955
|
* (OpenAI `error.code`, OpenRouter `error.metadata.error_type`) over message
|
|
870
956
|
* text, then falls back to status + message regex for providers that don't
|
|
@@ -877,6 +963,7 @@ function classifyCode(status, message, code) {
|
|
|
877
963
|
if (status === 429 || code === "rate_limit_exceeded" || /rate[_ ]limit|too many requests|resource[_ ]exhausted/i.test(message)) return "rate_limited";
|
|
878
964
|
if (status === 529 || code === "provider_overloaded" || /overloaded/i.test(message)) return "provider_overloaded";
|
|
879
965
|
if (status === 503 || status === 504 || code === "provider_unavailable" || /no healthy upstream|upstream request timeout|stream timeout|service unavailable|unavailable|gateway/i.test(message)) return "provider_unavailable";
|
|
966
|
+
if (code === "content_filter" || CONTENT_FILTER_MESSAGE.test(message)) return "content_filter";
|
|
880
967
|
return "provider_error";
|
|
881
968
|
}
|
|
882
969
|
/**
|
|
@@ -891,7 +978,7 @@ function isModelProviderError(err) {
|
|
|
891
978
|
if (numericStatus(e) !== null) return true;
|
|
892
979
|
if (e.error && typeof e.error === "object") return true;
|
|
893
980
|
const msg = messageText(err);
|
|
894
|
-
return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits/i.test(msg);
|
|
981
|
+
return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits/i.test(msg) || CONTENT_FILTER_MESSAGE.test(msg);
|
|
895
982
|
}
|
|
896
983
|
/** Build the structured, forwardable error from a thrown model-provider error. */
|
|
897
984
|
function toModelProviderError(err) {
|
|
@@ -947,7 +1034,10 @@ async function extractContent$1(content, log) {
|
|
|
947
1034
|
data: src.data
|
|
948
1035
|
});
|
|
949
1036
|
else if (src?.type === "url" && src.url) try {
|
|
950
|
-
images.push(await fetchImageAsBase64({
|
|
1037
|
+
images.push(await fetchImageAsBase64({
|
|
1038
|
+
url: src.url,
|
|
1039
|
+
userAgent: null
|
|
1040
|
+
}));
|
|
951
1041
|
} catch (err) {
|
|
952
1042
|
log.warn({
|
|
953
1043
|
event: "image_fetch_failed",
|
|
@@ -1036,7 +1126,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1036
1126
|
let parsed;
|
|
1037
1127
|
try {
|
|
1038
1128
|
parsed = JSON.parse(await request.text());
|
|
1039
|
-
} catch {
|
|
1129
|
+
} catch (_err) {
|
|
1040
1130
|
return jsonError(400, "invalid json");
|
|
1041
1131
|
}
|
|
1042
1132
|
const { messages, stream = false, model: modelInput, system, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -1059,7 +1149,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1059
1149
|
const baseSystemPrompt = extractSystemPrompt(system);
|
|
1060
1150
|
const { sessionId } = parseSessionId(request);
|
|
1061
1151
|
const shellEnv = parseShellEnv(request);
|
|
1062
|
-
const { session
|
|
1152
|
+
const { session } = await ctx.createSession({
|
|
1063
1153
|
cwd: ctx.cwd,
|
|
1064
1154
|
sessionId,
|
|
1065
1155
|
perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
|
|
@@ -1082,7 +1172,6 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1082
1172
|
if (stream) {
|
|
1083
1173
|
const response = runStream$2({
|
|
1084
1174
|
session,
|
|
1085
|
-
sm,
|
|
1086
1175
|
prompt,
|
|
1087
1176
|
images,
|
|
1088
1177
|
id,
|
|
@@ -1098,7 +1187,6 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1098
1187
|
}
|
|
1099
1188
|
const response = await runBlocking$2({
|
|
1100
1189
|
session,
|
|
1101
|
-
sm,
|
|
1102
1190
|
prompt,
|
|
1103
1191
|
images,
|
|
1104
1192
|
id,
|
|
@@ -1112,7 +1200,7 @@ async function handleMessages(request, ctx, registry) {
|
|
|
1112
1200
|
if (traceId) response.headers.set("x-trace-id", traceId);
|
|
1113
1201
|
return response;
|
|
1114
1202
|
}
|
|
1115
|
-
function runStream$2({ session,
|
|
1203
|
+
function runStream$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
|
|
1116
1204
|
return createSSEResponse(async (writer) => {
|
|
1117
1205
|
let contentBlockIndex = 0;
|
|
1118
1206
|
let textBlockOpen = false;
|
|
@@ -1158,12 +1246,15 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1158
1246
|
textBlockOpen = false;
|
|
1159
1247
|
};
|
|
1160
1248
|
let capturedModelError = null;
|
|
1249
|
+
let sawModelErrorStop = false;
|
|
1250
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
1161
1251
|
const emitProviderError = (providerError) => {
|
|
1162
1252
|
log.warn({
|
|
1163
1253
|
event: "model_provider_error",
|
|
1164
1254
|
code: providerError.code,
|
|
1165
1255
|
upstream_status: providerError.upstreamStatus,
|
|
1166
|
-
provider: providerError.provider
|
|
1256
|
+
provider: providerError.provider,
|
|
1257
|
+
retry_attempts: autoRetry.attempts()
|
|
1167
1258
|
}, "forwarding model-provider error to client");
|
|
1168
1259
|
sseEvent(writer, "error", {
|
|
1169
1260
|
type: "error",
|
|
@@ -1173,15 +1264,24 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1173
1264
|
x_model_provider_error: {
|
|
1174
1265
|
code: providerError.code,
|
|
1175
1266
|
provider: providerError.provider,
|
|
1176
|
-
upstream_status: providerError.upstreamStatus
|
|
1267
|
+
upstream_status: providerError.upstreamStatus,
|
|
1268
|
+
retry_attempts: autoRetry.attempts()
|
|
1177
1269
|
}
|
|
1178
1270
|
}
|
|
1179
1271
|
});
|
|
1180
1272
|
};
|
|
1181
1273
|
session.subscribe((event) => {
|
|
1182
1274
|
const ev = event;
|
|
1275
|
+
autoRetry.observe(ev);
|
|
1276
|
+
if (ev.type === "auto_retry_end" && ev.success === true) {
|
|
1277
|
+
capturedModelError = null;
|
|
1278
|
+
sawModelErrorStop = false;
|
|
1279
|
+
}
|
|
1183
1280
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
1184
|
-
if (endedMessage?.stopReason === "error" &&
|
|
1281
|
+
if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
|
|
1282
|
+
sawModelErrorStop = true;
|
|
1283
|
+
if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
|
|
1284
|
+
}
|
|
1185
1285
|
if (ev.type !== "message_update") return;
|
|
1186
1286
|
const inner = ev.assistantMessageEvent;
|
|
1187
1287
|
if (!inner) return;
|
|
@@ -1239,6 +1339,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1239
1339
|
postPrompt
|
|
1240
1340
|
});
|
|
1241
1341
|
if (capturedModelError !== null) emitProviderError(toModelProviderError(new Error(capturedModelError)));
|
|
1342
|
+
else if (sawModelErrorStop) emitProviderError(toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message")));
|
|
1242
1343
|
} catch (err) {
|
|
1243
1344
|
log.error({
|
|
1244
1345
|
err,
|
|
@@ -1266,7 +1367,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
|
|
|
1266
1367
|
sseEvent(writer, "message_stop", { type: "message_stop" });
|
|
1267
1368
|
});
|
|
1268
1369
|
}
|
|
1269
|
-
async function runBlocking$2({ session,
|
|
1370
|
+
async function runBlocking$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
|
|
1270
1371
|
let text = "";
|
|
1271
1372
|
const toolUseBlocks = [];
|
|
1272
1373
|
const tcByContentIdx = /* @__PURE__ */ new Map();
|
|
@@ -1317,7 +1418,7 @@ async function runBlocking$2({ session, sm, prompt, images, id, sessionId, model
|
|
|
1317
1418
|
}
|
|
1318
1419
|
for (const block of toolUseBlocks) if (block.type === "tool_use" && typeof block.input === "string") try {
|
|
1319
1420
|
block.input = JSON.parse(block.input);
|
|
1320
|
-
} catch {
|
|
1421
|
+
} catch (_err) {
|
|
1321
1422
|
block.input = {};
|
|
1322
1423
|
}
|
|
1323
1424
|
const content = [];
|
|
@@ -1379,6 +1480,110 @@ function create$2(options) {
|
|
|
1379
1480
|
};
|
|
1380
1481
|
}
|
|
1381
1482
|
//#endregion
|
|
1483
|
+
//#region src/billing-blocked-provider-signal.ts
|
|
1484
|
+
const BILLING_BLOCKED_PROVIDER_SIGNAL_PREFIX = "ANYONE_BILLING_BLOCKED_V1:";
|
|
1485
|
+
/**
|
|
1486
|
+
* Duplicated from the billing gate's `BlockReason` (apps/anyone/billing
|
|
1487
|
+
* gate/block-reason.ts): pi-server runs inside the sandbox and must not
|
|
1488
|
+
* depend on platform packages. A reason the proxy sends that predates this
|
|
1489
|
+
* build fails the enum and degrades to the untyped `billing_blocked`
|
|
1490
|
+
* handling — never a crash. Exported so agent-lifecycle's chat.test.ts can
|
|
1491
|
+
* pin this copy against its own (which billing's block-reason.test.ts in
|
|
1492
|
+
* turn pins against the gate), keeping every copy drift-checked.
|
|
1493
|
+
*/
|
|
1494
|
+
const BILLING_BLOCK_REASONS = [
|
|
1495
|
+
"insufficient_balance",
|
|
1496
|
+
"payment_failed",
|
|
1497
|
+
"hard_spend_limit_reached"
|
|
1498
|
+
];
|
|
1499
|
+
const payloadSchema = z.object({
|
|
1500
|
+
blockReason: z.enum(BILLING_BLOCK_REASONS),
|
|
1501
|
+
message: z.string()
|
|
1502
|
+
}).strict();
|
|
1503
|
+
const nestedProxyEnvelopeSchema = z.object({
|
|
1504
|
+
error: z.object({
|
|
1505
|
+
type: z.literal("billing_blocked"),
|
|
1506
|
+
code: z.literal("billing_blocked"),
|
|
1507
|
+
message: z.string()
|
|
1508
|
+
}).strict(),
|
|
1509
|
+
blockReason: z.enum(BILLING_BLOCK_REASONS),
|
|
1510
|
+
message: z.string()
|
|
1511
|
+
}).strict();
|
|
1512
|
+
/**
|
|
1513
|
+
* pi-ai's openai-compatible providers (openrouter, xai, groq, deepseek, …) can
|
|
1514
|
+
* NOT fold a proxy's non-2xx body into `error.message`, so `formatProviderError`
|
|
1515
|
+
* (@earendil-works/pi-ai utils/error-body) composes the display string as
|
|
1516
|
+
* `"<status>: <body>"` or, with a provider label, `"<prefix> (<status>): <body>"`.
|
|
1517
|
+
* The billing gate's 402 body therefore reaches us wrapped, e.g.
|
|
1518
|
+
* `"402: {\"error\":{...},\"blockReason\":\"insufficient_balance\",\"message\":...}"`
|
|
1519
|
+
* or `"OpenRouter (402): {...}"`. Neither the bare `JSON.parse` nor the `"402 "`
|
|
1520
|
+
* (space) strip below recognizes that, so a real billing block from an
|
|
1521
|
+
* openrouter-routed model degrades to the generic model-provider error. Peel a
|
|
1522
|
+
* single leading `"<status>: "` / `"<prefix> (<status>): "` wrapper off the
|
|
1523
|
+
* front so the recovered body flows through the existing shape checks. Returns
|
|
1524
|
+
* the message unchanged when no wrapper is present.
|
|
1525
|
+
*/
|
|
1526
|
+
function unwrapOpenAICompatStatusPrefix(message) {
|
|
1527
|
+
const withPrefix = message.match(/^.+ \(\d{3}\): ([\s\S]+)$/);
|
|
1528
|
+
if (withPrefix?.[1] !== void 0) return withPrefix[1];
|
|
1529
|
+
const bare = message.match(/^\d{3}: ([\s\S]+)$/);
|
|
1530
|
+
if (bare?.[1] !== void 0) return bare[1];
|
|
1531
|
+
return message;
|
|
1532
|
+
}
|
|
1533
|
+
function parsePrefixedSignal(message) {
|
|
1534
|
+
const normalized = message.startsWith("402 ") ? message.slice(4) : message;
|
|
1535
|
+
if (!normalized.startsWith(BILLING_BLOCKED_PROVIDER_SIGNAL_PREFIX)) return null;
|
|
1536
|
+
try {
|
|
1537
|
+
const parsed = payloadSchema.safeParse(JSON.parse(normalized.slice(26)));
|
|
1538
|
+
return parsed.success ? parsed.data : null;
|
|
1539
|
+
} catch (_err) {
|
|
1540
|
+
return null;
|
|
1541
|
+
}
|
|
1542
|
+
}
|
|
1543
|
+
/**
|
|
1544
|
+
* The sandbox proxy's other 402 envelope (verified prod 2026-09-20): the
|
|
1545
|
+
* billing gate's body surfaces with the type/code at TOP level and the
|
|
1546
|
+
* versioned signal nested inside `message` —
|
|
1547
|
+
* {"type":"billing_blocked","code":"billing_blocked","message":
|
|
1548
|
+
* "ANYONE_BILLING_BLOCKED_V1:{\"blockReason\":\"insufficient_balance\",…}"}
|
|
1549
|
+
* — with no blockReason/error wrapper for nestedProxyEnvelopeSchema, so it
|
|
1550
|
+
* used to fall through and degrade to a generic model-provider error that
|
|
1551
|
+
* read as "The model stopped responding" on an out-of-credit org. The outer
|
|
1552
|
+
* shape is validated with a schema instead of ad-hoc casting, and only the
|
|
1553
|
+
* top-level `message` is peeled; a signal nested inside `error.message` is
|
|
1554
|
+
* handled below through nestedProxyEnvelopeSchema's full integrity checks
|
|
1555
|
+
* (validating type, code, blockReason, and message agreement).
|
|
1556
|
+
*/
|
|
1557
|
+
const topLevelMessageEnvelopeSchema = z.object({ message: z.string().optional() }).passthrough();
|
|
1558
|
+
/**
|
|
1559
|
+
* Decode only the versioned platform envelope. Anthropic and OpenAI prefix the
|
|
1560
|
+
* nested error message with HTTP 402. Google's SDK instead preserves the full
|
|
1561
|
+
* proxy response as JSON, so that outer shape is validated separately.
|
|
1562
|
+
*/
|
|
1563
|
+
function parseBillingBlockedProviderSignal(message) {
|
|
1564
|
+
const unwrapped = unwrapOpenAICompatStatusPrefix(message);
|
|
1565
|
+
const direct = parsePrefixedSignal(unwrapped);
|
|
1566
|
+
if (direct) return direct;
|
|
1567
|
+
let body;
|
|
1568
|
+
try {
|
|
1569
|
+
body = JSON.parse(unwrapped);
|
|
1570
|
+
} catch (_err) {
|
|
1571
|
+
return null;
|
|
1572
|
+
}
|
|
1573
|
+
if (body === null || typeof body !== "object") return null;
|
|
1574
|
+
const nested = topLevelMessageEnvelopeSchema.safeParse(body);
|
|
1575
|
+
if (nested.success && nested.data.message) {
|
|
1576
|
+
const signal = parsePrefixedSignal(nested.data.message);
|
|
1577
|
+
if (signal) return signal;
|
|
1578
|
+
}
|
|
1579
|
+
const envelope = nestedProxyEnvelopeSchema.safeParse(body);
|
|
1580
|
+
if (!envelope.success) return null;
|
|
1581
|
+
const nestedError = parsePrefixedSignal(envelope.data.error.message);
|
|
1582
|
+
const topLevel = parsePrefixedSignal(envelope.data.message);
|
|
1583
|
+
if (!nestedError || nestedError.blockReason !== envelope.data.blockReason || envelope.data.message !== nestedError.message && (!topLevel || topLevel.blockReason !== nestedError.blockReason || topLevel.message !== nestedError.message)) return null;
|
|
1584
|
+
return nestedError;
|
|
1585
|
+
}
|
|
1586
|
+
//#endregion
|
|
1382
1587
|
//#region src/protocols/chat-completions.ts
|
|
1383
1588
|
/**
|
|
1384
1589
|
* OpenAI Chat Completions–compatible protocol handler.
|
|
@@ -1563,7 +1768,7 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
|
|
|
1563
1768
|
let args = {};
|
|
1564
1769
|
try {
|
|
1565
1770
|
args = tc.function.arguments ? JSON.parse(tc.function.arguments) : {};
|
|
1566
|
-
} catch {
|
|
1771
|
+
} catch (_err) {
|
|
1567
1772
|
args = { _raw: tc.function.arguments };
|
|
1568
1773
|
}
|
|
1569
1774
|
contentArr.push({
|
|
@@ -1605,6 +1810,33 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
|
|
|
1605
1810
|
}
|
|
1606
1811
|
}
|
|
1607
1812
|
}
|
|
1813
|
+
/**
|
|
1814
|
+
* Each pi/session message carries the real wall-clock `timestamp` the
|
|
1815
|
+
* provider stamped when it produced that message (see @earendil-works/pi-ai
|
|
1816
|
+
* Message). Forward it verbatim as `x_created_at` (epoch ms) on the OAI
|
|
1817
|
+
* trailer message so the consumer (anyone-messaging completeRun) can stamp
|
|
1818
|
+
* each persisted segment at its true emit time and interleave a mid-run steer
|
|
1819
|
+
* by createdAt — instead of positionally reconstructing per-turn times, which
|
|
1820
|
+
* drifts on parallel tool calls and provider-split replies.
|
|
1821
|
+
*/
|
|
1822
|
+
function messageTimestampMs(m) {
|
|
1823
|
+
const ts = m?.timestamp;
|
|
1824
|
+
return typeof ts === "number" && Number.isFinite(ts) ? ts : null;
|
|
1825
|
+
}
|
|
1826
|
+
/**
|
|
1827
|
+
* Drop assistant messages recorded as failed model calls (stopReason 'error')
|
|
1828
|
+
* from the session-trailer slice. pi keeps an aborted attempt in the session
|
|
1829
|
+
* when auto-retry re-runs the model call in place, so a recovered turn carries
|
|
1830
|
+
* both the errored attempt and the successful retry. The trailer is only built
|
|
1831
|
+
* on the clean path -- a turn that exhausts its retries fails and never emits
|
|
1832
|
+
* one -- so any stopReason 'error' assistant message in the slice is an
|
|
1833
|
+
* aborted attempt the retry replaced, not a message downstream should join.
|
|
1834
|
+
* A user abort ('aborted') is a cancel, not a retried provider failure, and is
|
|
1835
|
+
* kept, matching the terminal-error capture above.
|
|
1836
|
+
*/
|
|
1837
|
+
function dropAbortedModelAttempts(messages) {
|
|
1838
|
+
return messages.filter((m) => !(m?.role === "assistant" && m.stopReason === "error" && m.errorMessage !== "aborted"));
|
|
1839
|
+
}
|
|
1608
1840
|
function piMessagesToOpenAI(messages) {
|
|
1609
1841
|
const out = [];
|
|
1610
1842
|
for (const m of messages) {
|
|
@@ -1633,6 +1865,8 @@ function piMessagesToOpenAI(messages) {
|
|
|
1633
1865
|
};
|
|
1634
1866
|
if (toolCalls.length) msg.tool_calls = toolCalls;
|
|
1635
1867
|
if (thinking.length) msg.x_thinking = thinking;
|
|
1868
|
+
const createdAtMs = messageTimestampMs(m);
|
|
1869
|
+
if (createdAtMs !== null) msg.x_created_at = createdAtMs;
|
|
1636
1870
|
out.push(msg);
|
|
1637
1871
|
continue;
|
|
1638
1872
|
}
|
|
@@ -1644,6 +1878,8 @@ function piMessagesToOpenAI(messages) {
|
|
|
1644
1878
|
content: text
|
|
1645
1879
|
};
|
|
1646
1880
|
if (m.isError) tm.x_is_error = true;
|
|
1881
|
+
const createdAtMs = messageTimestampMs(m);
|
|
1882
|
+
if (createdAtMs !== null) tm.x_created_at = createdAtMs;
|
|
1647
1883
|
out.push(tm);
|
|
1648
1884
|
continue;
|
|
1649
1885
|
}
|
|
@@ -1658,7 +1894,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1658
1894
|
let parsed;
|
|
1659
1895
|
try {
|
|
1660
1896
|
parsed = JSON.parse(await request.text());
|
|
1661
|
-
} catch {
|
|
1897
|
+
} catch (_err) {
|
|
1662
1898
|
return jsonError(400, "invalid json");
|
|
1663
1899
|
}
|
|
1664
1900
|
const { messages, stream = false, model: modelInput, thinkingLevel: thinkingInput, context_window: contextWindowInput, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -1670,7 +1906,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1670
1906
|
}
|
|
1671
1907
|
if (lastUserIdx < 0) return jsonError(400, "no user message in history");
|
|
1672
1908
|
const lastUser = messages[lastUserIdx];
|
|
1673
|
-
const reqUA = request.headers.get("user-agent") ??
|
|
1909
|
+
const reqUA = request.headers.get("user-agent") ?? null;
|
|
1674
1910
|
const { text: prompt, images } = await extractContent(lastUser.content, {
|
|
1675
1911
|
userAgent: reqUA,
|
|
1676
1912
|
log
|
|
@@ -1701,7 +1937,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
|
|
|
1701
1937
|
const sessionId = parsedSession.source === "header" ? parsedSession.sessionId : `chatcmpl-${parsedSession.sessionId}`;
|
|
1702
1938
|
const shellEnv = parseShellEnv(request);
|
|
1703
1939
|
const sessionSetupStart = performance.now();
|
|
1704
|
-
const [{ session, sessionManager: sm }
|
|
1940
|
+
const [{ session, sessionManager: sm }] = await Promise.all([ctx.createSession({
|
|
1705
1941
|
cwd: ctx.cwd,
|
|
1706
1942
|
sessionId,
|
|
1707
1943
|
perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
|
|
@@ -1788,6 +2024,8 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1788
2024
|
let usage = null;
|
|
1789
2025
|
let lastFollowUpCount = 0;
|
|
1790
2026
|
let capturedModelError = null;
|
|
2027
|
+
let sawModelErrorStop = false;
|
|
2028
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
1791
2029
|
const sendChunk = (delta, finishReason) => {
|
|
1792
2030
|
if (firstChunkAt === null) {
|
|
1793
2031
|
firstChunkAt = performance.now();
|
|
@@ -1812,14 +2050,22 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1812
2050
|
};
|
|
1813
2051
|
const ensureRole = () => {
|
|
1814
2052
|
if (roleEmitted) return;
|
|
1815
|
-
sendChunk({ role: "assistant" });
|
|
2053
|
+
sendChunk({ role: "assistant" }, null);
|
|
1816
2054
|
roleEmitted = true;
|
|
1817
2055
|
};
|
|
1818
2056
|
let _evCount = 0;
|
|
1819
2057
|
session.subscribe((event) => {
|
|
1820
2058
|
const ev = event;
|
|
2059
|
+
autoRetry.observe(ev);
|
|
2060
|
+
if (ev.type === "auto_retry_end" && ev.success === true) {
|
|
2061
|
+
capturedModelError = null;
|
|
2062
|
+
sawModelErrorStop = false;
|
|
2063
|
+
}
|
|
1821
2064
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
1822
|
-
if (endedMessage?.stopReason === "error" &&
|
|
2065
|
+
if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
|
|
2066
|
+
sawModelErrorStop = true;
|
|
2067
|
+
if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
|
|
2068
|
+
}
|
|
1823
2069
|
if (ev.type === "queue_update") {
|
|
1824
2070
|
const steering = Array.isArray(ev.steering) ? ev.steering.length : 0;
|
|
1825
2071
|
const followUp = Array.isArray(ev.followUp) ? ev.followUp.length : 0;
|
|
@@ -1835,7 +2081,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1835
2081
|
steering,
|
|
1836
2082
|
follow_up: followUp,
|
|
1837
2083
|
...consumedIds.length > 0 ? { consumed_steer_ids: consumedIds } : {}
|
|
1838
|
-
} });
|
|
2084
|
+
} }, null);
|
|
1839
2085
|
return;
|
|
1840
2086
|
}
|
|
1841
2087
|
_evCount++;
|
|
@@ -1858,7 +2104,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1858
2104
|
...ev.type === "message_start" || ev.type === "message_end" ? { messageJson: JSON.stringify(ev.message).slice(0, 500) } : {}
|
|
1859
2105
|
}, "harness session event");
|
|
1860
2106
|
if (ev.type === "tool_execution_start") {
|
|
1861
|
-
sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } });
|
|
2107
|
+
sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } }, null);
|
|
1862
2108
|
return;
|
|
1863
2109
|
}
|
|
1864
2110
|
if (ev.type === "tool_execution_end") {
|
|
@@ -1867,7 +2113,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1867
2113
|
tool_call_id: String(ev.toolCallId ?? ""),
|
|
1868
2114
|
content: text,
|
|
1869
2115
|
...ev.isError ? { is_error: true } : {}
|
|
1870
|
-
} });
|
|
2116
|
+
} }, null);
|
|
1871
2117
|
return;
|
|
1872
2118
|
}
|
|
1873
2119
|
if (ev.type === "agent_end") {
|
|
@@ -1879,11 +2125,11 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1879
2125
|
const t = inner?.type;
|
|
1880
2126
|
if (t === "text_delta") {
|
|
1881
2127
|
ensureRole();
|
|
1882
|
-
sendChunk({ content: inner.delta ?? "" });
|
|
2128
|
+
sendChunk({ content: inner.delta ?? "" }, null);
|
|
1883
2129
|
return;
|
|
1884
2130
|
}
|
|
1885
2131
|
if (t === "thinking_delta") {
|
|
1886
|
-
sendChunk({ x_thinking_delta: inner.delta ?? "" });
|
|
2132
|
+
sendChunk({ x_thinking_delta: inner.delta ?? "" }, null);
|
|
1887
2133
|
return;
|
|
1888
2134
|
}
|
|
1889
2135
|
if (t === "toolcall_start") {
|
|
@@ -1901,7 +2147,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1901
2147
|
name: tc?.name ?? "",
|
|
1902
2148
|
arguments: ""
|
|
1903
2149
|
}
|
|
1904
|
-
}] });
|
|
2150
|
+
}] }, null);
|
|
1905
2151
|
return;
|
|
1906
2152
|
}
|
|
1907
2153
|
if (t === "toolcall_delta") {
|
|
@@ -1910,7 +2156,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1910
2156
|
sendChunk({ tool_calls: [{
|
|
1911
2157
|
index: openaiIdx,
|
|
1912
2158
|
function: { arguments: inner.delta ?? "" }
|
|
1913
|
-
}] });
|
|
2159
|
+
}] }, null);
|
|
1914
2160
|
}
|
|
1915
2161
|
});
|
|
1916
2162
|
try {
|
|
@@ -1922,6 +2168,22 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1922
2168
|
postPrompt
|
|
1923
2169
|
});
|
|
1924
2170
|
if (capturedModelError !== null) {
|
|
2171
|
+
const billingBlocked = parseBillingBlockedProviderSignal(capturedModelError);
|
|
2172
|
+
if (billingBlocked) {
|
|
2173
|
+
log.warn({
|
|
2174
|
+
event: "billing_blocked",
|
|
2175
|
+
chatcmpl_id: sessionId,
|
|
2176
|
+
source: "stop_reason_error"
|
|
2177
|
+
}, "forwarding platform billing block to client");
|
|
2178
|
+
emitBillingBlocked({
|
|
2179
|
+
writer,
|
|
2180
|
+
sessionId,
|
|
2181
|
+
created,
|
|
2182
|
+
billingBlocked
|
|
2183
|
+
});
|
|
2184
|
+
writer.write("data: [DONE]\n\n");
|
|
2185
|
+
return;
|
|
2186
|
+
}
|
|
1925
2187
|
const providerError = toModelProviderError(new Error(capturedModelError));
|
|
1926
2188
|
log.warn({
|
|
1927
2189
|
event: "model_provider_error",
|
|
@@ -1929,13 +2191,34 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1929
2191
|
code: providerError.code,
|
|
1930
2192
|
upstream_status: providerError.upstreamStatus,
|
|
1931
2193
|
provider: providerError.provider,
|
|
2194
|
+
retry_attempts: autoRetry.attempts(),
|
|
1932
2195
|
source: "stop_reason_error"
|
|
1933
2196
|
}, "forwarding model-provider error to client");
|
|
1934
2197
|
emitModelProviderError({
|
|
1935
2198
|
writer,
|
|
1936
2199
|
sessionId,
|
|
1937
2200
|
created,
|
|
1938
|
-
providerError
|
|
2201
|
+
providerError,
|
|
2202
|
+
retryAttempts: autoRetry.attempts()
|
|
2203
|
+
});
|
|
2204
|
+
writer.write("data: [DONE]\n\n");
|
|
2205
|
+
} else if (sawModelErrorStop) {
|
|
2206
|
+
const providerError = toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message"));
|
|
2207
|
+
log.warn({
|
|
2208
|
+
event: "model_provider_error",
|
|
2209
|
+
chatcmpl_id: sessionId,
|
|
2210
|
+
code: providerError.code,
|
|
2211
|
+
upstream_status: providerError.upstreamStatus,
|
|
2212
|
+
provider: providerError.provider,
|
|
2213
|
+
retry_attempts: autoRetry.attempts(),
|
|
2214
|
+
source: "stop_reason_error_no_message"
|
|
2215
|
+
}, "forwarding model-provider error to client");
|
|
2216
|
+
emitModelProviderError({
|
|
2217
|
+
writer,
|
|
2218
|
+
sessionId,
|
|
2219
|
+
created,
|
|
2220
|
+
providerError,
|
|
2221
|
+
retryAttempts: autoRetry.attempts()
|
|
1939
2222
|
});
|
|
1940
2223
|
writer.write("data: [DONE]\n\n");
|
|
1941
2224
|
} else {
|
|
@@ -1957,19 +2240,32 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1957
2240
|
err
|
|
1958
2241
|
}, "chat error");
|
|
1959
2242
|
if (isModelProviderError(err)) {
|
|
2243
|
+
const billingBlocked = parseBillingBlockedProviderSignal(err instanceof Error ? err.message : String(err));
|
|
2244
|
+
if (billingBlocked) {
|
|
2245
|
+
emitBillingBlocked({
|
|
2246
|
+
writer,
|
|
2247
|
+
sessionId,
|
|
2248
|
+
created,
|
|
2249
|
+
billingBlocked
|
|
2250
|
+
});
|
|
2251
|
+
writer.write("data: [DONE]\n\n");
|
|
2252
|
+
return;
|
|
2253
|
+
}
|
|
1960
2254
|
const providerError = toModelProviderError(err);
|
|
1961
2255
|
log.warn({
|
|
1962
2256
|
event: "model_provider_error",
|
|
1963
2257
|
chatcmpl_id: sessionId,
|
|
1964
2258
|
code: providerError.code,
|
|
1965
2259
|
upstream_status: providerError.upstreamStatus,
|
|
1966
|
-
provider: providerError.provider
|
|
2260
|
+
provider: providerError.provider,
|
|
2261
|
+
retry_attempts: autoRetry.attempts()
|
|
1967
2262
|
}, "forwarding model-provider error to client");
|
|
1968
2263
|
emitModelProviderError({
|
|
1969
2264
|
writer,
|
|
1970
2265
|
sessionId,
|
|
1971
2266
|
created,
|
|
1972
|
-
providerError
|
|
2267
|
+
providerError,
|
|
2268
|
+
retryAttempts: autoRetry.attempts()
|
|
1973
2269
|
});
|
|
1974
2270
|
} else sendChunk({ content: `\n[error: ${err?.message ?? err}]` }, "stop");
|
|
1975
2271
|
writer.write("data: [DONE]\n\n");
|
|
@@ -1986,7 +2282,8 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
|
|
|
1986
2282
|
});
|
|
1987
2283
|
}
|
|
1988
2284
|
function emitSessionMessagesTrailer({ writer, sm, baselineMessageCount, sessionId, created, usage }) {
|
|
1989
|
-
const
|
|
2285
|
+
const all = sm.buildSessionContext().messages;
|
|
2286
|
+
const sessionMessages = piMessagesToOpenAI(dropAbortedModelAttempts(all.slice(baselineMessageCount)));
|
|
1990
2287
|
if (sessionMessages.length === 0 && !usage) return;
|
|
1991
2288
|
const chunk = {
|
|
1992
2289
|
id: sessionId,
|
|
@@ -2010,7 +2307,7 @@ function emitSessionMessagesTrailer({ writer, sm, baselineMessageCount, sessionI
|
|
|
2010
2307
|
* cleanly closes the stream for any OpenAI-shaped reader that ignores the
|
|
2011
2308
|
* extension field.
|
|
2012
2309
|
*/
|
|
2013
|
-
function emitModelProviderError({ writer, sessionId, created, providerError }) {
|
|
2310
|
+
function emitModelProviderError({ writer, sessionId, created, providerError, retryAttempts }) {
|
|
2014
2311
|
const chunk = {
|
|
2015
2312
|
id: sessionId,
|
|
2016
2313
|
object: "chat.completion.chunk",
|
|
@@ -2025,7 +2322,25 @@ function emitModelProviderError({ writer, sessionId, created, providerError }) {
|
|
|
2025
2322
|
message: providerError.message,
|
|
2026
2323
|
provider: providerError.provider,
|
|
2027
2324
|
type: providerError.type,
|
|
2028
|
-
upstream_status: providerError.upstreamStatus
|
|
2325
|
+
upstream_status: providerError.upstreamStatus,
|
|
2326
|
+
retry_attempts: retryAttempts
|
|
2327
|
+
}
|
|
2328
|
+
};
|
|
2329
|
+
writer.write(`data: ${JSON.stringify(chunk)}\n\n`);
|
|
2330
|
+
}
|
|
2331
|
+
function emitBillingBlocked({ writer, sessionId, created, billingBlocked }) {
|
|
2332
|
+
const chunk = {
|
|
2333
|
+
id: sessionId,
|
|
2334
|
+
object: "chat.completion.chunk",
|
|
2335
|
+
created,
|
|
2336
|
+
choices: [{
|
|
2337
|
+
index: 0,
|
|
2338
|
+
delta: {},
|
|
2339
|
+
finish_reason: "stop"
|
|
2340
|
+
}],
|
|
2341
|
+
x_billing_blocked: {
|
|
2342
|
+
block_reason: billingBlocked.blockReason,
|
|
2343
|
+
message: billingBlocked.message
|
|
2029
2344
|
}
|
|
2030
2345
|
};
|
|
2031
2346
|
writer.write(`data: ${JSON.stringify(chunk)}\n\n`);
|
|
@@ -2091,7 +2406,7 @@ async function runBlocking$1({ session, sm, prompt, images, sessionId, created,
|
|
|
2091
2406
|
};
|
|
2092
2407
|
if (toolCalls.length) message.tool_calls = toolCalls;
|
|
2093
2408
|
const all = sm.buildSessionContext().messages;
|
|
2094
|
-
const sessionMessages = piMessagesToOpenAI(all.slice(baselineMessageCount));
|
|
2409
|
+
const sessionMessages = piMessagesToOpenAI(dropAbortedModelAttempts(all.slice(baselineMessageCount)));
|
|
2095
2410
|
const response = {
|
|
2096
2411
|
id: sessionId,
|
|
2097
2412
|
object: "chat.completion",
|
|
@@ -2122,12 +2437,43 @@ const steerContentPartSchema = z.union([z.object({
|
|
|
2122
2437
|
type: z.literal("image_url"),
|
|
2123
2438
|
image_url: z.object({ url: z.string() })
|
|
2124
2439
|
})]);
|
|
2440
|
+
const steerSenderSchema = z.object({
|
|
2441
|
+
kind: z.enum([
|
|
2442
|
+
"user",
|
|
2443
|
+
"system",
|
|
2444
|
+
"agent",
|
|
2445
|
+
"external"
|
|
2446
|
+
]),
|
|
2447
|
+
displayName: z.string().nullable(),
|
|
2448
|
+
participantId: z.string().nullable(),
|
|
2449
|
+
attribute: z.boolean()
|
|
2450
|
+
});
|
|
2125
2451
|
const steerRequestSchema = z.object({
|
|
2126
2452
|
content: z.union([z.string().min(1), z.array(steerContentPartSchema).min(1)]).optional(),
|
|
2127
2453
|
mode: z.enum(["steer", "follow-up"]).default("steer"),
|
|
2454
|
+
sender: steerSenderSchema.optional(),
|
|
2128
2455
|
text: z.string().optional()
|
|
2129
2456
|
}).refine((v) => v.content !== void 0 || v.text !== void 0, { message: "content or text is required" });
|
|
2130
|
-
|
|
2457
|
+
/**
|
|
2458
|
+
* Render a steer as the user-role text the model reads.
|
|
2459
|
+
*
|
|
2460
|
+
* Typed senders let the harness own this frame instead of callers embedding a
|
|
2461
|
+
* prefix in the text itself — nothing downstream string-matches a marker.
|
|
2462
|
+
* Forms mirror the worker's history projection of the same rows
|
|
2463
|
+
* (messageToOai / to-oai-history) so the live steer and its re-read next turn
|
|
2464
|
+
* agree: system-authored rows project as `[platform] <text>`; attributed
|
|
2465
|
+
* humans as `Name: <text>`. Agent steers keep the explicit not-the-user agent
|
|
2466
|
+
* form rather than the human `Name:` attribution, per the impersonation fix
|
|
2467
|
+
* this carries forward. An attributed steer with no text (attachment-only)
|
|
2468
|
+
* still carries its attribution over `(attachments only)`.
|
|
2469
|
+
*/
|
|
2470
|
+
function frameSteerText({ text, sender }) {
|
|
2471
|
+
if (!sender?.attribute) return text;
|
|
2472
|
+
const body = text.length > 0 ? text : "(attachments only)";
|
|
2473
|
+
if (sender.kind === "system") return `[platform] ${body}`;
|
|
2474
|
+
if (sender.kind === "agent") return sender.displayName ? `[platform] steer from agent ${sender.displayName}: ${body}` : `[platform] steer from agent: ${body}`;
|
|
2475
|
+
return `${sender.displayName ?? "unnamed participant"}: ${body}`;
|
|
2476
|
+
}
|
|
2131
2477
|
async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
2132
2478
|
const { log } = requestLogger(request);
|
|
2133
2479
|
const session = registry.get(sessionId);
|
|
@@ -2135,17 +2481,21 @@ async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
|
2135
2481
|
let body;
|
|
2136
2482
|
try {
|
|
2137
2483
|
body = JSON.parse(await request.text());
|
|
2138
|
-
} catch {
|
|
2484
|
+
} catch (_err) {
|
|
2139
2485
|
return jsonError(400, "invalid json");
|
|
2140
2486
|
}
|
|
2141
2487
|
const parsed = steerRequestSchema.safeParse(body);
|
|
2142
2488
|
if (!parsed.success) return jsonError(400, parsed.error.message);
|
|
2143
2489
|
const { mode } = parsed.data;
|
|
2144
2490
|
const { text, images } = await extractContent(parsed.data.content ?? parsed.data.text ?? "", {
|
|
2145
|
-
userAgent: request.headers.get("user-agent") ??
|
|
2491
|
+
userAgent: request.headers.get("user-agent") ?? null,
|
|
2146
2492
|
log
|
|
2147
2493
|
});
|
|
2148
2494
|
if (!text && images.length === 0) return jsonError(400, "steer has no content");
|
|
2495
|
+
const framedText = frameSteerText({
|
|
2496
|
+
text,
|
|
2497
|
+
sender: parsed.data.sender ?? null
|
|
2498
|
+
});
|
|
2149
2499
|
const steerId = randomUUID();
|
|
2150
2500
|
let queues = pendingSteerIds.get(sessionId);
|
|
2151
2501
|
if (!queues) {
|
|
@@ -2157,10 +2507,10 @@ async function handleSteer(request, sessionId, registry, pendingSteerIds) {
|
|
|
2157
2507
|
}
|
|
2158
2508
|
if (mode === "follow-up") {
|
|
2159
2509
|
queues.followUp.push(steerId);
|
|
2160
|
-
await session.followUp(
|
|
2510
|
+
await session.followUp(framedText, images.length > 0 ? images : void 0);
|
|
2161
2511
|
} else {
|
|
2162
2512
|
queues.steer.push(steerId);
|
|
2163
|
-
await session.steer(
|
|
2513
|
+
await session.steer(framedText, images.length > 0 ? images : void 0);
|
|
2164
2514
|
}
|
|
2165
2515
|
log.info({
|
|
2166
2516
|
event: "steer_accepted",
|
|
@@ -2295,7 +2645,10 @@ async function extractImagesFromContent(content, log) {
|
|
|
2295
2645
|
for (const part of content) if (part?.type === "input_image") {
|
|
2296
2646
|
const url = part.image_url ?? part.url;
|
|
2297
2647
|
if (typeof url === "string" && /^https?:\/\//i.test(url)) try {
|
|
2298
|
-
images.push(await fetchImageAsBase64({
|
|
2648
|
+
images.push(await fetchImageAsBase64({
|
|
2649
|
+
url,
|
|
2650
|
+
userAgent: null
|
|
2651
|
+
}));
|
|
2299
2652
|
} catch (err) {
|
|
2300
2653
|
log.warn({
|
|
2301
2654
|
event: "image_fetch_failed",
|
|
@@ -2316,7 +2669,7 @@ async function handleResponses(request, ctx, registry) {
|
|
|
2316
2669
|
let parsed;
|
|
2317
2670
|
try {
|
|
2318
2671
|
parsed = JSON.parse(await request.text());
|
|
2319
|
-
} catch {
|
|
2672
|
+
} catch (_err) {
|
|
2320
2673
|
return jsonError(400, "invalid json");
|
|
2321
2674
|
}
|
|
2322
2675
|
const { input, stream = false, model: modelInput, instructions, x_model: modelSpecInput } = parsed ?? {};
|
|
@@ -2388,6 +2741,7 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2388
2741
|
const tcByContentIdx = /* @__PURE__ */ new Map();
|
|
2389
2742
|
let nextOutputIndex = 0;
|
|
2390
2743
|
let capturedModelError = null;
|
|
2744
|
+
const autoRetry = createAutoRetryObserver(log);
|
|
2391
2745
|
const send = (event) => {
|
|
2392
2746
|
writer.write(`data: ${JSON.stringify(event)}\n\n`);
|
|
2393
2747
|
};
|
|
@@ -2396,7 +2750,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2396
2750
|
event: "model_provider_error",
|
|
2397
2751
|
code: providerError.code,
|
|
2398
2752
|
upstream_status: providerError.upstreamStatus,
|
|
2399
|
-
provider: providerError.provider
|
|
2753
|
+
provider: providerError.provider,
|
|
2754
|
+
retry_attempts: autoRetry.attempts()
|
|
2400
2755
|
}, "forwarding model-provider error to client");
|
|
2401
2756
|
send({
|
|
2402
2757
|
type: "response.failed",
|
|
@@ -2410,7 +2765,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2410
2765
|
x_model_provider_error: {
|
|
2411
2766
|
provider: providerError.provider,
|
|
2412
2767
|
type: providerError.type,
|
|
2413
|
-
upstream_status: providerError.upstreamStatus
|
|
2768
|
+
upstream_status: providerError.upstreamStatus,
|
|
2769
|
+
retry_attempts: autoRetry.attempts()
|
|
2414
2770
|
}
|
|
2415
2771
|
}
|
|
2416
2772
|
}
|
|
@@ -2446,8 +2802,10 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
|
|
|
2446
2802
|
};
|
|
2447
2803
|
session.subscribe((event) => {
|
|
2448
2804
|
const ev = event;
|
|
2805
|
+
autoRetry.observe(ev);
|
|
2806
|
+
if (ev.type === "auto_retry_end" && ev.success === true) capturedModelError = null;
|
|
2449
2807
|
const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
|
|
2450
|
-
if (endedMessage?.stopReason === "error"
|
|
2808
|
+
if (endedMessage?.stopReason === "error") capturedModelError = typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0 ? endedMessage.errorMessage : "model provider call failed without an error message";
|
|
2451
2809
|
if (ev.type !== "message_update") return;
|
|
2452
2810
|
const inner = ev.assistantMessageEvent;
|
|
2453
2811
|
if (!inner) return;
|
|
@@ -2851,4 +3209,4 @@ function chainMiddleware(middlewares) {
|
|
|
2851
3209
|
};
|
|
2852
3210
|
}
|
|
2853
3211
|
//#endregion
|
|
2854
|
-
export { DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, PI_AGENT_DIR, VALID_THINKING_LEVELS, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
|
|
3212
|
+
export { BILLING_BLOCK_REASONS, DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, PI_AGENT_DIR, VALID_THINKING_LEVELS, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
|