@skydiveai/pi-server 0.1.0-beta.267 → 0.1.0-beta.2728

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -208,30 +208,7 @@ declare function createPrewarm({
208
208
  */
209
209
  declare const KNOWN_PI_PROVIDERS: readonly ["amazon-bedrock", "anthropic", "google", "google-vertex", "openai", "azure-openai-responses", "openai-codex", "deepseek", "github-copilot", "xai", "groq", "cerebras", "openrouter", "vercel-ai-gateway", "zai", "mistral", "minimax", "minimax-cn", "moonshotai", "moonshotai-cn", "huggingface", "fireworks", "opencode", "opencode-go", "kimi-coding", "cloudflare-workers-ai", "cloudflare-ai-gateway", "xiaomi", "xiaomi-token-plan-cn", "xiaomi-token-plan-ams", "xiaomi-token-plan-sgp"];
210
210
  declare const modelSpecSchema: z.ZodDiscriminatedUnion<"api", [z.ZodObject<{
211
- api: z.ZodLiteral<"openai-completions">;
212
- compat: z.ZodOptional<z.ZodObject<{
213
- supportsReasoningEffort: z.ZodOptional<z.ZodBoolean>;
214
- requiresThinkingAsText: z.ZodOptional<z.ZodBoolean>;
215
- thinkingFormat: z.ZodOptional<z.ZodEnum<["openai", "openrouter", "deepseek", "zai", "qwen", "qwen-chat-template"]>>;
216
- supportsStrictMode: z.ZodOptional<z.ZodBoolean>;
217
- maxTokensField: z.ZodOptional<z.ZodEnum<["max_completion_tokens", "max_tokens"]>>;
218
- cacheControlFormat: z.ZodOptional<z.ZodLiteral<"anthropic">>;
219
- }, "strip", z.ZodTypeAny, {
220
- supportsReasoningEffort?: boolean | undefined;
221
- requiresThinkingAsText?: boolean | undefined;
222
- thinkingFormat?: "openai" | "deepseek" | "openrouter" | "zai" | "qwen" | "qwen-chat-template" | undefined;
223
- supportsStrictMode?: boolean | undefined;
224
- maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
225
- cacheControlFormat?: "anthropic" | undefined;
226
- }, {
227
- supportsReasoningEffort?: boolean | undefined;
228
- requiresThinkingAsText?: boolean | undefined;
229
- thinkingFormat?: "openai" | "deepseek" | "openrouter" | "zai" | "qwen" | "qwen-chat-template" | undefined;
230
- supportsStrictMode?: boolean | undefined;
231
- maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
232
- cacheControlFormat?: "anthropic" | undefined;
233
- }>>; /** Exact model id the serving endpoint expects. */
234
- id: z.ZodString; /** Display name shown in pi surfaces and session logs. */
211
+ /** Exact model id the serving endpoint expects. */id: z.ZodString; /** Display name shown in pi surfaces and session logs. */
235
212
  name: z.ZodString; /** pi-ai provider key — also the auth-storage key for `apiKey`. */
236
213
  provider: z.ZodEnum<["amazon-bedrock", "anthropic", "google", "google-vertex", "openai", "azure-openai-responses", "openai-codex", "deepseek", "github-copilot", "xai", "groq", "cerebras", "openrouter", "vercel-ai-gateway", "zai", "mistral", "minimax", "minimax-cn", "moonshotai", "moonshotai-cn", "huggingface", "fireworks", "opencode", "opencode-go", "kimi-coding", "cloudflare-workers-ai", "cloudflare-ai-gateway", "xiaomi", "xiaomi-token-plan-cn", "xiaomi-token-plan-ams", "xiaomi-token-plan-sgp"]>; /** Provider API root, e.g. "https://openrouter.ai/api/v1". */
237
214
  baseUrl: z.ZodString;
@@ -244,17 +221,17 @@ declare const modelSpecSchema: z.ZodDiscriminatedUnion<"api", [z.ZodObject<{
244
221
  high: z.ZodOptional<z.ZodNullable<z.ZodString>>;
245
222
  xhigh: z.ZodOptional<z.ZodNullable<z.ZodString>>;
246
223
  }, "strip", z.ZodTypeAny, {
247
- medium?: string | null | undefined;
248
224
  off?: string | null | undefined;
249
225
  minimal?: string | null | undefined;
250
226
  low?: string | null | undefined;
227
+ medium?: string | null | undefined;
251
228
  high?: string | null | undefined;
252
229
  xhigh?: string | null | undefined;
253
230
  }, {
254
- medium?: string | null | undefined;
255
231
  off?: string | null | undefined;
256
232
  minimal?: string | null | undefined;
257
233
  low?: string | null | undefined;
234
+ medium?: string | null | undefined;
258
235
  high?: string | null | undefined;
259
236
  xhigh?: string | null | undefined;
260
237
  }>>;
@@ -268,73 +245,85 @@ declare const modelSpecSchema: z.ZodDiscriminatedUnion<"api", [z.ZodObject<{
268
245
  * placeholder that layer recognizes — the value is opaque to this server.
269
246
  */
270
247
  apiKey: z.ZodOptional<z.ZodString>;
248
+ api: z.ZodLiteral<"openai-completions">;
249
+ compat: z.ZodOptional<z.ZodObject<{
250
+ supportsReasoningEffort: z.ZodOptional<z.ZodBoolean>;
251
+ requiresThinkingAsText: z.ZodOptional<z.ZodBoolean>;
252
+ thinkingFormat: z.ZodOptional<z.ZodEnum<["openai", "openrouter", "deepseek", "zai", "qwen", "qwen-chat-template"]>>;
253
+ supportsStrictMode: z.ZodOptional<z.ZodBoolean>;
254
+ maxTokensField: z.ZodOptional<z.ZodEnum<["max_completion_tokens", "max_tokens"]>>;
255
+ cacheControlFormat: z.ZodOptional<z.ZodLiteral<"anthropic">>;
256
+ }, "strip", z.ZodTypeAny, {
257
+ supportsReasoningEffort?: boolean | undefined;
258
+ requiresThinkingAsText?: boolean | undefined;
259
+ thinkingFormat?: "deepseek" | "openai" | "openrouter" | "qwen" | "qwen-chat-template" | "zai" | undefined;
260
+ supportsStrictMode?: boolean | undefined;
261
+ maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
262
+ cacheControlFormat?: "anthropic" | undefined;
263
+ }, {
264
+ supportsReasoningEffort?: boolean | undefined;
265
+ requiresThinkingAsText?: boolean | undefined;
266
+ thinkingFormat?: "deepseek" | "openai" | "openrouter" | "qwen" | "qwen-chat-template" | "zai" | undefined;
267
+ supportsStrictMode?: boolean | undefined;
268
+ maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
269
+ cacheControlFormat?: "anthropic" | undefined;
270
+ }>>;
271
271
  }, "strip", z.ZodTypeAny, {
272
- api: "openai-completions";
273
272
  id: string;
274
273
  name: string;
275
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
274
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
276
275
  baseUrl: string;
277
276
  reasoning: boolean;
278
- input: ["text" | "image", ...("text" | "image")[]];
277
+ thinkingLevelMap?: {
278
+ off?: string | null | undefined;
279
+ minimal?: string | null | undefined;
280
+ low?: string | null | undefined;
281
+ medium?: string | null | undefined;
282
+ high?: string | null | undefined;
283
+ xhigh?: string | null | undefined;
284
+ } | undefined;
285
+ input: ["image" | "text", ...("image" | "text")[]];
279
286
  contextWindow: number;
280
287
  maxTokens: number;
288
+ apiKey?: string | undefined;
289
+ api: "openai-completions";
281
290
  compat?: {
282
291
  supportsReasoningEffort?: boolean | undefined;
283
292
  requiresThinkingAsText?: boolean | undefined;
284
- thinkingFormat?: "openai" | "deepseek" | "openrouter" | "zai" | "qwen" | "qwen-chat-template" | undefined;
293
+ thinkingFormat?: "deepseek" | "openai" | "openrouter" | "qwen" | "qwen-chat-template" | "zai" | undefined;
285
294
  supportsStrictMode?: boolean | undefined;
286
295
  maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
287
296
  cacheControlFormat?: "anthropic" | undefined;
288
297
  } | undefined;
298
+ }, {
299
+ id: string;
300
+ name: string;
301
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
302
+ baseUrl: string;
303
+ reasoning: boolean;
289
304
  thinkingLevelMap?: {
290
- medium?: string | null | undefined;
291
305
  off?: string | null | undefined;
292
306
  minimal?: string | null | undefined;
293
307
  low?: string | null | undefined;
308
+ medium?: string | null | undefined;
294
309
  high?: string | null | undefined;
295
310
  xhigh?: string | null | undefined;
296
311
  } | undefined;
297
- apiKey?: string | undefined;
298
- }, {
299
- api: "openai-completions";
300
- id: string;
301
- name: string;
302
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
303
- baseUrl: string;
304
- reasoning: boolean;
305
- input: ["text" | "image", ...("text" | "image")[]];
312
+ input: ["image" | "text", ...("image" | "text")[]];
306
313
  contextWindow: number;
307
314
  maxTokens: number;
315
+ apiKey?: string | undefined;
316
+ api: "openai-completions";
308
317
  compat?: {
309
318
  supportsReasoningEffort?: boolean | undefined;
310
319
  requiresThinkingAsText?: boolean | undefined;
311
- thinkingFormat?: "openai" | "deepseek" | "openrouter" | "zai" | "qwen" | "qwen-chat-template" | undefined;
320
+ thinkingFormat?: "deepseek" | "openai" | "openrouter" | "qwen" | "qwen-chat-template" | "zai" | undefined;
312
321
  supportsStrictMode?: boolean | undefined;
313
322
  maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
314
323
  cacheControlFormat?: "anthropic" | undefined;
315
324
  } | undefined;
316
- thinkingLevelMap?: {
317
- medium?: string | null | undefined;
318
- off?: string | null | undefined;
319
- minimal?: string | null | undefined;
320
- low?: string | null | undefined;
321
- high?: string | null | undefined;
322
- xhigh?: string | null | undefined;
323
- } | undefined;
324
- apiKey?: string | undefined;
325
325
  }>, z.ZodObject<{
326
- api: z.ZodLiteral<"anthropic-messages">;
327
- compat: z.ZodOptional<z.ZodObject<{
328
- supportsEagerToolInputStreaming: z.ZodOptional<z.ZodBoolean>;
329
- supportsLongCacheRetention: z.ZodOptional<z.ZodBoolean>;
330
- }, "strip", z.ZodTypeAny, {
331
- supportsEagerToolInputStreaming?: boolean | undefined;
332
- supportsLongCacheRetention?: boolean | undefined;
333
- }, {
334
- supportsEagerToolInputStreaming?: boolean | undefined;
335
- supportsLongCacheRetention?: boolean | undefined;
336
- }>>; /** Exact model id the serving endpoint expects. */
337
- id: z.ZodString; /** Display name shown in pi surfaces and session logs. */
326
+ /** Exact model id the serving endpoint expects. */id: z.ZodString; /** Display name shown in pi surfaces and session logs. */
338
327
  name: z.ZodString; /** pi-ai provider key — also the auth-storage key for `apiKey`. */
339
328
  provider: z.ZodEnum<["amazon-bedrock", "anthropic", "google", "google-vertex", "openai", "azure-openai-responses", "openai-codex", "deepseek", "github-copilot", "xai", "groq", "cerebras", "openrouter", "vercel-ai-gateway", "zai", "mistral", "minimax", "minimax-cn", "moonshotai", "moonshotai-cn", "huggingface", "fireworks", "opencode", "opencode-go", "kimi-coding", "cloudflare-workers-ai", "cloudflare-ai-gateway", "xiaomi", "xiaomi-token-plan-cn", "xiaomi-token-plan-ams", "xiaomi-token-plan-sgp"]>; /** Provider API root, e.g. "https://openrouter.ai/api/v1". */
340
329
  baseUrl: z.ZodString;
@@ -347,17 +336,17 @@ declare const modelSpecSchema: z.ZodDiscriminatedUnion<"api", [z.ZodObject<{
347
336
  high: z.ZodOptional<z.ZodNullable<z.ZodString>>;
348
337
  xhigh: z.ZodOptional<z.ZodNullable<z.ZodString>>;
349
338
  }, "strip", z.ZodTypeAny, {
350
- medium?: string | null | undefined;
351
339
  off?: string | null | undefined;
352
340
  minimal?: string | null | undefined;
353
341
  low?: string | null | undefined;
342
+ medium?: string | null | undefined;
354
343
  high?: string | null | undefined;
355
344
  xhigh?: string | null | undefined;
356
345
  }, {
357
- medium?: string | null | undefined;
358
346
  off?: string | null | undefined;
359
347
  minimal?: string | null | undefined;
360
348
  low?: string | null | undefined;
349
+ medium?: string | null | undefined;
361
350
  high?: string | null | undefined;
362
351
  xhigh?: string | null | undefined;
363
352
  }>>;
@@ -371,52 +360,63 @@ declare const modelSpecSchema: z.ZodDiscriminatedUnion<"api", [z.ZodObject<{
371
360
  * placeholder that layer recognizes — the value is opaque to this server.
372
361
  */
373
362
  apiKey: z.ZodOptional<z.ZodString>;
363
+ api: z.ZodLiteral<"anthropic-messages">;
364
+ compat: z.ZodOptional<z.ZodObject<{
365
+ supportsEagerToolInputStreaming: z.ZodOptional<z.ZodBoolean>;
366
+ supportsLongCacheRetention: z.ZodOptional<z.ZodBoolean>;
367
+ }, "strip", z.ZodTypeAny, {
368
+ supportsEagerToolInputStreaming?: boolean | undefined;
369
+ supportsLongCacheRetention?: boolean | undefined;
370
+ }, {
371
+ supportsEagerToolInputStreaming?: boolean | undefined;
372
+ supportsLongCacheRetention?: boolean | undefined;
373
+ }>>;
374
374
  }, "strip", z.ZodTypeAny, {
375
- api: "anthropic-messages";
376
375
  id: string;
377
376
  name: string;
378
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
377
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
379
378
  baseUrl: string;
380
379
  reasoning: boolean;
381
- input: ["text" | "image", ...("text" | "image")[]];
382
- contextWindow: number;
383
- maxTokens: number;
384
- compat?: {
385
- supportsEagerToolInputStreaming?: boolean | undefined;
386
- supportsLongCacheRetention?: boolean | undefined;
387
- } | undefined;
388
380
  thinkingLevelMap?: {
389
- medium?: string | null | undefined;
390
381
  off?: string | null | undefined;
391
382
  minimal?: string | null | undefined;
392
383
  low?: string | null | undefined;
384
+ medium?: string | null | undefined;
393
385
  high?: string | null | undefined;
394
386
  xhigh?: string | null | undefined;
395
387
  } | undefined;
396
- apiKey?: string | undefined;
397
- }, {
398
- api: "anthropic-messages";
399
- id: string;
400
- name: string;
401
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
402
- baseUrl: string;
403
- reasoning: boolean;
404
- input: ["text" | "image", ...("text" | "image")[]];
388
+ input: ["image" | "text", ...("image" | "text")[]];
405
389
  contextWindow: number;
406
390
  maxTokens: number;
391
+ apiKey?: string | undefined;
392
+ api: "anthropic-messages";
407
393
  compat?: {
408
394
  supportsEagerToolInputStreaming?: boolean | undefined;
409
395
  supportsLongCacheRetention?: boolean | undefined;
410
396
  } | undefined;
397
+ }, {
398
+ id: string;
399
+ name: string;
400
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
401
+ baseUrl: string;
402
+ reasoning: boolean;
411
403
  thinkingLevelMap?: {
412
- medium?: string | null | undefined;
413
404
  off?: string | null | undefined;
414
405
  minimal?: string | null | undefined;
415
406
  low?: string | null | undefined;
407
+ medium?: string | null | undefined;
416
408
  high?: string | null | undefined;
417
409
  xhigh?: string | null | undefined;
418
410
  } | undefined;
411
+ input: ["image" | "text", ...("image" | "text")[]];
412
+ contextWindow: number;
413
+ maxTokens: number;
419
414
  apiKey?: string | undefined;
415
+ api: "anthropic-messages";
416
+ compat?: {
417
+ supportsEagerToolInputStreaming?: boolean | undefined;
418
+ supportsLongCacheRetention?: boolean | undefined;
419
+ } | undefined;
420
420
  }>]>;
421
421
  type ModelSpec = z.infer<typeof modelSpecSchema>;
422
422
  /**
@@ -426,16 +426,20 @@ type ModelSpec = z.infer<typeof modelSpecSchema>;
426
426
  */
427
427
  declare function parseModelSpec(input: unknown, log: Logger): ModelSpec | null;
428
428
  declare function buildModelFromSpec(spec: ModelSpec): {
429
- compat?: {
430
- supportsReasoningEffort?: boolean | undefined;
431
- requiresThinkingAsText?: boolean | undefined;
432
- thinkingFormat?: "openai" | "deepseek" | "openrouter" | "zai" | "qwen" | "qwen-chat-template" | undefined;
433
- supportsStrictMode?: boolean | undefined;
434
- maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
435
- cacheControlFormat?: "anthropic" | undefined;
429
+ id: string;
430
+ name: string;
431
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
432
+ baseUrl: string;
433
+ reasoning: boolean;
434
+ thinkingLevelMap?: {
435
+ off?: string | null | undefined;
436
+ minimal?: string | null | undefined;
437
+ low?: string | null | undefined;
438
+ medium?: string | null | undefined;
439
+ high?: string | null | undefined;
440
+ xhigh?: string | null | undefined;
436
441
  } | undefined;
437
- api: "openai-completions";
438
- input: ["text" | "image", ...("text" | "image")[]];
442
+ input: ["image" | "text", ...("image" | "text")[]];
439
443
  cost: {
440
444
  input: number;
441
445
  output: number;
@@ -444,26 +448,30 @@ declare function buildModelFromSpec(spec: ModelSpec): {
444
448
  };
445
449
  contextWindow: number;
446
450
  maxTokens: number;
451
+ api: "openai-completions";
452
+ compat?: {
453
+ supportsReasoningEffort?: boolean | undefined;
454
+ requiresThinkingAsText?: boolean | undefined;
455
+ thinkingFormat?: "deepseek" | "openai" | "openrouter" | "qwen" | "qwen-chat-template" | "zai" | undefined;
456
+ supportsStrictMode?: boolean | undefined;
457
+ maxTokensField?: "max_completion_tokens" | "max_tokens" | undefined;
458
+ cacheControlFormat?: "anthropic" | undefined;
459
+ } | undefined;
460
+ } | {
461
+ id: string;
462
+ name: string;
463
+ provider: "amazon-bedrock" | "anthropic" | "azure-openai-responses" | "cerebras" | "cloudflare-ai-gateway" | "cloudflare-workers-ai" | "deepseek" | "fireworks" | "github-copilot" | "google" | "google-vertex" | "groq" | "huggingface" | "kimi-coding" | "minimax" | "minimax-cn" | "mistral" | "moonshotai" | "moonshotai-cn" | "openai" | "openai-codex" | "opencode" | "opencode-go" | "openrouter" | "vercel-ai-gateway" | "xai" | "xiaomi" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-sgp" | "zai";
464
+ baseUrl: string;
465
+ reasoning: boolean;
447
466
  thinkingLevelMap?: {
448
- medium?: string | null | undefined;
449
467
  off?: string | null | undefined;
450
468
  minimal?: string | null | undefined;
451
469
  low?: string | null | undefined;
470
+ medium?: string | null | undefined;
452
471
  high?: string | null | undefined;
453
472
  xhigh?: string | null | undefined;
454
473
  } | undefined;
455
- id: string;
456
- name: string;
457
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
458
- baseUrl: string;
459
- reasoning: boolean;
460
- } | {
461
- compat?: {
462
- supportsEagerToolInputStreaming?: boolean | undefined;
463
- supportsLongCacheRetention?: boolean | undefined;
464
- } | undefined;
465
- api: "anthropic-messages";
466
- input: ["text" | "image", ...("text" | "image")[]];
474
+ input: ["image" | "text", ...("image" | "text")[]];
467
475
  cost: {
468
476
  input: number;
469
477
  output: number;
@@ -472,19 +480,11 @@ declare function buildModelFromSpec(spec: ModelSpec): {
472
480
  };
473
481
  contextWindow: number;
474
482
  maxTokens: number;
475
- thinkingLevelMap?: {
476
- medium?: string | null | undefined;
477
- off?: string | null | undefined;
478
- minimal?: string | null | undefined;
479
- low?: string | null | undefined;
480
- high?: string | null | undefined;
481
- xhigh?: string | null | undefined;
483
+ api: "anthropic-messages";
484
+ compat?: {
485
+ supportsEagerToolInputStreaming?: boolean | undefined;
486
+ supportsLongCacheRetention?: boolean | undefined;
482
487
  } | undefined;
483
- id: string;
484
- name: string;
485
- provider: "anthropic" | "amazon-bedrock" | "google" | "google-vertex" | "openai" | "azure-openai-responses" | "openai-codex" | "deepseek" | "github-copilot" | "xai" | "groq" | "cerebras" | "openrouter" | "vercel-ai-gateway" | "zai" | "mistral" | "minimax" | "minimax-cn" | "moonshotai" | "moonshotai-cn" | "huggingface" | "fireworks" | "opencode" | "opencode-go" | "kimi-coding" | "cloudflare-workers-ai" | "cloudflare-ai-gateway" | "xiaomi" | "xiaomi-token-plan-cn" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-sgp";
486
- baseUrl: string;
487
- reasoning: boolean;
488
488
  };
489
489
  /**
490
490
  * Resolve the model a request should run on: a valid `x_model` wins,
@@ -542,4 +542,27 @@ declare function getCurrentTraceparent(): string | null;
542
542
  /** Extract the trace id (second segment) from a `traceparent` value. */
543
543
  declare function parseTraceId(traceparent: string | null): string | null;
544
544
  //#endregion
545
- export { type A2AOptions, type AnthropicMessagesOptions, type ChatCompletionsOptions, type CreateSessionArgs, DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, type Middleware, type ModelSpec, PI_AGENT_DIR, type PostPromptCallback, type ProtocolHandler, type ProtocolsOptions, type ResponsesOptions, type SessionFactory, type SessionSetupCallback, type TraceContext, VALID_THINKING_LEVELS, type WebHandler, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
545
+ //#region src/billing-blocked-provider-signal.d.ts
546
+ /**
547
+ * Duplicated from the billing gate's `BlockReason` (apps/anyone/billing
548
+ * gate/block-reason.ts): pi-server runs inside the sandbox and must not
549
+ * depend on platform packages. A reason the proxy sends that predates this
550
+ * build fails the enum and degrades to the untyped `billing_blocked`
551
+ * handling — never a crash. Exported so agent-lifecycle's chat.test.ts can
552
+ * pin this copy against its own (which billing's block-reason.test.ts in
553
+ * turn pins against the gate), keeping every copy drift-checked.
554
+ */
555
+ declare const BILLING_BLOCK_REASONS: readonly ["insufficient_balance", "payment_failed", "hard_spend_limit_reached"];
556
+ declare const payloadSchema: z.ZodObject<{
557
+ blockReason: z.ZodEnum<["insufficient_balance", "payment_failed", "hard_spend_limit_reached"]>;
558
+ message: z.ZodString;
559
+ }, "strict", z.ZodTypeAny, {
560
+ blockReason: "hard_spend_limit_reached" | "insufficient_balance" | "payment_failed";
561
+ message: string;
562
+ }, {
563
+ blockReason: "hard_spend_limit_reached" | "insufficient_balance" | "payment_failed";
564
+ message: string;
565
+ }>;
566
+ type BillingBlockedProviderSignal = z.infer<typeof payloadSchema>;
567
+ //#endregion
568
+ export { type A2AOptions, type AnthropicMessagesOptions, BILLING_BLOCK_REASONS, type BillingBlockedProviderSignal, type ChatCompletionsOptions, type CreateSessionArgs, DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, type Middleware, type ModelSpec, PI_AGENT_DIR, type PostPromptCallback, type ProtocolHandler, type ProtocolsOptions, type ResponsesOptions, type SessionFactory, type SessionSetupCallback, type TraceContext, VALID_THINKING_LEVELS, type WebHandler, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
package/dist/index.mjs CHANGED
@@ -259,7 +259,7 @@ function parseShellEnv(request) {
259
259
  const env = {};
260
260
  for (const [k, v] of Object.entries(parsed)) if (typeof v === "string") env[k] = v;
261
261
  return Object.keys(env).length > 0 ? env : null;
262
- } catch {
262
+ } catch (_err) {
263
263
  return null;
264
264
  }
265
265
  }
@@ -283,12 +283,12 @@ function createSSEResponse(handler) {
283
283
  write(chunk) {
284
284
  try {
285
285
  controller.enqueue(encoder.encode(chunk));
286
- } catch {}
286
+ } catch (_err) {}
287
287
  },
288
288
  close() {
289
289
  try {
290
290
  controller.close();
291
- } catch {}
291
+ } catch (_err) {}
292
292
  }
293
293
  };
294
294
  const keepalive = setInterval(() => {
@@ -298,13 +298,13 @@ function createSSEResponse(handler) {
298
298
  clearInterval(keepalive);
299
299
  try {
300
300
  controller.close();
301
- } catch {}
301
+ } catch (_err) {}
302
302
  }, (err) => {
303
303
  logger.error({ err }, "SSE handler crashed — stream closed without response");
304
304
  clearInterval(keepalive);
305
305
  try {
306
306
  controller.close();
307
- } catch {}
307
+ } catch (_err) {}
308
308
  });
309
309
  return new Response(stream, {
310
310
  status: 200,
@@ -373,6 +373,57 @@ async function runConversation({ session, prompt, images, log, postPrompt }) {
373
373
  });
374
374
  }
375
375
  /**
376
+ * Track the agent session's built-in model-call auto-retry so it leaves a
377
+ * trace.
378
+ *
379
+ * `AgentSession` already restarts a failed assistant turn in place (via
380
+ * `agent.continue()`, so no prompt is replayed and no tool re-executes) for the
381
+ * transient provider/transport failures pi classifies as retryable -- dropped
382
+ * streams, `terminated`, 5xx, overloaded, rate limits. It is on by default,
383
+ * with its own budget and backoff, and it emits `auto_retry_start` /
384
+ * `auto_retry_end` around each attempt.
385
+ *
386
+ * Nothing consumed those events, so a retry left no trace anywhere: a call that
387
+ * succeeded first try and one that burned the whole budget before failing
388
+ * produced the same terminal error, and the fleet-wide retry rate was
389
+ * unmeasurable. That gap is why a 2026-08-16 investigation into three runs lost
390
+ * to `provider_error: terminated` could not tell whether the budget had run out
391
+ * (ANY-7101).
392
+ *
393
+ * Exposed as a handler rather than its own `session.subscribe` call so each
394
+ * protocol feeds it from the single subscription it already owns -- one
395
+ * subscriber, explicit ordering.
396
+ *
397
+ * `attempts()` reports what has been spent so far, so a terminal error can
398
+ * carry the count to the worker, where it lands in a log group we can query
399
+ * fleet-wide (the sandbox's own logs are not).
400
+ */
401
+ function createAutoRetryObserver(log) {
402
+ let attempts = 0;
403
+ return {
404
+ observe(event) {
405
+ if (event.type === "auto_retry_start") {
406
+ attempts = typeof event.attempt === "number" ? event.attempt : attempts + 1;
407
+ log.warn({
408
+ event: "model_call_auto_retry",
409
+ attempt: event.attempt,
410
+ max_attempts: event.maxAttempts,
411
+ delay_ms: event.delayMs,
412
+ error_message: event.errorMessage
413
+ }, "retrying failed model call in place");
414
+ return;
415
+ }
416
+ if (event.type === "auto_retry_end") log.warn({
417
+ event: "model_call_auto_retry_end",
418
+ attempt: event.attempt,
419
+ success: event.success,
420
+ final_error: event.finalError
421
+ }, event.success ? "model call recovered after retry" : "model call retries exhausted");
422
+ },
423
+ attempts: () => attempts
424
+ };
425
+ }
426
+ /**
376
427
  * Hard-stop the in-flight turn for a session. Shared by every protocol's
377
428
  * `/:id/abort` route: a cancel signals the stop explicitly instead of relying
378
429
  * on a dropped connection. `session.abort()` interrupts the turn and resolves
@@ -419,7 +470,10 @@ async function extractParts(parts, log) {
419
470
  }
420
471
  if (part.url != null && part.mediaType?.startsWith("image/")) {
421
472
  try {
422
- images.push(await fetchImageAsBase64({ url: part.url }));
473
+ images.push(await fetchImageAsBase64({
474
+ url: part.url,
475
+ userAgent: null
476
+ }));
423
477
  } catch (err) {
424
478
  log.warn({
425
479
  event: "a2a_image_fetch_failed",
@@ -877,6 +931,7 @@ function classifyCode(status, message, code) {
877
931
  if (status === 429 || code === "rate_limit_exceeded" || /rate[_ ]limit|too many requests|resource[_ ]exhausted/i.test(message)) return "rate_limited";
878
932
  if (status === 529 || code === "provider_overloaded" || /overloaded/i.test(message)) return "provider_overloaded";
879
933
  if (status === 503 || status === 504 || code === "provider_unavailable" || /no healthy upstream|upstream request timeout|stream timeout|service unavailable|unavailable|gateway/i.test(message)) return "provider_unavailable";
934
+ if (code === "content_filter" || /finish_reason:\s*content_filter|content[_ ]filter|content[_ ]management[_ ]policy/i.test(message)) return "content_filter";
880
935
  return "provider_error";
881
936
  }
882
937
  /**
@@ -891,7 +946,7 @@ function isModelProviderError(err) {
891
946
  if (numericStatus(e) !== null) return true;
892
947
  if (e.error && typeof e.error === "object") return true;
893
948
  const msg = messageText(err);
894
- return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits/i.test(msg);
949
+ return /prompt is too long|context[_ ]length[_ ]exceeded|exceeds maximum token limit|input token count|rate[_ ]limit|overloaded|no healthy upstream|upstream request timeout|insufficient credits|requires more credits|credit balance is too low|payment required|purchase more credits|finish_reason:\s*content_filter|content[_ ]filter|content[_ ]management[_ ]policy/i.test(msg);
895
950
  }
896
951
  /** Build the structured, forwardable error from a thrown model-provider error. */
897
952
  function toModelProviderError(err) {
@@ -947,7 +1002,10 @@ async function extractContent$1(content, log) {
947
1002
  data: src.data
948
1003
  });
949
1004
  else if (src?.type === "url" && src.url) try {
950
- images.push(await fetchImageAsBase64({ url: src.url }));
1005
+ images.push(await fetchImageAsBase64({
1006
+ url: src.url,
1007
+ userAgent: null
1008
+ }));
951
1009
  } catch (err) {
952
1010
  log.warn({
953
1011
  event: "image_fetch_failed",
@@ -1036,7 +1094,7 @@ async function handleMessages(request, ctx, registry) {
1036
1094
  let parsed;
1037
1095
  try {
1038
1096
  parsed = JSON.parse(await request.text());
1039
- } catch {
1097
+ } catch (_err) {
1040
1098
  return jsonError(400, "invalid json");
1041
1099
  }
1042
1100
  const { messages, stream = false, model: modelInput, system, x_model: modelSpecInput } = parsed ?? {};
@@ -1059,7 +1117,7 @@ async function handleMessages(request, ctx, registry) {
1059
1117
  const baseSystemPrompt = extractSystemPrompt(system);
1060
1118
  const { sessionId } = parseSessionId(request);
1061
1119
  const shellEnv = parseShellEnv(request);
1062
- const { session, sessionManager: sm } = await ctx.createSession({
1120
+ const { session } = await ctx.createSession({
1063
1121
  cwd: ctx.cwd,
1064
1122
  sessionId,
1065
1123
  perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
@@ -1082,7 +1140,6 @@ async function handleMessages(request, ctx, registry) {
1082
1140
  if (stream) {
1083
1141
  const response = runStream$2({
1084
1142
  session,
1085
- sm,
1086
1143
  prompt,
1087
1144
  images,
1088
1145
  id,
@@ -1098,7 +1155,6 @@ async function handleMessages(request, ctx, registry) {
1098
1155
  }
1099
1156
  const response = await runBlocking$2({
1100
1157
  session,
1101
- sm,
1102
1158
  prompt,
1103
1159
  images,
1104
1160
  id,
@@ -1112,7 +1168,7 @@ async function handleMessages(request, ctx, registry) {
1112
1168
  if (traceId) response.headers.set("x-trace-id", traceId);
1113
1169
  return response;
1114
1170
  }
1115
- function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
1171
+ function runStream$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
1116
1172
  return createSSEResponse(async (writer) => {
1117
1173
  let contentBlockIndex = 0;
1118
1174
  let textBlockOpen = false;
@@ -1158,12 +1214,15 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
1158
1214
  textBlockOpen = false;
1159
1215
  };
1160
1216
  let capturedModelError = null;
1217
+ let sawModelErrorStop = false;
1218
+ const autoRetry = createAutoRetryObserver(log);
1161
1219
  const emitProviderError = (providerError) => {
1162
1220
  log.warn({
1163
1221
  event: "model_provider_error",
1164
1222
  code: providerError.code,
1165
1223
  upstream_status: providerError.upstreamStatus,
1166
- provider: providerError.provider
1224
+ provider: providerError.provider,
1225
+ retry_attempts: autoRetry.attempts()
1167
1226
  }, "forwarding model-provider error to client");
1168
1227
  sseEvent(writer, "error", {
1169
1228
  type: "error",
@@ -1173,15 +1232,20 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
1173
1232
  x_model_provider_error: {
1174
1233
  code: providerError.code,
1175
1234
  provider: providerError.provider,
1176
- upstream_status: providerError.upstreamStatus
1235
+ upstream_status: providerError.upstreamStatus,
1236
+ retry_attempts: autoRetry.attempts()
1177
1237
  }
1178
1238
  }
1179
1239
  });
1180
1240
  };
1181
1241
  session.subscribe((event) => {
1182
1242
  const ev = event;
1243
+ autoRetry.observe(ev);
1183
1244
  const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
1184
- if (endedMessage?.stopReason === "error" && typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
1245
+ if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
1246
+ sawModelErrorStop = true;
1247
+ if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
1248
+ }
1185
1249
  if (ev.type !== "message_update") return;
1186
1250
  const inner = ev.assistantMessageEvent;
1187
1251
  if (!inner) return;
@@ -1239,6 +1303,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
1239
1303
  postPrompt
1240
1304
  });
1241
1305
  if (capturedModelError !== null) emitProviderError(toModelProviderError(new Error(capturedModelError)));
1306
+ else if (sawModelErrorStop) emitProviderError(toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message")));
1242
1307
  } catch (err) {
1243
1308
  log.error({
1244
1309
  err,
@@ -1266,7 +1331,7 @@ function runStream$2({ session, sm, prompt, images, id, sessionId, modelName, lo
1266
1331
  sseEvent(writer, "message_stop", { type: "message_stop" });
1267
1332
  });
1268
1333
  }
1269
- async function runBlocking$2({ session, sm, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
1334
+ async function runBlocking$2({ session, prompt, images, id, sessionId, modelName, log, postPrompt, registry }) {
1270
1335
  let text = "";
1271
1336
  const toolUseBlocks = [];
1272
1337
  const tcByContentIdx = /* @__PURE__ */ new Map();
@@ -1317,7 +1382,7 @@ async function runBlocking$2({ session, sm, prompt, images, id, sessionId, model
1317
1382
  }
1318
1383
  for (const block of toolUseBlocks) if (block.type === "tool_use" && typeof block.input === "string") try {
1319
1384
  block.input = JSON.parse(block.input);
1320
- } catch {
1385
+ } catch (_err) {
1321
1386
  block.input = {};
1322
1387
  }
1323
1388
  const content = [];
@@ -1379,6 +1444,87 @@ function create$2(options) {
1379
1444
  };
1380
1445
  }
1381
1446
  //#endregion
1447
+ //#region src/billing-blocked-provider-signal.ts
1448
+ const BILLING_BLOCKED_PROVIDER_SIGNAL_PREFIX = "ANYONE_BILLING_BLOCKED_V1:";
1449
+ /**
1450
+ * Duplicated from the billing gate's `BlockReason` (apps/anyone/billing
1451
+ * gate/block-reason.ts): pi-server runs inside the sandbox and must not
1452
+ * depend on platform packages. A reason the proxy sends that predates this
1453
+ * build fails the enum and degrades to the untyped `billing_blocked`
1454
+ * handling — never a crash. Exported so agent-lifecycle's chat.test.ts can
1455
+ * pin this copy against its own (which billing's block-reason.test.ts in
1456
+ * turn pins against the gate), keeping every copy drift-checked.
1457
+ */
1458
+ const BILLING_BLOCK_REASONS = [
1459
+ "insufficient_balance",
1460
+ "payment_failed",
1461
+ "hard_spend_limit_reached"
1462
+ ];
1463
+ const payloadSchema = z.object({
1464
+ blockReason: z.enum(BILLING_BLOCK_REASONS),
1465
+ message: z.string()
1466
+ }).strict();
1467
+ const nestedProxyEnvelopeSchema = z.object({
1468
+ error: z.object({
1469
+ type: z.literal("billing_blocked"),
1470
+ code: z.literal("billing_blocked"),
1471
+ message: z.string()
1472
+ }).strict(),
1473
+ blockReason: z.enum(BILLING_BLOCK_REASONS),
1474
+ message: z.string()
1475
+ }).strict();
1476
+ /**
1477
+ * pi-ai's openai-compatible providers (openrouter, xai, groq, deepseek, …) can
1478
+ * NOT fold a proxy's non-2xx body into `error.message`, so `formatProviderError`
1479
+ * (@earendil-works/pi-ai utils/error-body) composes the display string as
1480
+ * `"<status>: <body>"` or, with a provider label, `"<prefix> (<status>): <body>"`.
1481
+ * The billing gate's 402 body therefore reaches us wrapped, e.g.
1482
+ * `"402: {\"error\":{...},\"blockReason\":\"insufficient_balance\",\"message\":...}"`
1483
+ * or `"OpenRouter (402): {...}"`. Neither the bare `JSON.parse` nor the `"402 "`
1484
+ * (space) strip below recognizes that, so a real billing block from an
1485
+ * openrouter-routed model degrades to the generic model-provider error. Peel a
1486
+ * single leading `"<status>: "` / `"<prefix> (<status>): "` wrapper off the
1487
+ * front so the recovered body flows through the existing shape checks. Returns
1488
+ * the message unchanged when no wrapper is present.
1489
+ */
1490
+ function unwrapOpenAICompatStatusPrefix(message) {
1491
+ const withPrefix = message.match(/^.+ \(\d{3}\): ([\s\S]+)$/);
1492
+ if (withPrefix?.[1] !== void 0) return withPrefix[1];
1493
+ const bare = message.match(/^\d{3}: ([\s\S]+)$/);
1494
+ if (bare?.[1] !== void 0) return bare[1];
1495
+ return message;
1496
+ }
1497
+ function parsePrefixedSignal(message) {
1498
+ const normalized = message.startsWith("402 ") ? message.slice(4) : message;
1499
+ if (!normalized.startsWith(BILLING_BLOCKED_PROVIDER_SIGNAL_PREFIX)) return null;
1500
+ try {
1501
+ const parsed = payloadSchema.safeParse(JSON.parse(normalized.slice(26)));
1502
+ return parsed.success ? parsed.data : null;
1503
+ } catch (_err) {
1504
+ return null;
1505
+ }
1506
+ }
1507
+ /**
1508
+ * Decode only the versioned platform envelope. Anthropic and OpenAI prefix the
1509
+ * nested error message with HTTP 402. Google's SDK instead preserves the full
1510
+ * proxy response as JSON, so that outer shape is validated separately.
1511
+ */
1512
+ function parseBillingBlockedProviderSignal(message) {
1513
+ const unwrapped = unwrapOpenAICompatStatusPrefix(message);
1514
+ const direct = parsePrefixedSignal(unwrapped);
1515
+ if (direct) return direct;
1516
+ try {
1517
+ const envelope = nestedProxyEnvelopeSchema.safeParse(JSON.parse(unwrapped));
1518
+ if (!envelope.success) return null;
1519
+ const nested = parsePrefixedSignal(envelope.data.error.message);
1520
+ const topLevel = parsePrefixedSignal(envelope.data.message);
1521
+ if (!nested || nested.blockReason !== envelope.data.blockReason || envelope.data.message !== nested.message && (!topLevel || topLevel.blockReason !== nested.blockReason || topLevel.message !== nested.message)) return null;
1522
+ return nested;
1523
+ } catch (_err) {
1524
+ return null;
1525
+ }
1526
+ }
1527
+ //#endregion
1382
1528
  //#region src/protocols/chat-completions.ts
1383
1529
  /**
1384
1530
  * OpenAI Chat Completions–compatible protocol handler.
@@ -1563,7 +1709,7 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
1563
1709
  let args = {};
1564
1710
  try {
1565
1711
  args = tc.function.arguments ? JSON.parse(tc.function.arguments) : {};
1566
- } catch {
1712
+ } catch (_err) {
1567
1713
  args = { _raw: tc.function.arguments };
1568
1714
  }
1569
1715
  contentArr.push({
@@ -1605,6 +1751,19 @@ async function loadHistoryIntoSession({ sm, messages, upToIdxExclusive, modelNam
1605
1751
  }
1606
1752
  }
1607
1753
  }
1754
+ /**
1755
+ * Each pi/session message carries the real wall-clock `timestamp` the
1756
+ * provider stamped when it produced that message (see @earendil-works/pi-ai
1757
+ * Message). Forward it verbatim as `x_created_at` (epoch ms) on the OAI
1758
+ * trailer message so the consumer (anyone-messaging completeRun) can stamp
1759
+ * each persisted segment at its true emit time and interleave a mid-run steer
1760
+ * by createdAt — instead of positionally reconstructing per-turn times, which
1761
+ * drifts on parallel tool calls and provider-split replies.
1762
+ */
1763
+ function messageTimestampMs(m) {
1764
+ const ts = m?.timestamp;
1765
+ return typeof ts === "number" && Number.isFinite(ts) ? ts : null;
1766
+ }
1608
1767
  function piMessagesToOpenAI(messages) {
1609
1768
  const out = [];
1610
1769
  for (const m of messages) {
@@ -1633,6 +1792,8 @@ function piMessagesToOpenAI(messages) {
1633
1792
  };
1634
1793
  if (toolCalls.length) msg.tool_calls = toolCalls;
1635
1794
  if (thinking.length) msg.x_thinking = thinking;
1795
+ const createdAtMs = messageTimestampMs(m);
1796
+ if (createdAtMs !== null) msg.x_created_at = createdAtMs;
1636
1797
  out.push(msg);
1637
1798
  continue;
1638
1799
  }
@@ -1644,6 +1805,8 @@ function piMessagesToOpenAI(messages) {
1644
1805
  content: text
1645
1806
  };
1646
1807
  if (m.isError) tm.x_is_error = true;
1808
+ const createdAtMs = messageTimestampMs(m);
1809
+ if (createdAtMs !== null) tm.x_created_at = createdAtMs;
1647
1810
  out.push(tm);
1648
1811
  continue;
1649
1812
  }
@@ -1658,7 +1821,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
1658
1821
  let parsed;
1659
1822
  try {
1660
1823
  parsed = JSON.parse(await request.text());
1661
- } catch {
1824
+ } catch (_err) {
1662
1825
  return jsonError(400, "invalid json");
1663
1826
  }
1664
1827
  const { messages, stream = false, model: modelInput, thinkingLevel: thinkingInput, context_window: contextWindowInput, x_model: modelSpecInput } = parsed ?? {};
@@ -1670,7 +1833,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
1670
1833
  }
1671
1834
  if (lastUserIdx < 0) return jsonError(400, "no user message in history");
1672
1835
  const lastUser = messages[lastUserIdx];
1673
- const reqUA = request.headers.get("user-agent") ?? void 0;
1836
+ const reqUA = request.headers.get("user-agent") ?? null;
1674
1837
  const { text: prompt, images } = await extractContent(lastUser.content, {
1675
1838
  userAgent: reqUA,
1676
1839
  log
@@ -1701,7 +1864,7 @@ async function handleChatCompletions(request, ctx, registry, pendingSteerIds) {
1701
1864
  const sessionId = parsedSession.source === "header" ? parsedSession.sessionId : `chatcmpl-${parsedSession.sessionId}`;
1702
1865
  const shellEnv = parseShellEnv(request);
1703
1866
  const sessionSetupStart = performance.now();
1704
- const [{ session, sessionManager: sm }, customTools] = await Promise.all([ctx.createSession({
1867
+ const [{ session, sessionManager: sm }] = await Promise.all([ctx.createSession({
1705
1868
  cwd: ctx.cwd,
1706
1869
  sessionId,
1707
1870
  perRequestApiKeys: withSpecApiKey(resolvePerRequestApiKeys(request), modelSpec),
@@ -1788,6 +1951,8 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1788
1951
  let usage = null;
1789
1952
  let lastFollowUpCount = 0;
1790
1953
  let capturedModelError = null;
1954
+ let sawModelErrorStop = false;
1955
+ const autoRetry = createAutoRetryObserver(log);
1791
1956
  const sendChunk = (delta, finishReason) => {
1792
1957
  if (firstChunkAt === null) {
1793
1958
  firstChunkAt = performance.now();
@@ -1812,14 +1977,18 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1812
1977
  };
1813
1978
  const ensureRole = () => {
1814
1979
  if (roleEmitted) return;
1815
- sendChunk({ role: "assistant" });
1980
+ sendChunk({ role: "assistant" }, null);
1816
1981
  roleEmitted = true;
1817
1982
  };
1818
1983
  let _evCount = 0;
1819
1984
  session.subscribe((event) => {
1820
1985
  const ev = event;
1986
+ autoRetry.observe(ev);
1821
1987
  const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
1822
- if (endedMessage?.stopReason === "error" && typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
1988
+ if (endedMessage?.stopReason === "error" && endedMessage.errorMessage !== "aborted") {
1989
+ sawModelErrorStop = true;
1990
+ if (typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
1991
+ }
1823
1992
  if (ev.type === "queue_update") {
1824
1993
  const steering = Array.isArray(ev.steering) ? ev.steering.length : 0;
1825
1994
  const followUp = Array.isArray(ev.followUp) ? ev.followUp.length : 0;
@@ -1835,7 +2004,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1835
2004
  steering,
1836
2005
  follow_up: followUp,
1837
2006
  ...consumedIds.length > 0 ? { consumed_steer_ids: consumedIds } : {}
1838
- } });
2007
+ } }, null);
1839
2008
  return;
1840
2009
  }
1841
2010
  _evCount++;
@@ -1858,7 +2027,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1858
2027
  ...ev.type === "message_start" || ev.type === "message_end" ? { messageJson: JSON.stringify(ev.message).slice(0, 500) } : {}
1859
2028
  }, "harness session event");
1860
2029
  if (ev.type === "tool_execution_start") {
1861
- sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } });
2030
+ sendChunk({ x_tool_execution_start: { tool_call_id: String(ev.toolCallId ?? "") } }, null);
1862
2031
  return;
1863
2032
  }
1864
2033
  if (ev.type === "tool_execution_end") {
@@ -1867,7 +2036,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1867
2036
  tool_call_id: String(ev.toolCallId ?? ""),
1868
2037
  content: text,
1869
2038
  ...ev.isError ? { is_error: true } : {}
1870
- } });
2039
+ } }, null);
1871
2040
  return;
1872
2041
  }
1873
2042
  if (ev.type === "agent_end") {
@@ -1879,11 +2048,11 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1879
2048
  const t = inner?.type;
1880
2049
  if (t === "text_delta") {
1881
2050
  ensureRole();
1882
- sendChunk({ content: inner.delta ?? "" });
2051
+ sendChunk({ content: inner.delta ?? "" }, null);
1883
2052
  return;
1884
2053
  }
1885
2054
  if (t === "thinking_delta") {
1886
- sendChunk({ x_thinking_delta: inner.delta ?? "" });
2055
+ sendChunk({ x_thinking_delta: inner.delta ?? "" }, null);
1887
2056
  return;
1888
2057
  }
1889
2058
  if (t === "toolcall_start") {
@@ -1901,7 +2070,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1901
2070
  name: tc?.name ?? "",
1902
2071
  arguments: ""
1903
2072
  }
1904
- }] });
2073
+ }] }, null);
1905
2074
  return;
1906
2075
  }
1907
2076
  if (t === "toolcall_delta") {
@@ -1910,7 +2079,7 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1910
2079
  sendChunk({ tool_calls: [{
1911
2080
  index: openaiIdx,
1912
2081
  function: { arguments: inner.delta ?? "" }
1913
- }] });
2082
+ }] }, null);
1914
2083
  }
1915
2084
  });
1916
2085
  try {
@@ -1922,6 +2091,22 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1922
2091
  postPrompt
1923
2092
  });
1924
2093
  if (capturedModelError !== null) {
2094
+ const billingBlocked = parseBillingBlockedProviderSignal(capturedModelError);
2095
+ if (billingBlocked) {
2096
+ log.warn({
2097
+ event: "billing_blocked",
2098
+ chatcmpl_id: sessionId,
2099
+ source: "stop_reason_error"
2100
+ }, "forwarding platform billing block to client");
2101
+ emitBillingBlocked({
2102
+ writer,
2103
+ sessionId,
2104
+ created,
2105
+ billingBlocked
2106
+ });
2107
+ writer.write("data: [DONE]\n\n");
2108
+ return;
2109
+ }
1925
2110
  const providerError = toModelProviderError(new Error(capturedModelError));
1926
2111
  log.warn({
1927
2112
  event: "model_provider_error",
@@ -1929,13 +2114,34 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1929
2114
  code: providerError.code,
1930
2115
  upstream_status: providerError.upstreamStatus,
1931
2116
  provider: providerError.provider,
2117
+ retry_attempts: autoRetry.attempts(),
1932
2118
  source: "stop_reason_error"
1933
2119
  }, "forwarding model-provider error to client");
1934
2120
  emitModelProviderError({
1935
2121
  writer,
1936
2122
  sessionId,
1937
2123
  created,
1938
- providerError
2124
+ providerError,
2125
+ retryAttempts: autoRetry.attempts()
2126
+ });
2127
+ writer.write("data: [DONE]\n\n");
2128
+ } else if (sawModelErrorStop) {
2129
+ const providerError = toModelProviderError(/* @__PURE__ */ new Error("model provider call failed without an error message"));
2130
+ log.warn({
2131
+ event: "model_provider_error",
2132
+ chatcmpl_id: sessionId,
2133
+ code: providerError.code,
2134
+ upstream_status: providerError.upstreamStatus,
2135
+ provider: providerError.provider,
2136
+ retry_attempts: autoRetry.attempts(),
2137
+ source: "stop_reason_error_no_message"
2138
+ }, "forwarding model-provider error to client");
2139
+ emitModelProviderError({
2140
+ writer,
2141
+ sessionId,
2142
+ created,
2143
+ providerError,
2144
+ retryAttempts: autoRetry.attempts()
1939
2145
  });
1940
2146
  writer.write("data: [DONE]\n\n");
1941
2147
  } else {
@@ -1957,19 +2163,32 @@ function runStream$1({ session, sm, prompt, images, sessionId, created, reqStart
1957
2163
  err
1958
2164
  }, "chat error");
1959
2165
  if (isModelProviderError(err)) {
2166
+ const billingBlocked = parseBillingBlockedProviderSignal(err instanceof Error ? err.message : String(err));
2167
+ if (billingBlocked) {
2168
+ emitBillingBlocked({
2169
+ writer,
2170
+ sessionId,
2171
+ created,
2172
+ billingBlocked
2173
+ });
2174
+ writer.write("data: [DONE]\n\n");
2175
+ return;
2176
+ }
1960
2177
  const providerError = toModelProviderError(err);
1961
2178
  log.warn({
1962
2179
  event: "model_provider_error",
1963
2180
  chatcmpl_id: sessionId,
1964
2181
  code: providerError.code,
1965
2182
  upstream_status: providerError.upstreamStatus,
1966
- provider: providerError.provider
2183
+ provider: providerError.provider,
2184
+ retry_attempts: autoRetry.attempts()
1967
2185
  }, "forwarding model-provider error to client");
1968
2186
  emitModelProviderError({
1969
2187
  writer,
1970
2188
  sessionId,
1971
2189
  created,
1972
- providerError
2190
+ providerError,
2191
+ retryAttempts: autoRetry.attempts()
1973
2192
  });
1974
2193
  } else sendChunk({ content: `\n[error: ${err?.message ?? err}]` }, "stop");
1975
2194
  writer.write("data: [DONE]\n\n");
@@ -2010,7 +2229,7 @@ function emitSessionMessagesTrailer({ writer, sm, baselineMessageCount, sessionI
2010
2229
  * cleanly closes the stream for any OpenAI-shaped reader that ignores the
2011
2230
  * extension field.
2012
2231
  */
2013
- function emitModelProviderError({ writer, sessionId, created, providerError }) {
2232
+ function emitModelProviderError({ writer, sessionId, created, providerError, retryAttempts }) {
2014
2233
  const chunk = {
2015
2234
  id: sessionId,
2016
2235
  object: "chat.completion.chunk",
@@ -2025,7 +2244,25 @@ function emitModelProviderError({ writer, sessionId, created, providerError }) {
2025
2244
  message: providerError.message,
2026
2245
  provider: providerError.provider,
2027
2246
  type: providerError.type,
2028
- upstream_status: providerError.upstreamStatus
2247
+ upstream_status: providerError.upstreamStatus,
2248
+ retry_attempts: retryAttempts
2249
+ }
2250
+ };
2251
+ writer.write(`data: ${JSON.stringify(chunk)}\n\n`);
2252
+ }
2253
+ function emitBillingBlocked({ writer, sessionId, created, billingBlocked }) {
2254
+ const chunk = {
2255
+ id: sessionId,
2256
+ object: "chat.completion.chunk",
2257
+ created,
2258
+ choices: [{
2259
+ index: 0,
2260
+ delta: {},
2261
+ finish_reason: "stop"
2262
+ }],
2263
+ x_billing_blocked: {
2264
+ block_reason: billingBlocked.blockReason,
2265
+ message: billingBlocked.message
2029
2266
  }
2030
2267
  };
2031
2268
  writer.write(`data: ${JSON.stringify(chunk)}\n\n`);
@@ -2127,7 +2364,6 @@ const steerRequestSchema = z.object({
2127
2364
  mode: z.enum(["steer", "follow-up"]).default("steer"),
2128
2365
  text: z.string().optional()
2129
2366
  }).refine((v) => v.content !== void 0 || v.text !== void 0, { message: "content or text is required" });
2130
- z.string().uuid();
2131
2367
  async function handleSteer(request, sessionId, registry, pendingSteerIds) {
2132
2368
  const { log } = requestLogger(request);
2133
2369
  const session = registry.get(sessionId);
@@ -2135,14 +2371,14 @@ async function handleSteer(request, sessionId, registry, pendingSteerIds) {
2135
2371
  let body;
2136
2372
  try {
2137
2373
  body = JSON.parse(await request.text());
2138
- } catch {
2374
+ } catch (_err) {
2139
2375
  return jsonError(400, "invalid json");
2140
2376
  }
2141
2377
  const parsed = steerRequestSchema.safeParse(body);
2142
2378
  if (!parsed.success) return jsonError(400, parsed.error.message);
2143
2379
  const { mode } = parsed.data;
2144
2380
  const { text, images } = await extractContent(parsed.data.content ?? parsed.data.text ?? "", {
2145
- userAgent: request.headers.get("user-agent") ?? void 0,
2381
+ userAgent: request.headers.get("user-agent") ?? null,
2146
2382
  log
2147
2383
  });
2148
2384
  if (!text && images.length === 0) return jsonError(400, "steer has no content");
@@ -2295,7 +2531,10 @@ async function extractImagesFromContent(content, log) {
2295
2531
  for (const part of content) if (part?.type === "input_image") {
2296
2532
  const url = part.image_url ?? part.url;
2297
2533
  if (typeof url === "string" && /^https?:\/\//i.test(url)) try {
2298
- images.push(await fetchImageAsBase64({ url }));
2534
+ images.push(await fetchImageAsBase64({
2535
+ url,
2536
+ userAgent: null
2537
+ }));
2299
2538
  } catch (err) {
2300
2539
  log.warn({
2301
2540
  event: "image_fetch_failed",
@@ -2316,7 +2555,7 @@ async function handleResponses(request, ctx, registry) {
2316
2555
  let parsed;
2317
2556
  try {
2318
2557
  parsed = JSON.parse(await request.text());
2319
- } catch {
2558
+ } catch (_err) {
2320
2559
  return jsonError(400, "invalid json");
2321
2560
  }
2322
2561
  const { input, stream = false, model: modelInput, instructions, x_model: modelSpecInput } = parsed ?? {};
@@ -2388,6 +2627,7 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
2388
2627
  const tcByContentIdx = /* @__PURE__ */ new Map();
2389
2628
  let nextOutputIndex = 0;
2390
2629
  let capturedModelError = null;
2630
+ const autoRetry = createAutoRetryObserver(log);
2391
2631
  const send = (event) => {
2392
2632
  writer.write(`data: ${JSON.stringify(event)}\n\n`);
2393
2633
  };
@@ -2396,7 +2636,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
2396
2636
  event: "model_provider_error",
2397
2637
  code: providerError.code,
2398
2638
  upstream_status: providerError.upstreamStatus,
2399
- provider: providerError.provider
2639
+ provider: providerError.provider,
2640
+ retry_attempts: autoRetry.attempts()
2400
2641
  }, "forwarding model-provider error to client");
2401
2642
  send({
2402
2643
  type: "response.failed",
@@ -2410,7 +2651,8 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
2410
2651
  x_model_provider_error: {
2411
2652
  provider: providerError.provider,
2412
2653
  type: providerError.type,
2413
- upstream_status: providerError.upstreamStatus
2654
+ upstream_status: providerError.upstreamStatus,
2655
+ retry_attempts: autoRetry.attempts()
2414
2656
  }
2415
2657
  }
2416
2658
  }
@@ -2446,8 +2688,9 @@ function runStream({ session, prompt, images, responseId, sessionId, modelName,
2446
2688
  };
2447
2689
  session.subscribe((event) => {
2448
2690
  const ev = event;
2691
+ autoRetry.observe(ev);
2449
2692
  const endedMessage = ev.message ?? (Array.isArray(ev.messages) ? ev.messages[ev.messages.length - 1] : null);
2450
- if (endedMessage?.stopReason === "error" && typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0) capturedModelError = endedMessage.errorMessage;
2693
+ if (endedMessage?.stopReason === "error") capturedModelError = typeof endedMessage.errorMessage === "string" && endedMessage.errorMessage.length > 0 ? endedMessage.errorMessage : "model provider call failed without an error message";
2451
2694
  if (ev.type !== "message_update") return;
2452
2695
  const inner = ev.assistantMessageEvent;
2453
2696
  if (!inner) return;
@@ -2851,4 +3094,4 @@ function chainMiddleware(middlewares) {
2851
3094
  };
2852
3095
  }
2853
3096
  //#endregion
2854
- export { DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, PI_AGENT_DIR, VALID_THINKING_LEVELS, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
3097
+ export { BILLING_BLOCK_REASONS, DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_THINKING_LEVEL, KNOWN_PI_PROVIDERS, PI_AGENT_DIR, VALID_THINKING_LEVELS, buildAgentCard, buildModelFromSpec, chainMiddleware, composeHandlers, createAgentExecutor, createMapSessionRegistry, createPrewarm, createProtocolHandlers, createProtocols, deriveTraceContext, getCurrentTraceparent, logger, modelSpecSchema, mountAt, newTraceContext, parseModelSpec, parseTraceId, requestHeaders, requestLogger, requestUrl, resolveRequestModel, runInTraceContext, setCurrentTraceparent, webHandlerToMiddleware, withSpecApiKey };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@skydiveai/pi-server",
3
- "version": "0.1.0-beta.267",
3
+ "version": "0.1.0-beta.2728",
4
4
  "homepage": "https://skydive.com",
5
5
  "license": "MIT",
6
6
  "author": "Create, Inc.",
@@ -17,20 +17,13 @@
17
17
  },
18
18
  "publishConfig": {
19
19
  "access": "public",
20
- "exports": {
21
- ".": {
22
- "types": "./dist/index.d.mts",
23
- "default": "./dist/index.mjs"
24
- }
25
- },
26
20
  "registry": "https://registry.npmjs.org"
27
21
  },
28
22
  "scripts": {
29
23
  "build": "tsdown",
30
24
  "typecheck": "tsgo --noEmit",
31
25
  "test:unit": "vitest run --passWithNoTests",
32
- "test:ci": "vitest run --coverage --coverage.reporter=lcovonly --reporter=default --reporter=github-actions --minWorkers=1 --maxWorkers=2 --passWithNoTests",
33
- "publish:system-artifacts": "doppler run --preserve-env -- node ../../scripts/anyone/publish-system-artifact.mjs"
26
+ "test:ci": "vitest run --coverage --coverage.reporter=lcovonly --reporter=default --reporter=github-actions --minWorkers=1 --maxWorkers=2 --passWithNoTests"
34
27
  },
35
28
  "dependencies": {
36
29
  "@a2a-js/sdk": "^0.3.13",