@gajae-code/ai 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/dist/types/model-thinking.d.ts +2 -1
  3. package/dist/types/providers/azure-openai-responses.d.ts +1 -1
  4. package/dist/types/providers/ollama.d.ts +1 -1
  5. package/dist/types/providers/openai-chat-server-schema.d.ts +1 -0
  6. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
  7. package/dist/types/providers/openai-codex-responses.d.ts +1 -1
  8. package/dist/types/providers/openai-completions.d.ts +1 -1
  9. package/dist/types/providers/openai-responses.d.ts +1 -1
  10. package/dist/types/providers/pi-native-server.d.ts +1 -1
  11. package/dist/types/types.d.ts +2 -0
  12. package/dist/types/utils.d.ts +1 -1
  13. package/package.json +2 -2
  14. package/src/model-thinking.ts +48 -6
  15. package/src/models.json +115 -2
  16. package/src/provider-models/descriptors.ts +4 -4
  17. package/src/provider-models/openai-compat.ts +1 -1
  18. package/src/providers/amazon-bedrock.ts +1 -0
  19. package/src/providers/anthropic.ts +2 -2
  20. package/src/providers/azure-openai-responses.ts +1 -1
  21. package/src/providers/google-gemini-headers.ts +1 -1
  22. package/src/providers/ollama.ts +2 -1
  23. package/src/providers/openai-chat-server-schema.ts +1 -1
  24. package/src/providers/openai-chat-server.ts +8 -1
  25. package/src/providers/openai-codex/request-transformer.ts +1 -1
  26. package/src/providers/openai-codex-responses.ts +1 -1
  27. package/src/providers/openai-completions-compat.ts +3 -1
  28. package/src/providers/openai-completions.ts +4 -3
  29. package/src/providers/openai-responses-server.ts +8 -1
  30. package/src/providers/openai-responses.ts +5 -9
  31. package/src/providers/pi-native-client.ts +2 -3
  32. package/src/providers/pi-native-server.ts +1 -1
  33. package/src/stream.ts +4 -1
  34. package/src/types.ts +2 -0
  35. package/src/utils.ts +3 -1
package/CHANGELOG.md CHANGED
@@ -2,6 +2,20 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.4.0] - 2026-06-06
6
+
7
+ ### Added
8
+
9
+ - Added minimax-m3 model support across MiniMax providers.
10
+ - Honored the `GJC_CACHE_RETENTION` environment variable and `cacheRetention` model config so hosts can control provider prompt-cache retention (#379/#381).
11
+ - Added an Opus max reasoning preset to the model thinking presets (#372).
12
+ - Refreshed the generated models schema for the new model/config surface (#382).
13
+
14
+ ### Changed
15
+
16
+ - Pinned the OpenAI Codex provider default to GPT-5.5 at `xhigh` reasoning effort (#352). This changes the default model and effort for Codex users (latency/cost/quality impact) and is a behavior change, not an API break; pass an explicit model/effort to override.
17
+ - Bumped the spoofed Gemini CLI user-agent version to 0.45.2 to track the upstream release.
18
+
5
19
  ## [0.3.0] - 2026-06-03
6
20
 
7
21
  ### Added
@@ -5,7 +5,8 @@ export declare const enum Effort {
5
5
  Low = "low",
6
6
  Medium = "medium",
7
7
  High = "high",
8
- XHigh = "xhigh"
8
+ XHigh = "xhigh",
9
+ Max = "max"
9
10
  }
10
11
  export declare const THINKING_EFFORTS: readonly Effort[];
11
12
  /**
@@ -1,6 +1,6 @@
1
1
  import type { ServiceTier, StreamFunction, StreamOptions, ToolChoice } from "../types";
2
2
  export interface AzureOpenAIResponsesOptions extends StreamOptions {
3
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
3
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
4
4
  reasoningSummary?: "auto" | "detailed" | "concise" | null;
5
5
  azureApiVersion?: string;
6
6
  azureResourceName?: string;
@@ -1,6 +1,6 @@
1
1
  import type { StreamFunction, StreamOptions, ToolChoice } from "../types";
2
2
  export interface OllamaChatOptions extends StreamOptions {
3
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
3
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
4
4
  toolChoice?: ToolChoice;
5
5
  }
6
6
  export declare const streamOllama: StreamFunction<"ollama-chat">;
@@ -777,6 +777,7 @@ export declare const openaiChatRequestSchema: z.ZodObject<{
777
777
  reasoning_effort: z.ZodOptional<z.ZodEnum<{
778
778
  high: "high";
779
779
  low: "low";
780
+ max: "max";
780
781
  medium: "medium";
781
782
  minimal: "minimal";
782
783
  xhigh: "xhigh";
@@ -1,6 +1,6 @@
1
1
  import type { Api, Model } from "../../types";
2
2
  export interface ReasoningConfig {
3
- effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
3
+ effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
4
4
  summary?: "auto" | "concise" | "detailed";
5
5
  }
6
6
  export interface CodexRequestOptions {
@@ -1,7 +1,7 @@
1
1
  import type { ResponseInput } from "openai/resources/responses/responses";
2
2
  import { type Context, type Model, type ProviderSessionState, type ServiceTier, type StreamFunction, type StreamOptions, type Tool, type ToolChoice } from "../types";
3
3
  export interface OpenAICodexResponsesOptions extends StreamOptions {
4
- reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
4
+ reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
5
5
  reasoningSummary?: "auto" | "concise" | "detailed" | null;
6
6
  textVerbosity?: "low" | "medium" | "high";
7
7
  include?: string[];
@@ -17,7 +17,7 @@ import { type ResolvedOpenAICompat } from "./openai-completions-compat";
17
17
  export declare function isOpenAICompletionsProgressChunk(chunk: unknown): boolean;
18
18
  export interface OpenAICompletionsOptions extends StreamOptions {
19
19
  toolChoice?: ToolChoice;
20
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
20
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
21
21
  /** Force-disable reasoning where supported, or request the lowest effort on generic effort endpoints. */
22
22
  disableReasoning?: boolean;
23
23
  serviceTier?: ServiceTier;
@@ -3,7 +3,7 @@ import type { Model, ServiceTier, StreamFunction, StreamOptions, Tool, ToolChoic
3
3
  import { type OpenAIResponsesToolChoice } from "../utils/tool-choice";
4
4
  export declare function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined;
5
5
  export interface OpenAIResponsesOptions extends StreamOptions {
6
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
6
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
7
7
  reasoningSummary?: "auto" | "detailed" | "concise" | null;
8
8
  serviceTier?: ServiceTier;
9
9
  toolChoice?: ToolChoice;
@@ -4,7 +4,7 @@
4
4
  * Where the OpenAI / Anthropic / Responses route modules translate foreign
5
5
  * wire shapes through pi-ai's canonical {@link Context}, this module accepts
6
6
  * the canonical shape *directly* — for clients that already speak pi-ai
7
- * (containerized gjc, the swarm extension, robogjc's sidecar auth-gateway).
7
+ * (containerized gjc and robogjc's sidecar auth-gateway).
8
8
  * Skipping the wire-format → Context → wire-format round-trip cuts
9
9
  * per-request CPU but, more importantly, avoids the quantization that those
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
@@ -751,6 +751,8 @@ export interface Model<TApi extends Api = any> {
751
751
  wireModelId?: string;
752
752
  /** Declarative request shaping for OpenAI-compatible proxy providers. */
753
753
  requestTransform?: ModelRequestTransform;
754
+ /** Default prompt-cache retention preference for this model when the request omits one. */
755
+ cacheRetention?: CacheRetention;
754
756
  /** Provider-assigned priority value (lower = higher priority). */
755
757
  priority?: number;
756
758
  /** Canonical thinking capability metadata for this model. */
@@ -22,7 +22,7 @@ export declare function getOpenAIResponsesHistoryPayload(providerPayload: Provid
22
22
  export declare function getOpenAIResponsesHistoryItems(providerPayload: ProviderPayload | undefined, currentProvider: string, fallbackProvider?: string): Array<Record<string, unknown>> | undefined;
23
23
  /**
24
24
  * Resolve cache retention preference.
25
- * Defaults to "short" and uses PI_CACHE_RETENTION for backward compatibility.
25
+ * Defaults to "short" and uses GJC_CACHE_RETENTION, with PI_CACHE_RETENTION as a legacy fallback.
26
26
  */
27
27
  export declare function resolveCacheRetention(cacheRetention?: CacheRetention): CacheRetention;
28
28
  export declare function isAnthropicOAuthToken(key: string): boolean;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.3.1",
4
+ "version": "0.4.0",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.3.1",
46
+ "@gajae-code/utils": "0.4.0",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -8,6 +8,7 @@ export const enum Effort {
8
8
  Medium = "medium",
9
9
  High = "high",
10
10
  XHigh = "xhigh",
11
+ Max = "max",
11
12
  }
12
13
 
13
14
  export const THINKING_EFFORTS: readonly Effort[] = [
@@ -16,6 +17,7 @@ export const THINKING_EFFORTS: readonly Effort[] = [
16
17
  Effort.Medium,
17
18
  Effort.High,
18
19
  Effort.XHigh,
20
+ Effort.Max,
19
21
  ];
20
22
 
21
23
  const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
@@ -26,9 +28,26 @@ const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [
26
28
  Effort.High,
27
29
  Effort.XHigh,
28
30
  ];
31
+ const DEFAULT_REASONING_EFFORTS_WITH_MAX: readonly Effort[] = [
32
+ Effort.Minimal,
33
+ Effort.Low,
34
+ Effort.Medium,
35
+ Effort.High,
36
+ Effort.Max,
37
+ ];
38
+ const DEFAULT_REASONING_EFFORTS_WITH_XHIGH_AND_MAX: readonly Effort[] = [
39
+ Effort.Minimal,
40
+ Effort.Low,
41
+ Effort.Medium,
42
+ Effort.High,
43
+ Effort.XHigh,
44
+ Effort.Max,
45
+ ];
29
46
  const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
30
47
  const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
31
48
  const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
49
+ const GPT_5_5_DEFAULT_EFFORT = Effort.XHigh;
50
+
32
51
  const GPT_5_1_CODEX_MINI_EFFORTS: readonly Effort[] = [Effort.Medium, Effort.High];
33
52
  const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
34
53
 
@@ -281,6 +300,7 @@ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
281
300
  return "MEDIUM";
282
301
  case Effort.High:
283
302
  case Effort.XHigh:
303
+ case Effort.Max:
284
304
  return "HIGH";
285
305
  }
286
306
  }
@@ -299,10 +319,8 @@ export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
299
319
  case Effort.High:
300
320
  return "high";
301
321
  case Effort.XHigh:
302
- // Opus 4.7+ introduced a distinct "xhigh" effort level (between "high" and "max").
303
- // The Anthropic docs scope this to the Messages API only, so Bedrock Converse and
304
- // older adaptive-thinking Opus 4.6 models keep the legacy "max" alias.
305
- return anthropicModelHasRealXHighEffort(model) ? "xhigh" : "max";
322
+ case Effort.Max:
323
+ return effort === Effort.XHigh ? "xhigh" : "max";
306
324
  }
307
325
  }
308
326
 
@@ -433,6 +451,17 @@ function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel
433
451
  }
434
452
  }
435
453
 
454
+ function inferDefaultEffort<TApi extends Api>(model: ApiModel<TApi>, parsedModel: ParsedModel): Effort | undefined {
455
+ if (
456
+ parsedModel.family === "openai" &&
457
+ model.provider === "openai-codex" &&
458
+ semverEqual(parsedModel.version, "5.5")
459
+ ) {
460
+ return GPT_5_5_DEFAULT_EFFORT;
461
+ }
462
+ return undefined;
463
+ }
464
+
436
465
  function inferModelThinking<TApi extends Api>(model: ApiModel<TApi>): ThinkingConfig {
437
466
  const parsedModel = parseKnownModel(model.id);
438
467
  const efforts = inferSupportedEfforts(parsedModel, model);
@@ -446,6 +475,10 @@ function inferModelThinking<TApi extends Api>(model: ApiModel<TApi>): ThinkingCo
446
475
  minLevel,
447
476
  maxLevel,
448
477
  };
478
+ const defaultLevel = inferDefaultEffort(model, parsedModel);
479
+ if (defaultLevel && efforts.includes(defaultLevel)) {
480
+ config.defaultLevel = defaultLevel;
481
+ }
449
482
  // Encode explicit levels only when the inferred set has gaps the min..max range cannot represent.
450
483
  const minIndex = THINKING_EFFORTS.indexOf(minLevel);
451
484
  const maxIndex = THINKING_EFFORTS.indexOf(maxLevel);
@@ -466,7 +499,13 @@ function normalizeThinkingConfig(thinking: ThinkingConfig | undefined): Thinking
466
499
  function thinkingsEqual(left: ThinkingConfig | undefined, right: ThinkingConfig | undefined): boolean {
467
500
  if (left === right) return true;
468
501
  if (!left || !right) return false;
469
- if (left.mode !== right.mode || left.minLevel !== right.minLevel || left.maxLevel !== right.maxLevel) return false;
502
+ if (
503
+ left.mode !== right.mode ||
504
+ left.minLevel !== right.minLevel ||
505
+ left.maxLevel !== right.maxLevel ||
506
+ left.defaultLevel !== right.defaultLevel
507
+ )
508
+ return false;
470
509
  const leftLevels = left.levels;
471
510
  const rightLevels = right.levels;
472
511
  if (leftLevels === rightLevels) return true;
@@ -525,7 +564,10 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
525
564
  (model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
526
565
  semverGte(parsedModel.version, "4.6")
527
566
  ) {
528
- return parsedModel.kind === "opus" ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
567
+ if (parsedModel.kind !== "opus") return DEFAULT_REASONING_EFFORTS;
568
+ return anthropicModelHasRealXHighEffort(model)
569
+ ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH_AND_MAX
570
+ : DEFAULT_REASONING_EFFORTS_WITH_MAX;
529
571
  }
530
572
  return inferFallbackEfforts(model);
531
573
  }
package/src/models.json CHANGED
@@ -34657,6 +34657,31 @@
34657
34657
  "minLevel": "minimal",
34658
34658
  "maxLevel": "xhigh"
34659
34659
  }
34660
+ },
34661
+ "minimax-m3": {
34662
+ "id": "minimax-m3",
34663
+ "name": "MiniMax M3",
34664
+ "api": "anthropic-messages",
34665
+ "provider": "minimax",
34666
+ "baseUrl": "https://api.minimax.io/anthropic",
34667
+ "reasoning": true,
34668
+ "input": [
34669
+ "text",
34670
+ "image"
34671
+ ],
34672
+ "cost": {
34673
+ "input": 0.6,
34674
+ "output": 2.4,
34675
+ "cacheRead": 0.12,
34676
+ "cacheWrite": 0
34677
+ },
34678
+ "contextWindow": 512000,
34679
+ "maxTokens": 128000,
34680
+ "thinking": {
34681
+ "mode": "budget",
34682
+ "minLevel": "minimal",
34683
+ "maxLevel": "xhigh"
34684
+ }
34660
34685
  }
34661
34686
  },
34662
34687
  "minimax-cn": {
@@ -34827,6 +34852,31 @@
34827
34852
  "minLevel": "minimal",
34828
34853
  "maxLevel": "xhigh"
34829
34854
  }
34855
+ },
34856
+ "minimax-m3": {
34857
+ "id": "minimax-m3",
34858
+ "name": "MiniMax M3",
34859
+ "api": "anthropic-messages",
34860
+ "provider": "minimax-cn",
34861
+ "baseUrl": "https://api.minimaxi.com/anthropic",
34862
+ "reasoning": true,
34863
+ "input": [
34864
+ "text",
34865
+ "image"
34866
+ ],
34867
+ "cost": {
34868
+ "input": 0.6,
34869
+ "output": 2.4,
34870
+ "cacheRead": 0.12,
34871
+ "cacheWrite": 0
34872
+ },
34873
+ "contextWindow": 512000,
34874
+ "maxTokens": 128000,
34875
+ "thinking": {
34876
+ "mode": "budget",
34877
+ "minLevel": "minimal",
34878
+ "maxLevel": "xhigh"
34879
+ }
34830
34880
  }
34831
34881
  },
34832
34882
  "minimax-code": {
@@ -35069,6 +35119,37 @@
35069
35119
  "minLevel": "minimal",
35070
35120
  "maxLevel": "high"
35071
35121
  }
35122
+ },
35123
+ "minimax-m3": {
35124
+ "id": "minimax-m3",
35125
+ "name": "MiniMax M3",
35126
+ "api": "openai-completions",
35127
+ "provider": "minimax-code",
35128
+ "baseUrl": "https://api.minimax.io/v1",
35129
+ "reasoning": true,
35130
+ "input": [
35131
+ "text",
35132
+ "image"
35133
+ ],
35134
+ "cost": {
35135
+ "input": 0,
35136
+ "output": 0,
35137
+ "cacheRead": 0,
35138
+ "cacheWrite": 0
35139
+ },
35140
+ "contextWindow": 512000,
35141
+ "maxTokens": 128000,
35142
+ "compat": {
35143
+ "supportsStore": false,
35144
+ "supportsDeveloperRole": false,
35145
+ "supportsReasoningEffort": false,
35146
+ "reasoningContentField": "reasoning_content"
35147
+ },
35148
+ "thinking": {
35149
+ "mode": "effort",
35150
+ "minLevel": "minimal",
35151
+ "maxLevel": "high"
35152
+ }
35072
35153
  }
35073
35154
  },
35074
35155
  "minimax-code-cn": {
@@ -35311,6 +35392,37 @@
35311
35392
  "minLevel": "minimal",
35312
35393
  "maxLevel": "high"
35313
35394
  }
35395
+ },
35396
+ "minimax-m3": {
35397
+ "id": "minimax-m3",
35398
+ "name": "MiniMax M3",
35399
+ "api": "openai-completions",
35400
+ "provider": "minimax-code-cn",
35401
+ "baseUrl": "https://api.minimaxi.com/v1",
35402
+ "reasoning": true,
35403
+ "input": [
35404
+ "text",
35405
+ "image"
35406
+ ],
35407
+ "cost": {
35408
+ "input": 0,
35409
+ "output": 0,
35410
+ "cacheRead": 0,
35411
+ "cacheWrite": 0
35412
+ },
35413
+ "contextWindow": 512000,
35414
+ "maxTokens": 128000,
35415
+ "compat": {
35416
+ "supportsStore": false,
35417
+ "supportsDeveloperRole": false,
35418
+ "supportsReasoningEffort": false,
35419
+ "reasoningContentField": "reasoning_content"
35420
+ },
35421
+ "thinking": {
35422
+ "mode": "effort",
35423
+ "minLevel": "minimal",
35424
+ "maxLevel": "high"
35425
+ }
35314
35426
  }
35315
35427
  },
35316
35428
  "mistral": {
@@ -53159,7 +53271,8 @@
53159
53271
  "thinking": {
53160
53272
  "mode": "effort",
53161
53273
  "minLevel": "low",
53162
- "maxLevel": "xhigh"
53274
+ "maxLevel": "xhigh",
53275
+ "defaultLevel": "xhigh"
53163
53276
  },
53164
53277
  "applyPatchToolType": "freeform",
53165
53278
  "contextPromotionTarget": "openai-codex/gpt-5.4"
@@ -74638,4 +74751,4 @@
74638
74751
  }
74639
74752
  }
74640
74753
  }
74641
- }
74754
+ }
@@ -301,9 +301,9 @@ export const DEFAULT_MODEL_PER_PROVIDER: Record<KnownProvider, string> = {
301
301
  "google-antigravity": "gemini-3-pro-high",
302
302
  "google-gemini-cli": "gemini-2.5-pro",
303
303
  "google-vertex": "gemini-3-pro-preview",
304
- minimax: "MiniMax-M2.5",
305
- "minimax-code": "MiniMax-M2.5",
306
- "minimax-code-cn": "MiniMax-M2.5",
307
- "openai-codex": "gpt-5.4",
304
+ minimax: "minimax-m3",
305
+ "minimax-code": "minimax-m3",
306
+ "minimax-code-cn": "minimax-m3",
307
+ "openai-codex": "gpt-5.5",
308
308
  "gitlab-duo": "duo-chat-sonnet-4-5",
309
309
  } as Record<KnownProvider, string>;
@@ -2135,7 +2135,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
2135
2135
  // sending unsupported effort strings.
2136
2136
  supportsDeveloperRole: false,
2137
2137
  supportsReasoningEffort: true,
2138
- reasoningEffortMap: { minimal: "high", low: "high", medium: "high", high: "high", xhigh: "max" },
2138
+ reasoningEffortMap: { minimal: "high", low: "high", medium: "high", high: "high", xhigh: "max", max: "max" },
2139
2139
  maxTokensField: "max_tokens",
2140
2140
  // DeepSeek V4 thinking mode rejects the `tool_choice` control parameter.
2141
2141
  // Tool calls still work without it; the API defaults to auto when tools exist.
@@ -791,6 +791,7 @@ function buildAdditionalModelRequestFields(
791
791
  medium: 8192,
792
792
  high: 16384,
793
793
  xhigh: 32768,
794
+ max: 32768,
794
795
  };
795
796
  const budget = options.thinkingBudgets?.[level] ?? defaultBudgets[level];
796
797
 
@@ -2001,8 +2001,8 @@ function buildParams(
2001
2001
  }
2002
2002
  params.thinking = adaptive as typeof params.thinking;
2003
2003
  if (effort) {
2004
- // SDK's OutputConfig.effort type is not yet widened to include the new "xhigh"
2005
- // level introduced with Anthropic model Opus 4.7. Cast until the SDK catches up.
2004
+ // SDK OutputConfig.effort typings may lag Anthropic's adaptive effort literals.
2005
+ // Cast so newly supported levels can pass through before the SDK catches up.
2006
2006
  params.output_config = { effort } as typeof params.output_config;
2007
2007
  }
2008
2008
  } else {
@@ -70,7 +70,7 @@ function resolveDeploymentName(model: Model<"azure-openai-responses">, options?:
70
70
 
71
71
  // Azure OpenAI Responses-specific options
72
72
  export interface AzureOpenAIResponsesOptions extends StreamOptions {
73
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
73
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
74
74
  reasoningSummary?: "auto" | "detailed" | "concise" | null;
75
75
  azureApiVersion?: string;
76
76
  azureResourceName?: string;
@@ -4,7 +4,7 @@
4
4
  * GeminiCLI/VERSION/MODEL (PLATFORM; ARCH; SURFACE)
5
5
  */
6
6
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
7
- const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.44.1";
7
+ const version = process.env.PI_AI_GEMINI_CLI_VERSION || "0.45.2";
8
8
  const platform = process.platform === "win32" ? "win32" : process.platform;
9
9
  const arch = process.arch === "x64" ? "x64" : process.arch;
10
10
  return `GeminiCLI/${version}/${modelId} (${platform}; ${arch}; terminal)`;
@@ -23,7 +23,7 @@ import { toolWireSchema } from "../utils/schema/wire";
23
23
  import { transformMessages } from "./transform-messages";
24
24
 
25
25
  export interface OllamaChatOptions extends StreamOptions {
26
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
26
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
27
27
  toolChoice?: ToolChoice;
28
28
  }
29
29
 
@@ -95,6 +95,7 @@ function mapReasoning(reasoning: OllamaChatOptions["reasoning"]): boolean | "low
95
95
  return "medium";
96
96
  case "high":
97
97
  case "xhigh":
98
+ case "max":
98
99
  return "high";
99
100
  default:
100
101
  return undefined;
@@ -210,7 +210,7 @@ export const openaiChatRequestSchema = z.object({
210
210
  frequency_penalty: z.number().optional(),
211
211
  logit_bias: z.record(z.string(), z.number()).optional(),
212
212
  user: z.string().optional(),
213
- reasoning_effort: z.enum(["minimal", "low", "medium", "high", "xhigh"]).optional(),
213
+ reasoning_effort: z.enum(["minimal", "low", "medium", "high", "xhigh", "max"]).optional(),
214
214
  parallel_tool_calls: z.boolean().optional(),
215
215
  service_tier: z.enum(["auto", "default", "flex", "scale", "priority"]).optional(),
216
216
  metadata: z.record(z.string(), z.unknown()).optional(),
@@ -33,7 +33,14 @@ export type { ParsedRequest };
33
33
  type ReasoningEffort = NonNullable<ParsedRequest["options"]["reasoning"]>;
34
34
 
35
35
  function isReasoningEffort(value: unknown): value is ReasoningEffort {
36
- return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh";
36
+ return (
37
+ value === "minimal" ||
38
+ value === "low" ||
39
+ value === "medium" ||
40
+ value === "high" ||
41
+ value === "xhigh" ||
42
+ value === "max"
43
+ );
37
44
  }
38
45
 
39
46
  function isServiceTier(value: unknown): value is ResolvedServiceTier {
@@ -3,7 +3,7 @@ import { requireSupportedEffort } from "../../model-thinking";
3
3
  import type { Api, Model } from "../../types";
4
4
 
5
5
  export interface ReasoningConfig {
6
- effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
6
+ effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
7
7
  summary?: "auto" | "concise" | "detailed";
8
8
  }
9
9
 
@@ -73,7 +73,7 @@ import {
73
73
  import { transformMessages } from "./transform-messages";
74
74
 
75
75
  export interface OpenAICodexResponsesOptions extends StreamOptions {
76
- reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
76
+ reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
77
77
  reasoningSummary?: "auto" | "concise" | "detailed" | null;
78
78
  textVerbosity?: "low" | "medium" | "high";
79
79
  include?: string[];
@@ -1,6 +1,6 @@
1
1
  import type { Model, OpenAICompat } from "../types";
2
2
 
3
- type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
3
+ type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
4
4
  type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
5
5
 
6
6
  export type ResolvedOpenAICompat = Required<
@@ -162,6 +162,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
162
162
  medium: "default",
163
163
  high: "default",
164
164
  xhigh: "default",
165
+ max: "default",
165
166
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
166
167
  : isDeepseekFamily && model.reasoning
167
168
  ? ({
@@ -170,6 +171,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
170
171
  medium: "high",
171
172
  high: "high",
172
173
  xhigh: "max",
174
+ max: "max",
173
175
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
174
176
  : isFireworks
175
177
  ? ({
@@ -249,13 +249,13 @@ export function isOpenAICompletionsProgressChunk(chunk: unknown): boolean {
249
249
 
250
250
  export interface OpenAICompletionsOptions extends StreamOptions {
251
251
  toolChoice?: ToolChoice;
252
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
252
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
253
253
  /** Force-disable reasoning where supported, or request the lowest effort on generic effort endpoints. */
254
254
  disableReasoning?: boolean;
255
255
  serviceTier?: ServiceTier;
256
256
  }
257
257
 
258
- type OpenAICompletionsParams = OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming & {
258
+ type OpenAICompletionsParams = Omit<OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming, "reasoning_effort"> & {
259
259
  top_k?: number;
260
260
  min_p?: number;
261
261
  repetition_penalty?: number;
@@ -263,6 +263,7 @@ type OpenAICompletionsParams = OpenAI.Chat.Completions.ChatCompletionCreateParam
263
263
  enable_thinking?: boolean;
264
264
  chat_template_kwargs?: { enable_thinking: boolean };
265
265
  reasoning?: { effort?: string } | { enabled: false };
266
+ reasoning_effort?: OpenAICompletionsOptions["reasoning"];
266
267
  provider?: OpenAICompat["openRouterRouting"];
267
268
  providerOptions?: { gateway?: { only?: string[]; order?: string[] } };
268
269
  };
@@ -479,7 +480,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
479
480
  body: params,
480
481
  };
481
482
  const { data, response, request_id } = await client.chat.completions
482
- .create(params, { signal: requestSignal })
483
+ .create(params as OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming, { signal: requestSignal })
483
484
  .withResponse();
484
485
  await notifyProviderResponse(options, response, model, request_id);
485
486
  return data;
@@ -37,7 +37,14 @@ export type { ParsedRequest };
37
37
  // ─── narrow guards ──────────────────────────────────────────────────────────
38
38
 
39
39
  function isReasoningEffort(value: unknown): value is NonNullable<ParsedRequest["options"]["reasoning"]> {
40
- return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh";
40
+ return (
41
+ value === "minimal" ||
42
+ value === "low" ||
43
+ value === "medium" ||
44
+ value === "high" ||
45
+ value === "xhigh" ||
46
+ value === "max"
47
+ );
41
48
  }
42
49
 
43
50
  function isServiceTier(value: unknown): value is NonNullable<ParsedRequest["options"]["serviceTier"]> {
@@ -95,7 +95,7 @@ export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undef
95
95
 
96
96
  // OpenAI Responses-specific options
97
97
  export interface OpenAIResponsesOptions extends StreamOptions {
98
- reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
98
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
99
99
  reasoningSummary?: "auto" | "detailed" | "concise" | null;
100
100
  serviceTier?: ServiceTier;
101
101
  toolChoice?: ToolChoice;
@@ -425,11 +425,9 @@ function createClient(
425
425
  }
426
426
 
427
427
  function getOpenAIResponsesCacheSessionId(
428
- options: Pick<OpenAIResponsesOptions, "cacheRetention" | "sessionId"> | undefined,
428
+ options: Pick<OpenAIResponsesOptions, "sessionId"> | undefined,
429
429
  ): string | undefined {
430
- return resolveCacheRetention(options?.cacheRetention) === "none"
431
- ? undefined
432
- : normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
430
+ return normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
433
431
  }
434
432
 
435
433
  function buildParams(
@@ -470,7 +468,7 @@ function buildParams(
470
468
  }
471
469
  }
472
470
 
473
- const cacheRetention = resolveCacheRetention(options?.cacheRetention);
471
+ const cacheRetention = resolveCacheRetention(options?.cacheRetention ?? model.cacheRetention);
474
472
  const promptCacheKey = getOpenAIResponsesCacheSessionId(options);
475
473
  const params: OpenAIResponsesSamplingParams = {
476
474
  model: model.wireModelId ?? model.id,
@@ -478,9 +476,7 @@ function buildParams(
478
476
  instructions: systemInstructions,
479
477
  stream: true,
480
478
  prompt_cache_key: promptCacheKey,
481
- prompt_cache_retention: promptCacheKey
482
- ? getPromptCacheRetention(resolvedBaseUrl || model.baseUrl, cacheRetention)
483
- : undefined,
479
+ prompt_cache_retention: getPromptCacheRetention(resolvedBaseUrl || model.baseUrl, cacheRetention),
484
480
  store: false,
485
481
  stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
486
482
  };
@@ -11,9 +11,8 @@
11
11
  *
12
12
  * Activated when a {@link Model} has `transport: "pi-native"` set; the
13
13
  * dispatch hook lives in `streamSimple()` (see `../stream.ts`). Used by
14
- * containerized gjc deployments (robogjc slots, the swarm extension) that
15
- * route every LLM call through a credential-holding sidecar so the slot
16
- * itself stays credential-free.
14
+ * containerized gjc deployments (e.g. robogjc slots) that route every LLM call
15
+ * through a credential-holding sidecar so the slot itself stays credential-free.
17
16
  */
18
17
  import { readSseJson } from "@gajae-code/utils";
19
18
  import type {
@@ -4,7 +4,7 @@
4
4
  * Where the OpenAI / Anthropic / Responses route modules translate foreign
5
5
  * wire shapes through pi-ai's canonical {@link Context}, this module accepts
6
6
  * the canonical shape *directly* — for clients that already speak pi-ai
7
- * (containerized gjc, the swarm extension, robogjc's sidecar auth-gateway).
7
+ * (containerized gjc and robogjc's sidecar auth-gateway).
8
8
  * Skipping the wire-format → Context → wire-format round-trip cuts
9
9
  * per-request CPU but, more importantly, avoids the quantization that those
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
package/src/stream.ts CHANGED
@@ -459,6 +459,7 @@ export const ANTHROPIC_THINKING: Record<Effort, number> = {
459
459
  medium: 8192,
460
460
  high: 16384,
461
461
  xhigh: 32768,
462
+ max: 65536,
462
463
  };
463
464
 
464
465
  const GOOGLE_THINKING: Record<Effort, number> = {
@@ -467,6 +468,7 @@ const GOOGLE_THINKING: Record<Effort, number> = {
467
468
  medium: 8192,
468
469
  high: 16384,
469
470
  xhigh: 24575,
471
+ max: 24575,
470
472
  };
471
473
 
472
474
  const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
@@ -475,6 +477,7 @@ const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
475
477
  medium: 8192,
476
478
  high: 16384,
477
479
  xhigh: 16384,
480
+ max: 32768,
478
481
  };
479
482
 
480
483
  function resolveBedrockThinkingBudget(
@@ -559,7 +562,7 @@ function mapOptionsForApi<TApi extends Api>(
559
562
  maxTokens: options?.maxTokens || Math.min(model.maxTokens, 32000),
560
563
  signal: options?.signal,
561
564
  apiKey: apiKey || options?.apiKey,
562
- cacheRetention: options?.cacheRetention,
565
+ cacheRetention: options?.cacheRetention ?? model.cacheRetention,
563
566
  headers: options?.headers,
564
567
  initiatorOverride: options?.initiatorOverride,
565
568
  maxRetryDelayMs: options?.maxRetryDelayMs,
package/src/types.ts CHANGED
@@ -888,6 +888,8 @@ export interface Model<TApi extends Api = any> {
888
888
  wireModelId?: string;
889
889
  /** Declarative request shaping for OpenAI-compatible proxy providers. */
890
890
  requestTransform?: ModelRequestTransform;
891
+ /** Default prompt-cache retention preference for this model when the request omits one. */
892
+ cacheRetention?: CacheRetention;
891
893
  /** Provider-assigned priority value (lower = higher priority). */
892
894
  priority?: number;
893
895
  /** Canonical thinking capability metadata for this model. */
package/src/utils.ts CHANGED
@@ -153,10 +153,12 @@ export function getOpenAIResponsesHistoryItems(
153
153
 
154
154
  /**
155
155
  * Resolve cache retention preference.
156
- * Defaults to "short" and uses PI_CACHE_RETENTION for backward compatibility.
156
+ * Defaults to "short" and uses GJC_CACHE_RETENTION, with PI_CACHE_RETENTION as a legacy fallback.
157
157
  */
158
158
  export function resolveCacheRetention(cacheRetention?: CacheRetention): CacheRetention {
159
159
  if (cacheRetention) return cacheRetention;
160
+ if ($env.GJC_CACHE_RETENTION === "long") return "long";
161
+ if ($env.GJC_CACHE_RETENTION !== undefined) return "short";
160
162
  if ($env.PI_CACHE_RETENTION === "long") return "long";
161
163
  return "short";
162
164
  }