indusagi 0.13.10 → 0.13.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/agent.js CHANGED
@@ -4694,6 +4694,33 @@ var MODELS = {
4694
4694
  contextWindow: 1e6,
4695
4695
  maxTokens: 128e3
4696
4696
  },
4697
+ "gpt-6-astra": {
4698
+ id: "gpt-6-astra",
4699
+ name: "GPT-6 Astra",
4700
+ api: "openai-responses",
4701
+ provider: "openai",
4702
+ baseUrl: "https://api.openai.com/v1",
4703
+ reasoning: true,
4704
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
4705
+ input: ["text", "image"],
4706
+ cost: {
4707
+ input: 10,
4708
+ output: 50,
4709
+ cacheRead: 1,
4710
+ cacheWrite: 12.5
4711
+ },
4712
+ longContextPricing: {
4713
+ threshold: 272e3,
4714
+ multipliers: {
4715
+ input: 2,
4716
+ output: 1.5,
4717
+ cacheRead: 2,
4718
+ cacheWrite: 2
4719
+ }
4720
+ },
4721
+ contextWindow: 105e4,
4722
+ maxTokens: 128e3
4723
+ },
4697
4724
  "gpt-5.6-sol": {
4698
4725
  id: "gpt-5.6-sol",
4699
4726
  name: "GPT-5.6 Sol",
@@ -5156,6 +5183,28 @@ var MODELS = {
5156
5183
  },
5157
5184
  contextWindow: 105e4,
5158
5185
  maxTokens: 128e3
5186
+ },
5187
+ "gpt-6-astra": {
5188
+ id: "gpt-6-astra",
5189
+ name: "GPT-6 Astra",
5190
+ api: "openai-codex-responses",
5191
+ provider: "openai-codex",
5192
+ baseUrl: "https://chatgpt.com/backend-api",
5193
+ reasoning: true,
5194
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
5195
+ input: ["text", "image"],
5196
+ cost: {
5197
+ input: 10,
5198
+ output: 50,
5199
+ cacheRead: 1,
5200
+ cacheWrite: 12.5
5201
+ },
5202
+ longContextPricing: {
5203
+ threshold: 272e3,
5204
+ multipliers: { input: 2, output: 1.5, cacheRead: 2, cacheWrite: 2 }
5205
+ },
5206
+ contextWindow: 105e4,
5207
+ maxTokens: 128e3
5159
5208
  }
5160
5209
  },
5161
5210
  "opencode": {
@@ -12668,9 +12717,9 @@ function getModels(provider) {
12668
12717
  function calculateCost(model, usage) {
12669
12718
  return modelRegistry.calculateCost(model, usage);
12670
12719
  }
12671
- var XHIGH_MODEL_IDS = /* @__PURE__ */ new Set(["gpt-5.1-codex-max", "gpt-5.2", "gpt-5.2-codex"]);
12720
+ var XHIGH_MODEL_IDS = /* @__PURE__ */ new Set(["gpt-5.1-codex-max", "gpt-5.2", "gpt-5.2-codex", "gpt-6-astra"]);
12672
12721
  function supportsXhigh(model) {
12673
- return XHIGH_MODEL_IDS.has(model.id);
12722
+ return model.reasoningEfforts?.includes("xhigh") ?? XHIGH_MODEL_IDS.has(model.id);
12674
12723
  }
12675
12724
 
12676
12725
  // src/facade/ml/types.ts
@@ -13358,12 +13407,12 @@ function buildBaseOptions(model, options, apiKey) {
13358
13407
  return base;
13359
13408
  }
13360
13409
  function clampReasoning(effort) {
13361
- if (effort === "xhigh") return "high";
13410
+ if (effort === "xhigh" || effort === "max") return "high";
13362
13411
  return effort;
13363
13412
  }
13364
13413
  function mapThinkingLevel(level, provider = "clamp-xhigh") {
13365
13414
  if (!level) return void 0;
13366
- if (provider === "supports-xhigh") return level;
13415
+ if (provider === "supports-xhigh") return level === "max" ? "high" : level;
13367
13416
  return clampReasoning(level);
13368
13417
  }
13369
13418
  var RESERVED_ANSWER_TOKENS = 1024;
@@ -13375,8 +13424,10 @@ var REASONING_BUDGET_DEFAULTS = {
13375
13424
  // 2Ki
13376
13425
  medium: BUDGET_UNIT * 8,
13377
13426
  // 8Ki
13378
- high: BUDGET_UNIT * 16
13427
+ high: BUDGET_UNIT * 16,
13379
13428
  // 16Ki
13429
+ max: BUDGET_UNIT * 16
13430
+ // clamped to high for generic token-budget providers
13380
13431
  };
13381
13432
  function adjustMaxTokensForThinking(baseMaxTokens, modelMaxTokens, reasoningLevel, customBudgets) {
13382
13433
  const budgets = { ...REASONING_BUDGET_DEFAULTS, ...customBudgets };
@@ -16575,7 +16626,7 @@ var OpenAIResponsesParamBuilder = class {
16575
16626
  if (this.options?.maxTokens) {
16576
16627
  params.max_output_tokens = this.options.maxTokens;
16577
16628
  }
16578
- if (this.options?.temperature !== void 0) {
16629
+ if (this.model.id !== "gpt-6-astra" && this.options?.temperature !== void 0) {
16579
16630
  params.temperature = this.options.temperature;
16580
16631
  }
16581
16632
  if (this.options?.serviceTier !== void 0) {
@@ -16629,7 +16680,11 @@ var streamSimpleOpenAIResponses = (model, context, options) => {
16629
16680
  throw new Error(`No API key for provider: ${model.provider}`);
16630
16681
  }
16631
16682
  const base = buildBaseOptions(model, options, apiKey);
16632
- const reasoningEffort = mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
16683
+ const requestedEffort = options?.reasoning;
16684
+ const reasoningEffort = model.reasoningEfforts ? requestedEffort !== void 0 && requestedEffort !== "minimal" && model.reasoningEfforts.includes(requestedEffort) ? requestedEffort : void 0 : mapThinkingLevel(
16685
+ requestedEffort === "max" ? "high" : requestedEffort,
16686
+ supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh"
16687
+ );
16633
16688
  return streamOpenAIResponses(model, context, {
16634
16689
  ...base,
16635
16690
  reasoningEffort
@@ -16664,8 +16719,8 @@ if (typeof process !== "undefined" && (process.versions?.node || process.version
16664
16719
  }
16665
16720
  var CODEX_URL = "https://chatgpt.com/backend-api/codex/responses";
16666
16721
  var JWT_CLAIM_PATH = "https://api.openai.com/auth";
16667
- var MAX_RETRIES = 3;
16668
- var BASE_DELAY_MS = 1e3;
16722
+ var MAX_RETRIES = 4;
16723
+ var RETRY_DELAY_MS = 1e4;
16669
16724
  var CODEX_TOOL_CALL_PROVIDERS = /* @__PURE__ */ new Set(["openai", "openai-codex", "opencode"]);
16670
16725
  var CODEX_RESPONSE_STATUSES = /* @__PURE__ */ new Set([
16671
16726
  "completed",
@@ -16679,21 +16734,29 @@ var TRANSIENT_ERROR_PATTERNS = [
16679
16734
  /rate.?limit/i,
16680
16735
  /overloaded/i,
16681
16736
  /service.?unavailable/i,
16737
+ /server.?error/i,
16738
+ /terminated/i,
16682
16739
  /upstream.?connect/i,
16683
16740
  /connection.?refused/i
16684
16741
  ];
16685
16742
  var RETRYABLE_STATUS_CODES = /* @__PURE__ */ new Set([429, 500, 502, 503, 504]);
16686
16743
  var CodexRetryExecutor = class _CodexRetryExecutor {
16744
+ static terminalErrorMarker = /* @__PURE__ */ Symbol("codexTerminalError");
16687
16745
  static isRetryable(status, errorText) {
16688
16746
  if (RETRYABLE_STATUS_CODES.has(status)) {
16689
16747
  return true;
16690
16748
  }
16691
16749
  return TRANSIENT_ERROR_PATTERNS.some((pattern) => pattern.test(errorText));
16692
16750
  }
16693
- // Exponential backoff where every retry doubles the delay. A left shift is
16694
- // used because the exponent is always a small non-negative whole number.
16695
- static backoffDelay(attempt) {
16696
- return BASE_DELAY_MS * (1 << attempt);
16751
+ static backoffDelay(_attempt) {
16752
+ return RETRY_DELAY_MS;
16753
+ }
16754
+ static markTerminal(error) {
16755
+ error[_CodexRetryExecutor.terminalErrorMarker] = true;
16756
+ return error;
16757
+ }
16758
+ static isTerminal(error) {
16759
+ return Boolean(error[_CodexRetryExecutor.terminalErrorMarker]);
16697
16760
  }
16698
16761
  static sleep(ms, signal) {
16699
16762
  return new Promise((resolve5, reject) => {
@@ -16729,12 +16792,15 @@ var CodexRetryExecutor = class _CodexRetryExecutor {
16729
16792
  statusText: response.statusText
16730
16793
  });
16731
16794
  const info = await CodexErrorInterpreter.parse(fakeResponse);
16732
- throw new Error(info.friendlyMessage || info.message);
16795
+ throw _CodexRetryExecutor.markTerminal(new Error(info.friendlyMessage || info.message));
16733
16796
  } catch (error) {
16734
16797
  if (error instanceof Error && (error.name === "AbortError" || error.message === "Request was aborted")) {
16735
16798
  throw new Error("Request was aborted");
16736
16799
  }
16737
16800
  lastError = error instanceof Error ? error : new Error(String(error));
16801
+ if (_CodexRetryExecutor.isTerminal(lastError)) {
16802
+ throw lastError;
16803
+ }
16738
16804
  if (attempt < MAX_RETRIES && !lastError.message.includes("usage limit")) {
16739
16805
  await _CodexRetryExecutor.sleep(_CodexRetryExecutor.backoffDelay(attempt), options?.signal);
16740
16806
  continue;
@@ -16802,20 +16868,45 @@ var streamOpenAICodexResponses = (model, context, options) => {
16802
16868
  const requestFactory = new CodexRequestFactory(model, context, options, apiKey);
16803
16869
  const { body, bodyJson, headers } = requestFactory.buildPayload();
16804
16870
  options?.onPayload?.(body);
16805
- const response = await executeWithCodexRetry(
16806
- () => fetch(CODEX_URL, {
16807
- method: "POST",
16808
- headers,
16809
- body: bodyJson,
16810
- signal: options?.signal
16811
- }),
16812
- options?.signal
16813
- );
16814
- if (!response.body) {
16815
- throw new Error("No response body");
16871
+ let completedOutput;
16872
+ let completedEvents;
16873
+ for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
16874
+ const response = await executeWithCodexRetry(
16875
+ () => fetch(CODEX_URL, {
16876
+ method: "POST",
16877
+ headers,
16878
+ body: bodyJson,
16879
+ signal: options?.signal
16880
+ }),
16881
+ options?.signal
16882
+ );
16883
+ if (!response.body) {
16884
+ throw new Error("No response body");
16885
+ }
16886
+ const attemptOutput = createAssistantMessageOutput(model);
16887
+ const attemptStream = new AssistantMessageEventStream();
16888
+ try {
16889
+ await processStream(response, attemptOutput, attemptStream, model);
16890
+ completedOutput = attemptOutput;
16891
+ completedEvents = [...attemptStream.getHistory()];
16892
+ break;
16893
+ } catch (error) {
16894
+ if (options?.signal?.aborted) {
16895
+ throw new Error("Request was aborted");
16896
+ }
16897
+ const message = error instanceof Error ? error.message : JSON.stringify(error);
16898
+ if (attempt >= MAX_RETRIES || !CodexRetryExecutor.isRetryable(200, message)) {
16899
+ throw error;
16900
+ }
16901
+ await CodexRetryExecutor.sleep(CodexRetryExecutor.backoffDelay(attempt), options?.signal);
16902
+ }
16903
+ }
16904
+ if (!completedOutput || !completedEvents) {
16905
+ throw new Error("Failed after retries");
16816
16906
  }
16907
+ Object.assign(output, completedOutput);
16817
16908
  events.start();
16818
- await processStream(response, output, stream, model);
16909
+ for (const event of completedEvents) stream.push(event);
16819
16910
  if (options?.signal?.aborted) {
16820
16911
  throw new Error("Request was aborted");
16821
16912
  }
@@ -16838,7 +16929,8 @@ var streamSimpleOpenAICodexResponses = (model, context, options) => {
16838
16929
  throw new Error(`No API key for provider: ${model.provider}`);
16839
16930
  }
16840
16931
  const base = buildBaseOptions(model, options, apiKey);
16841
- const reasoningEffort = mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
16932
+ const mappedEffort = options?.reasoning === "max" && model.reasoningEfforts?.includes("max") ? "max" : mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
16933
+ const reasoningEffort = mappedEffort && model.reasoningEfforts && (mappedEffort === "minimal" || !model.reasoningEfforts.includes(mappedEffort)) ? void 0 : mappedEffort;
16842
16934
  return streamOpenAICodexResponses(model, context, {
16843
16935
  ...base,
16844
16936
  reasoningEffort
@@ -16860,7 +16952,10 @@ function buildRequestBody(model, context, options) {
16860
16952
  tool_choice: "auto",
16861
16953
  parallel_tool_calls: true
16862
16954
  };
16863
- if (options?.temperature !== void 0) {
16955
+ if (options?.maxTokens !== void 0) {
16956
+ body.max_output_tokens = options.maxTokens;
16957
+ }
16958
+ if (options?.temperature !== void 0 && model.id !== "gpt-6-astra") {
16864
16959
  body.temperature = options.temperature;
16865
16960
  }
16866
16961
  if (context.tools) {
@@ -17517,7 +17612,8 @@ var CLAUDE_THINKING_BUDGETS = {
17517
17612
  low: 1500,
17518
17613
  medium: 6e3,
17519
17614
  high: 2e4,
17520
- xhigh: 2e4
17615
+ xhigh: 2e4,
17616
+ max: 2e4
17521
17617
  };
17522
17618
  function resolveThinkingPayload(model, options) {
17523
17619
  if (!options.reasoning || !model.reasoning) {
package/dist/ai.js CHANGED
@@ -4818,6 +4818,33 @@ var MODELS = {
4818
4818
  contextWindow: 1e6,
4819
4819
  maxTokens: 128e3
4820
4820
  },
4821
+ "gpt-6-astra": {
4822
+ id: "gpt-6-astra",
4823
+ name: "GPT-6 Astra",
4824
+ api: "openai-responses",
4825
+ provider: "openai",
4826
+ baseUrl: "https://api.openai.com/v1",
4827
+ reasoning: true,
4828
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
4829
+ input: ["text", "image"],
4830
+ cost: {
4831
+ input: 10,
4832
+ output: 50,
4833
+ cacheRead: 1,
4834
+ cacheWrite: 12.5
4835
+ },
4836
+ longContextPricing: {
4837
+ threshold: 272e3,
4838
+ multipliers: {
4839
+ input: 2,
4840
+ output: 1.5,
4841
+ cacheRead: 2,
4842
+ cacheWrite: 2
4843
+ }
4844
+ },
4845
+ contextWindow: 105e4,
4846
+ maxTokens: 128e3
4847
+ },
4821
4848
  "gpt-5.6-sol": {
4822
4849
  id: "gpt-5.6-sol",
4823
4850
  name: "GPT-5.6 Sol",
@@ -5280,6 +5307,28 @@ var MODELS = {
5280
5307
  },
5281
5308
  contextWindow: 105e4,
5282
5309
  maxTokens: 128e3
5310
+ },
5311
+ "gpt-6-astra": {
5312
+ id: "gpt-6-astra",
5313
+ name: "GPT-6 Astra",
5314
+ api: "openai-codex-responses",
5315
+ provider: "openai-codex",
5316
+ baseUrl: "https://chatgpt.com/backend-api",
5317
+ reasoning: true,
5318
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
5319
+ input: ["text", "image"],
5320
+ cost: {
5321
+ input: 10,
5322
+ output: 50,
5323
+ cacheRead: 1,
5324
+ cacheWrite: 12.5
5325
+ },
5326
+ longContextPricing: {
5327
+ threshold: 272e3,
5328
+ multipliers: { input: 2, output: 1.5, cacheRead: 2, cacheWrite: 2 }
5329
+ },
5330
+ contextWindow: 105e4,
5331
+ maxTokens: 128e3
5283
5332
  }
5284
5333
  },
5285
5334
  "opencode": {
@@ -12810,9 +12859,9 @@ function loadCustomModels(models) {
12810
12859
  function calculateCost(model, usage) {
12811
12860
  return modelRegistry.calculateCost(model, usage);
12812
12861
  }
12813
- var XHIGH_MODEL_IDS = /* @__PURE__ */ new Set(["gpt-5.1-codex-max", "gpt-5.2", "gpt-5.2-codex"]);
12862
+ var XHIGH_MODEL_IDS = /* @__PURE__ */ new Set(["gpt-5.1-codex-max", "gpt-5.2", "gpt-5.2-codex", "gpt-6-astra"]);
12814
12863
  function supportsXhigh(model) {
12815
- return XHIGH_MODEL_IDS.has(model.id);
12864
+ return model.reasoningEfforts?.includes("xhigh") ?? XHIGH_MODEL_IDS.has(model.id);
12816
12865
  }
12817
12866
  function modelsAreEqual(a, b) {
12818
12867
  if (!a || !b) return false;
@@ -13551,12 +13600,12 @@ function buildBaseOptions(model, options, apiKey) {
13551
13600
  return base;
13552
13601
  }
13553
13602
  function clampReasoning(effort) {
13554
- if (effort === "xhigh") return "high";
13603
+ if (effort === "xhigh" || effort === "max") return "high";
13555
13604
  return effort;
13556
13605
  }
13557
13606
  function mapThinkingLevel(level, provider = "clamp-xhigh") {
13558
13607
  if (!level) return void 0;
13559
- if (provider === "supports-xhigh") return level;
13608
+ if (provider === "supports-xhigh") return level === "max" ? "high" : level;
13560
13609
  return clampReasoning(level);
13561
13610
  }
13562
13611
  var RESERVED_ANSWER_TOKENS = 1024;
@@ -13568,8 +13617,10 @@ var REASONING_BUDGET_DEFAULTS = {
13568
13617
  // 2Ki
13569
13618
  medium: BUDGET_UNIT * 8,
13570
13619
  // 8Ki
13571
- high: BUDGET_UNIT * 16
13620
+ high: BUDGET_UNIT * 16,
13572
13621
  // 16Ki
13622
+ max: BUDGET_UNIT * 16
13623
+ // clamped to high for generic token-budget providers
13573
13624
  };
13574
13625
  function adjustMaxTokensForThinking(baseMaxTokens, modelMaxTokens, reasoningLevel, customBudgets) {
13575
13626
  const budgets = { ...REASONING_BUDGET_DEFAULTS, ...customBudgets };
@@ -16788,7 +16839,7 @@ var OpenAIResponsesParamBuilder = class {
16788
16839
  if (this.options?.maxTokens) {
16789
16840
  params.max_output_tokens = this.options.maxTokens;
16790
16841
  }
16791
- if (this.options?.temperature !== void 0) {
16842
+ if (this.model.id !== "gpt-6-astra" && this.options?.temperature !== void 0) {
16792
16843
  params.temperature = this.options.temperature;
16793
16844
  }
16794
16845
  if (this.options?.serviceTier !== void 0) {
@@ -16842,7 +16893,11 @@ var streamSimpleOpenAIResponses = (model, context, options) => {
16842
16893
  throw new Error(`No API key for provider: ${model.provider}`);
16843
16894
  }
16844
16895
  const base = buildBaseOptions(model, options, apiKey);
16845
- const reasoningEffort = mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
16896
+ const requestedEffort = options?.reasoning;
16897
+ const reasoningEffort = model.reasoningEfforts ? requestedEffort !== void 0 && requestedEffort !== "minimal" && model.reasoningEfforts.includes(requestedEffort) ? requestedEffort : void 0 : mapThinkingLevel(
16898
+ requestedEffort === "max" ? "high" : requestedEffort,
16899
+ supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh"
16900
+ );
16846
16901
  return streamOpenAIResponses(model, context, {
16847
16902
  ...base,
16848
16903
  reasoningEffort
@@ -16877,8 +16932,8 @@ if (typeof process !== "undefined" && (process.versions?.node || process.version
16877
16932
  }
16878
16933
  var CODEX_URL = "https://chatgpt.com/backend-api/codex/responses";
16879
16934
  var JWT_CLAIM_PATH = "https://api.openai.com/auth";
16880
- var MAX_RETRIES = 3;
16881
- var BASE_DELAY_MS = 1e3;
16935
+ var MAX_RETRIES = 4;
16936
+ var RETRY_DELAY_MS = 1e4;
16882
16937
  var CODEX_TOOL_CALL_PROVIDERS = /* @__PURE__ */ new Set(["openai", "openai-codex", "opencode"]);
16883
16938
  var CODEX_RESPONSE_STATUSES = /* @__PURE__ */ new Set([
16884
16939
  "completed",
@@ -16892,21 +16947,29 @@ var TRANSIENT_ERROR_PATTERNS = [
16892
16947
  /rate.?limit/i,
16893
16948
  /overloaded/i,
16894
16949
  /service.?unavailable/i,
16950
+ /server.?error/i,
16951
+ /terminated/i,
16895
16952
  /upstream.?connect/i,
16896
16953
  /connection.?refused/i
16897
16954
  ];
16898
16955
  var RETRYABLE_STATUS_CODES = /* @__PURE__ */ new Set([429, 500, 502, 503, 504]);
16899
16956
  var CodexRetryExecutor = class _CodexRetryExecutor {
16957
+ static terminalErrorMarker = /* @__PURE__ */ Symbol("codexTerminalError");
16900
16958
  static isRetryable(status, errorText) {
16901
16959
  if (RETRYABLE_STATUS_CODES.has(status)) {
16902
16960
  return true;
16903
16961
  }
16904
16962
  return TRANSIENT_ERROR_PATTERNS.some((pattern) => pattern.test(errorText));
16905
16963
  }
16906
- // Exponential backoff where every retry doubles the delay. A left shift is
16907
- // used because the exponent is always a small non-negative whole number.
16908
- static backoffDelay(attempt) {
16909
- return BASE_DELAY_MS * (1 << attempt);
16964
+ static backoffDelay(_attempt) {
16965
+ return RETRY_DELAY_MS;
16966
+ }
16967
+ static markTerminal(error) {
16968
+ error[_CodexRetryExecutor.terminalErrorMarker] = true;
16969
+ return error;
16970
+ }
16971
+ static isTerminal(error) {
16972
+ return Boolean(error[_CodexRetryExecutor.terminalErrorMarker]);
16910
16973
  }
16911
16974
  static sleep(ms, signal) {
16912
16975
  return new Promise((resolve, reject) => {
@@ -16942,12 +17005,15 @@ var CodexRetryExecutor = class _CodexRetryExecutor {
16942
17005
  statusText: response.statusText
16943
17006
  });
16944
17007
  const info = await CodexErrorInterpreter.parse(fakeResponse);
16945
- throw new Error(info.friendlyMessage || info.message);
17008
+ throw _CodexRetryExecutor.markTerminal(new Error(info.friendlyMessage || info.message));
16946
17009
  } catch (error) {
16947
17010
  if (error instanceof Error && (error.name === "AbortError" || error.message === "Request was aborted")) {
16948
17011
  throw new Error("Request was aborted");
16949
17012
  }
16950
17013
  lastError = error instanceof Error ? error : new Error(String(error));
17014
+ if (_CodexRetryExecutor.isTerminal(lastError)) {
17015
+ throw lastError;
17016
+ }
16951
17017
  if (attempt < MAX_RETRIES && !lastError.message.includes("usage limit")) {
16952
17018
  await _CodexRetryExecutor.sleep(_CodexRetryExecutor.backoffDelay(attempt), options?.signal);
16953
17019
  continue;
@@ -17015,20 +17081,45 @@ var streamOpenAICodexResponses = (model, context, options) => {
17015
17081
  const requestFactory = new CodexRequestFactory(model, context, options, apiKey);
17016
17082
  const { body, bodyJson, headers } = requestFactory.buildPayload();
17017
17083
  options?.onPayload?.(body);
17018
- const response = await executeWithCodexRetry(
17019
- () => fetch(CODEX_URL, {
17020
- method: "POST",
17021
- headers,
17022
- body: bodyJson,
17023
- signal: options?.signal
17024
- }),
17025
- options?.signal
17026
- );
17027
- if (!response.body) {
17028
- throw new Error("No response body");
17084
+ let completedOutput;
17085
+ let completedEvents;
17086
+ for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
17087
+ const response = await executeWithCodexRetry(
17088
+ () => fetch(CODEX_URL, {
17089
+ method: "POST",
17090
+ headers,
17091
+ body: bodyJson,
17092
+ signal: options?.signal
17093
+ }),
17094
+ options?.signal
17095
+ );
17096
+ if (!response.body) {
17097
+ throw new Error("No response body");
17098
+ }
17099
+ const attemptOutput = createAssistantMessageOutput(model);
17100
+ const attemptStream = new AssistantMessageEventStream();
17101
+ try {
17102
+ await processStream(response, attemptOutput, attemptStream, model);
17103
+ completedOutput = attemptOutput;
17104
+ completedEvents = [...attemptStream.getHistory()];
17105
+ break;
17106
+ } catch (error) {
17107
+ if (options?.signal?.aborted) {
17108
+ throw new Error("Request was aborted");
17109
+ }
17110
+ const message = error instanceof Error ? error.message : JSON.stringify(error);
17111
+ if (attempt >= MAX_RETRIES || !CodexRetryExecutor.isRetryable(200, message)) {
17112
+ throw error;
17113
+ }
17114
+ await CodexRetryExecutor.sleep(CodexRetryExecutor.backoffDelay(attempt), options?.signal);
17115
+ }
17116
+ }
17117
+ if (!completedOutput || !completedEvents) {
17118
+ throw new Error("Failed after retries");
17029
17119
  }
17120
+ Object.assign(output, completedOutput);
17030
17121
  events.start();
17031
- await processStream(response, output, stream2, model);
17122
+ for (const event of completedEvents) stream2.push(event);
17032
17123
  if (options?.signal?.aborted) {
17033
17124
  throw new Error("Request was aborted");
17034
17125
  }
@@ -17051,7 +17142,8 @@ var streamSimpleOpenAICodexResponses = (model, context, options) => {
17051
17142
  throw new Error(`No API key for provider: ${model.provider}`);
17052
17143
  }
17053
17144
  const base = buildBaseOptions(model, options, apiKey);
17054
- const reasoningEffort = mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
17145
+ const mappedEffort = options?.reasoning === "max" && model.reasoningEfforts?.includes("max") ? "max" : mapThinkingLevel(options?.reasoning, supportsXhigh(model) ? "supports-xhigh" : "clamp-xhigh");
17146
+ const reasoningEffort = mappedEffort && model.reasoningEfforts && (mappedEffort === "minimal" || !model.reasoningEfforts.includes(mappedEffort)) ? void 0 : mappedEffort;
17055
17147
  return streamOpenAICodexResponses(model, context, {
17056
17148
  ...base,
17057
17149
  reasoningEffort
@@ -17073,7 +17165,10 @@ function buildRequestBody(model, context, options) {
17073
17165
  tool_choice: "auto",
17074
17166
  parallel_tool_calls: true
17075
17167
  };
17076
- if (options?.temperature !== void 0) {
17168
+ if (options?.maxTokens !== void 0) {
17169
+ body.max_output_tokens = options.maxTokens;
17170
+ }
17171
+ if (options?.temperature !== void 0 && model.id !== "gpt-6-astra") {
17077
17172
  body.temperature = options.temperature;
17078
17173
  }
17079
17174
  if (context.tools) {
@@ -17730,7 +17825,8 @@ var CLAUDE_THINKING_BUDGETS = {
17730
17825
  low: 1500,
17731
17826
  medium: 6e3,
17732
17827
  high: 2e4,
17733
- xhigh: 2e4
17828
+ xhigh: 2e4,
17829
+ max: 2e4
17734
17830
  };
17735
17831
  function resolveThinkingPayload(model, options) {
17736
17832
  if (!options.reasoning || !model.reasoning) {
package/dist/cli.js CHANGED
@@ -410,6 +410,28 @@ var MODEL_CARDS = [
410
410
  cacheReadPerMTok: 0.275
411
411
  }
412
412
  },
413
+ {
414
+ id: "gpt-6-astra",
415
+ provider: "openai",
416
+ api: "openai-responses",
417
+ displayName: "GPT-6 Astra",
418
+ baseUrl: "https://api.openai.com/v1",
419
+ contextWindow: 105e4,
420
+ maxOutputTokens: 128e3,
421
+ modalities: ["text", "image"],
422
+ reasoning: true,
423
+ reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
424
+ longContextPricing: {
425
+ threshold: 272e3,
426
+ multipliers: { input: 2, output: 1.5, cacheRead: 2, cacheWrite: 2 }
427
+ },
428
+ cost: {
429
+ inputPerMTok: 10,
430
+ outputPerMTok: 50,
431
+ cacheReadPerMTok: 1,
432
+ cacheWritePerMTok: 12.5
433
+ }
434
+ },
413
435
  // ── Google (Gemini) ─────────────────────────────────────────────────────
414
436
  {
415
437
  id: "gemini-2.5-pro",
@@ -1113,6 +1135,7 @@ function anthropicThinkingBudget(level, maxOutput) {
1113
1135
  low: 0.2,
1114
1136
  medium: 0.4,
1115
1137
  high: 0.6,
1138
+ xhigh: 0.8,
1116
1139
  max: 0.8
1117
1140
  };
1118
1141
  if (level === "off") {
@@ -1917,13 +1940,17 @@ function readNumber2(o, key) {
1917
1940
  const v = o[key];
1918
1941
  return typeof v === "number" && Number.isFinite(v) ? v : void 0;
1919
1942
  }
1920
- function reasoningEffort(level) {
1943
+ function reasoningEffort(model, level) {
1944
+ if (model.reasoningEfforts?.includes(level)) {
1945
+ return level;
1946
+ }
1921
1947
  switch (level) {
1922
1948
  case "low":
1923
1949
  return "low";
1924
1950
  case "medium":
1925
1951
  return "medium";
1926
1952
  case "high":
1953
+ case "xhigh":
1927
1954
  case "max":
1928
1955
  return "high";
1929
1956
  }
@@ -1964,14 +1991,17 @@ function buildBody2(model, c, opts) {
1964
1991
  if (ceiling > 0) {
1965
1992
  body.max_output_tokens = ceiling;
1966
1993
  }
1967
- if (opts.temperature !== void 0) {
1994
+ if (model.id !== "gpt-6-astra" && opts.temperature !== void 0) {
1968
1995
  body.temperature = opts.temperature;
1969
1996
  }
1970
- if (opts.topP !== void 0) {
1997
+ if (model.id !== "gpt-6-astra" && opts.topP !== void 0) {
1971
1998
  body.top_p = opts.topP;
1972
1999
  }
1973
2000
  if (model.reasoning && opts.thinking !== void 0 && opts.thinking !== "off") {
1974
- body.reasoning = { effort: reasoningEffort(opts.thinking) };
2001
+ const effort = reasoningEffort(model, opts.thinking);
2002
+ if (effort !== void 0) {
2003
+ body.reasoning = { effort };
2004
+ }
1975
2005
  }
1976
2006
  return body;
1977
2007
  }
@@ -2246,6 +2276,7 @@ function geminiThinkingBudget(level, maxOutput) {
2246
2276
  low: 0.2,
2247
2277
  medium: 0.4,
2248
2278
  high: 0.6,
2279
+ xhigh: 0.8,
2249
2280
  max: 0.8
2250
2281
  };
2251
2282
  const budget = Math.floor(maxOutput * fraction[level]);
@@ -2539,6 +2570,7 @@ function thinkingBudget(level, maxOutput) {
2539
2570
  low: 0.2,
2540
2571
  medium: 0.4,
2541
2572
  high: 0.6,
2573
+ xhigh: 0.8,
2542
2574
  max: 0.8
2543
2575
  };
2544
2576
  return Math.max(0, Math.floor(maxOutput * fraction[level]));