@oh-my-pi/pi-ai 18.3.1 → 18.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.3.2] - 2026-09-25
6
+
7
+ ### Fixed
8
+
9
+ - Fixed capped Anthropic and Bedrock Claude requests with thinking enabled, including on-demand compaction, ending at `max_tokens` with no answer; every capped request now gets its effort's thinking budget on top of the requested output ([#13300](https://github.com/can1357/oh-my-pi/pull/13300) by [@alphastorm](https://github.com/alphastorm))
10
+
5
11
  ## [18.3.1] - 2026-09-25
6
12
 
7
13
  ### Added
@@ -8,6 +8,10 @@
8
8
  * capacity right now" with a 429 `slot_busy` or a 529 overload that the client
9
9
  * retries after the server-stated interval.
10
10
  *
11
+ * Before the slow lane, the server may grant a small wrap-up allowance (drawn
12
+ * from the weekly limit) past the session limit: 2xx responses then carry
13
+ * `anthropic-ratelimit-unified-grace-{5h,7d}-utilization` above zero.
14
+ *
11
15
  * This module owns the wire contract only (header names, parsing). The mode's
12
16
  * state machine lives with the caller, which plugs into the Anthropic provider
13
17
  * through {@link AnthropicSlowModeHooks}.
@@ -43,6 +47,17 @@ export interface AnthropicSlowModeSignal {
43
47
  unifiedLimitClaim: boolean;
44
48
  /** True when extra usage (overage) is serving this account. */
45
49
  overageInUse: boolean;
50
+ /**
51
+ * Wrap-up allowance usage (0..1) past the 5-hour and weekly limits; any
52
+ * value above zero means the request ran on the allowance. Present only on
53
+ * responses carrying `anthropic-ratelimit-unified-status`.
54
+ */
55
+ graceUtilization?: {
56
+ fiveHour: number;
57
+ sevenDay: number;
58
+ };
59
+ /** True when `anthropic-ratelimit-unified-overage-status` lets extra usage serve requests. */
60
+ overageAllowed?: boolean;
46
61
  }
47
62
  /** One pre-content failure the provider hands to {@link AnthropicSlowModeHooks.onFailure}. */
48
63
  export interface AnthropicSlowModeFailure {
@@ -443,9 +443,10 @@ export interface StreamOptions {
443
443
  */
444
444
  fallbackCreditRedemption?: AnthropicFallbackCreditHandle;
445
445
  /**
446
- * Anthropic subscription slow-mode state machine (Claude Code `/low-priority`).
447
- * Consulted only for first-party OAuth `anthropic` requests: stamps
448
- * `anthropic-usage-limit: slow` while active and decides capacity waits.
446
+ * Anthropic subscription usage-limit state machine (wrap-up allowance and
447
+ * Claude Code's `/low-priority`). Consulted only for first-party OAuth
448
+ * `anthropic` requests: stamps `anthropic-usage-limit: slow` while active,
449
+ * observes limit headers, and decides capacity waits.
449
450
  */
450
451
  anthropicSlowMode?: AnthropicSlowModeHooks;
451
452
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.3.1",
3
+ "version": "18.3.3",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -155,11 +155,11 @@
155
155
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
156
156
  },
157
157
  "dependencies": {
158
- "@oh-my-pi/omptype": "18.3.1",
159
- "@oh-my-pi/pi-catalog": "18.3.1",
160
- "@oh-my-pi/pi-natives": "18.3.1",
161
- "@oh-my-pi/pi-utils": "18.3.1",
162
- "@oh-my-pi/pi-wire": "18.3.1"
158
+ "@oh-my-pi/omptype": "18.3.3",
159
+ "@oh-my-pi/pi-catalog": "18.3.3",
160
+ "@oh-my-pi/pi-natives": "18.3.3",
161
+ "@oh-my-pi/pi-utils": "18.3.3",
162
+ "@oh-my-pi/pi-wire": "18.3.3"
163
163
  },
164
164
  "devDependencies": {
165
165
  "@types/bun": "^1.3.14"
@@ -8,6 +8,10 @@
8
8
  * capacity right now" with a 429 `slot_busy` or a 529 overload that the client
9
9
  * retries after the server-stated interval.
10
10
  *
11
+ * Before the slow lane, the server may grant a small wrap-up allowance (drawn
12
+ * from the weekly limit) past the session limit: 2xx responses then carry
13
+ * `anthropic-ratelimit-unified-grace-{5h,7d}-utilization` above zero.
14
+ *
11
15
  * This module owns the wire contract only (header names, parsing). The mode's
12
16
  * state machine lives with the caller, which plugs into the Anthropic provider
13
17
  * through {@link AnthropicSlowModeHooks}.
@@ -55,6 +59,14 @@ export interface AnthropicSlowModeSignal {
55
59
  unifiedLimitClaim: boolean;
56
60
  /** True when extra usage (overage) is serving this account. */
57
61
  overageInUse: boolean;
62
+ /**
63
+ * Wrap-up allowance usage (0..1) past the 5-hour and weekly limits; any
64
+ * value above zero means the request ran on the allowance. Present only on
65
+ * responses carrying `anthropic-ratelimit-unified-status`.
66
+ */
67
+ graceUtilization?: { fiveHour: number; sevenDay: number };
68
+ /** True when `anthropic-ratelimit-unified-overage-status` lets extra usage serve requests. */
69
+ overageAllowed?: boolean;
58
70
  }
59
71
 
60
72
  /** One pre-content failure the provider hands to {@link AnthropicSlowModeHooks.onFailure}. */
@@ -125,6 +137,12 @@ function readNonNegative(headers: HeadersLike, name: string): number | undefined
125
137
  return Number.isFinite(value) && value >= 0 ? value : undefined;
126
138
  }
127
139
 
140
+ /** Utilization header clamped to 0..1; absent or malformed reads as 0. */
141
+ function readUtilization(headers: HeadersLike, name: string): number {
142
+ const value = readNonNegative(headers, name);
143
+ return value === undefined ? 0 : Math.min(1, value);
144
+ }
145
+
128
146
  function parseStatus(raw: string | undefined): AnthropicSlowStatus | undefined {
129
147
  if (raw === undefined) return undefined;
130
148
  switch (raw.trim()) {
@@ -168,6 +186,15 @@ export function parseAnthropicSlowModeHeaders(headers: HeadersLike): AnthropicSl
168
186
  readHeader(headers, "anthropic-ratelimit-unified-overage-status"),
169
187
  );
170
188
  const overageInUse = readHeader(headers, "anthropic-ratelimit-unified-overage-in-use")?.trim() === "true";
189
+ const overageStatus = readHeader(headers, "anthropic-ratelimit-unified-overage-status")?.trim();
190
+ const overageAllowed = overageStatus === "allowed" || overageStatus === "allowed_warning";
191
+ const graceUtilization =
192
+ readHeader(headers, "anthropic-ratelimit-unified-status") === undefined
193
+ ? undefined
194
+ : {
195
+ fiveHour: readUtilization(headers, "anthropic-ratelimit-unified-grace-5h-utilization"),
196
+ sevenDay: readUtilization(headers, "anthropic-ratelimit-unified-grace-7d-utilization"),
197
+ };
171
198
  const signal: AnthropicSlowModeSignal = {
172
199
  ...(offer !== undefined ? { offer } : {}),
173
200
  ...(status !== undefined ? { status } : {}),
@@ -180,6 +207,8 @@ export function parseAnthropicSlowModeHeaders(headers: HeadersLike): AnthropicSl
180
207
  ...(weeklyResetAtSec !== undefined ? { weeklyResetAtSec } : {}),
181
208
  unifiedLimitClaim,
182
209
  overageInUse,
210
+ ...(overageAllowed ? { overageAllowed } : {}),
211
+ ...(graceUtilization !== undefined ? { graceUtilization } : {}),
183
212
  };
184
213
  const hasSlowFacts =
185
214
  offer !== undefined ||
@@ -188,7 +217,15 @@ export function parseAnthropicSlowModeHeaders(headers: HeadersLike): AnthropicSl
188
217
  maxWaitSec !== undefined ||
189
218
  budgetUtilization !== undefined ||
190
219
  budgetResetAtSec !== undefined;
191
- if (!hasSlowFacts && !unifiedLimitClaim && fiveHourResetAtSec === undefined && unifiedResetAtSec === undefined) {
220
+ // A unified-status response always carries wrap-up facts, even when both
221
+ // grace readings are zero: that is how the controller sees the window close.
222
+ if (
223
+ !hasSlowFacts &&
224
+ graceUtilization === undefined &&
225
+ !unifiedLimitClaim &&
226
+ fiveHourResetAtSec === undefined &&
227
+ unifiedResetAtSec === undefined
228
+ ) {
192
229
  return undefined;
193
230
  }
194
231
  return signal;
package/src/stream.ts CHANGED
@@ -2138,11 +2138,21 @@ function mapOptionsForApi<TApi extends Api>(
2138
2138
  ? mapEffortToAnthropicAdaptiveEffort(model, reasoning)
2139
2139
  : undefined;
2140
2140
 
2141
+ // A caller's maxTokens is the output it asked for, but thinking spends the
2142
+ // same max_tokens: adaptive thinking can use all of it and leave no answer.
2143
+ // Give thinking its budget on top, as the budget-only path below does. An
2144
+ // uncapped request keeps the provider default.
2145
+ const maxTokensWithThinking =
2146
+ base.maxTokens === undefined
2147
+ ? undefined
2148
+ : maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
2149
+
2141
2150
  // For Opus 4.6+ and Sonnet 4.6+: use adaptive thinking with effort level
2142
2151
  // For older models: use budget-based thinking
2143
2152
  if (thinkingMode === "anthropic-adaptive") {
2144
2153
  return castApi<"anthropic-messages">({
2145
2154
  ...base,
2155
+ maxTokens: maxTokensWithThinking,
2146
2156
  requestModelId: resolveWireModelId(model, reasoning),
2147
2157
  thinkingEnabled: true,
2148
2158
  effort,
@@ -2155,6 +2165,7 @@ function mapOptionsForApi<TApi extends Api>(
2155
2165
  if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
2156
2166
  return castApi<"anthropic-messages">({
2157
2167
  ...base,
2168
+ maxTokens: maxTokensWithThinking,
2158
2169
  requestModelId: resolveWireModelId(model, reasoning),
2159
2170
  thinkingEnabled: true,
2160
2171
  thinkingBudgetTokens: thinkingBudget,
@@ -2214,8 +2225,25 @@ function mapOptionsForApi<TApi extends Api>(
2214
2225
  guardrailTrace: model.guardrailTrace ?? options?.guardrailTrace,
2215
2226
  requestMetadata: options?.requestMetadata,
2216
2227
  };
2217
- // Effort modes send effort directly, no budget_tokens — skip budget inflation.
2218
- if (model.thinking?.mode === "effort" || model.thinking?.mode === "anthropic-adaptive") {
2228
+ // Adaptive Claude shares max_tokens between thinking and the answer, like
2229
+ // the anthropic-messages adaptive path: a caller's cap is the output it
2230
+ // wants, so add the effort's budget on top. Uncapped requests keep the
2231
+ // provider default.
2232
+ if (model.thinking?.mode === "anthropic-adaptive") {
2233
+ const reasoning = bedrockBase.reasoning;
2234
+ const budget = reasoning
2235
+ ? (options?.thinkingBudgets?.[reasoning] ?? BEDROCK_CLAUDE_THINKING[reasoning])
2236
+ : 0;
2237
+ if (!model.reasoning || bedrockBase.maxTokens === undefined || budget <= 0) {
2238
+ return castApi<"bedrock-converse-stream">(bedrockBase);
2239
+ }
2240
+ return castApi<"bedrock-converse-stream">({
2241
+ ...bedrockBase,
2242
+ maxTokens: maxTokensWithThinkingBudget(bedrockBase.maxTokens, model.maxTokens, budget),
2243
+ });
2244
+ }
2245
+ // Effort mode sends effort directly, no budget_tokens — skip budget inflation.
2246
+ if (model.thinking?.mode === "effort") {
2219
2247
  return castApi<"bedrock-converse-stream">(bedrockBase);
2220
2248
  }
2221
2249
  const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
package/src/types.ts CHANGED
@@ -639,9 +639,10 @@ export interface StreamOptions {
639
639
  */
640
640
  fallbackCreditRedemption?: AnthropicFallbackCreditHandle;
641
641
  /**
642
- * Anthropic subscription slow-mode state machine (Claude Code `/low-priority`).
643
- * Consulted only for first-party OAuth `anthropic` requests: stamps
644
- * `anthropic-usage-limit: slow` while active and decides capacity waits.
642
+ * Anthropic subscription usage-limit state machine (wrap-up allowance and
643
+ * Claude Code's `/low-priority`). Consulted only for first-party OAuth
644
+ * `anthropic` requests: stamps `anthropic-usage-limit: slow` while active,
645
+ * observes limit headers, and decides capacity waits.
645
646
  */
646
647
  anthropicSlowMode?: AnthropicSlowModeHooks;
647
648
  }