@oh-my-pi/pi-ai 18.3.0 → 18.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/THIRD-PARTY-NOTICES.txt +2 -2
  3. package/dist/types/auth/policy.d.ts +8 -1
  4. package/dist/types/auth/pool.d.ts +8 -0
  5. package/dist/types/auth/types.d.ts +17 -7
  6. package/dist/types/auth/usage.d.ts +2 -0
  7. package/dist/types/auth-broker/discover.d.ts +22 -1
  8. package/dist/types/auth-storage.d.ts +35 -11
  9. package/dist/types/error/rate-limit.d.ts +3 -2
  10. package/dist/types/index.d.ts +1 -0
  11. package/dist/types/providers/anthropic-slow-mode.d.ts +109 -0
  12. package/dist/types/providers/anthropic-wire.d.ts +4 -0
  13. package/dist/types/providers/mock.d.ts +3 -1
  14. package/dist/types/providers/openai-codex/live-steering.d.ts +77 -0
  15. package/dist/types/providers/transform-messages.d.ts +9 -1
  16. package/dist/types/types.d.ts +56 -0
  17. package/dist/types/usage/xai-oauth.d.ts +6 -1
  18. package/dist/types/utils/http-inspector.d.ts +6 -0
  19. package/package.json +6 -6
  20. package/src/auth/policy.ts +27 -6
  21. package/src/auth/pool.ts +35 -2
  22. package/src/auth/refresh.ts +2 -2
  23. package/src/auth/types.ts +17 -7
  24. package/src/auth/usage.ts +5 -0
  25. package/src/auth-broker/discover.ts +57 -27
  26. package/src/auth-storage.ts +121 -35
  27. package/src/error/flags.ts +2 -1
  28. package/src/error/rate-limit.ts +8 -4
  29. package/src/index.ts +1 -0
  30. package/src/providers/anthropic-slow-mode.ts +232 -0
  31. package/src/providers/anthropic-wire.ts +4 -0
  32. package/src/providers/anthropic.ts +322 -22
  33. package/src/providers/cowork-fetch.ts +11 -4
  34. package/src/providers/google-shared.ts +30 -7
  35. package/src/providers/inference-headers.ts +7 -1
  36. package/src/providers/mock.ts +4 -0
  37. package/src/providers/openai-codex/live-steering.ts +237 -0
  38. package/src/providers/openai-codex-responses.ts +372 -73
  39. package/src/providers/transform-messages.ts +27 -7
  40. package/src/stream.ts +32 -2
  41. package/src/types.ts +59 -0
  42. package/src/usage/registry.ts +2 -1
  43. package/src/usage/xai-oauth.ts +31 -1
  44. package/src/utils/http-inspector.ts +21 -2
  45. package/src/utils/openrouter-headers.ts +3 -3
@@ -1,3 +1,4 @@
1
+ import { AsyncLocalStorage } from "node:async_hooks";
1
2
  import { renderDemotedThinking } from "../dialect/demotion";
2
3
  import type {
3
4
  Api,
@@ -437,14 +438,23 @@ function hasPlausibleCredentialEntropy(token: string): boolean {
437
438
  }
438
439
 
439
440
  /**
440
- * Whether outbound credential-pattern redaction is active. Off by default;
441
- * hosts opt in explicitly (the coding agent wires this to the
442
- * `secrets.enabled` setting).
441
+ * Whether outbound credential-pattern redaction is active outside any
442
+ * {@link withCredentialRedaction} scope. Off by default; hosts opt in
443
+ * explicitly (the coding agent wires this to the `secrets.enabled` setting).
443
444
  */
444
445
  let credentialRedactionEnabled = false;
445
446
 
447
+ /** Per-request override of {@link credentialRedactionEnabled}; see {@link withCredentialRedaction}. */
448
+ const credentialRedactionScope = new AsyncLocalStorage<boolean>();
449
+
450
+ /** Redaction policy for the request being built: its scope's, else the process-wide switch. */
451
+ function isCredentialRedactionActive(): boolean {
452
+ return credentialRedactionScope.getStore() ?? credentialRedactionEnabled;
453
+ }
454
+
446
455
  /**
447
- * Toggle outbound credential-pattern redaction. When disabled (the default),
456
+ * Toggle process-wide outbound credential-pattern redaction (requests outside
457
+ * any {@link withCredentialRedaction} scope). When disabled (the default),
448
458
  * {@link redactSensitiveCredentials} and {@link redactSensitiveInObject} are
449
459
  * pass-throughs and outbound messages/system prompts leave the process
450
460
  * unmodified.
@@ -453,8 +463,18 @@ export function configureCredentialRedaction(enabled: boolean): void {
453
463
  credentialRedactionEnabled = enabled;
454
464
  }
455
465
 
466
+ /**
467
+ * Runs `fn` with outbound credential-pattern redaction forced on or off for
468
+ * every request it starts (including the async work those requests spawn),
469
+ * overriding {@link configureCredentialRedaction}. Lets concurrent sessions in
470
+ * one process each apply their own policy.
471
+ */
472
+ export function withCredentialRedaction<T>(enabled: boolean, fn: () => T): T {
473
+ return credentialRedactionScope.run(enabled, fn);
474
+ }
475
+
456
476
  export function redactSensitiveCredentials(text: string): string {
457
- if (!credentialRedactionEnabled) return text;
477
+ if (!isCredentialRedactionActive()) return text;
458
478
  return text.replace(SENSITIVE_TOKEN_RE, match => {
459
479
  if (!hasPlausibleCredentialEntropy(match)) return match;
460
480
  const lower = match.toLowerCase();
@@ -475,7 +495,7 @@ export function redactSensitiveCredentials(text: string): string {
475
495
  }
476
496
 
477
497
  export function redactSensitiveInObject(val: unknown): { result: unknown; changed: boolean } {
478
- if (!credentialRedactionEnabled) return { result: val, changed: false };
498
+ if (!isCredentialRedactionActive()) return { result: val, changed: false };
479
499
  if (typeof val === "string") {
480
500
  const redacted = redactSensitiveCredentials(val);
481
501
  return { result: redacted, changed: redacted !== val };
@@ -503,7 +523,7 @@ export function redactSensitiveInObject(val: unknown): { result: unknown; change
503
523
  }
504
524
 
505
525
  function redactSensitiveCredentialsInMessages(messages: Message[]): Message[] {
506
- if (!credentialRedactionEnabled) return messages;
526
+ if (!isCredentialRedactionActive()) return messages;
507
527
  return messages.map((msg): Message => {
508
528
  if (msg.role === "user" || msg.role === "developer") {
509
529
  const userMsg = msg as UserMessage | DeveloperMessage;
package/src/stream.ts CHANGED
@@ -2083,6 +2083,7 @@ function mapOptionsForApi<TApi extends Api>(
2083
2083
  streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
2084
2084
  codexSseMaxAttempts: options?.codexSseMaxAttempts,
2085
2085
  providerSessionState: options?.providerSessionState,
2086
+ liveSteering: options?.liveSteering,
2086
2087
  maxInFlightRequests: options?.maxInFlightRequests,
2087
2088
  toolNamespacesInfo: options?.toolNamespacesInfo,
2088
2089
  onPayload: options?.onPayload,
@@ -2095,6 +2096,7 @@ function mapOptionsForApi<TApi extends Api>(
2095
2096
  anthropicCacheRefreshRequest: options?.anthropicCacheRefreshRequest,
2096
2097
  anthropicPrefixMismatchBehavior: options?.anthropicPrefixMismatchBehavior,
2097
2098
  anthropicCompaction: options?.anthropicCompaction,
2099
+ anthropicSlowMode: options?.anthropicSlowMode,
2098
2100
  userProfileId: options?.userProfileId,
2099
2101
  ...simpleProviderOptions,
2100
2102
  };
@@ -2136,11 +2138,21 @@ function mapOptionsForApi<TApi extends Api>(
2136
2138
  ? mapEffortToAnthropicAdaptiveEffort(model, reasoning)
2137
2139
  : undefined;
2138
2140
 
2141
+ // A caller's maxTokens is the output it asked for, but thinking spends the
2142
+ // same max_tokens: adaptive thinking can use all of it and leave no answer.
2143
+ // Give thinking its budget on top, as the budget-only path below does. An
2144
+ // uncapped request keeps the provider default.
2145
+ const maxTokensWithThinking =
2146
+ base.maxTokens === undefined
2147
+ ? undefined
2148
+ : maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
2149
+
2139
2150
  // For Opus 4.6+ and Sonnet 4.6+: use adaptive thinking with effort level
2140
2151
  // For older models: use budget-based thinking
2141
2152
  if (thinkingMode === "anthropic-adaptive") {
2142
2153
  return castApi<"anthropic-messages">({
2143
2154
  ...base,
2155
+ maxTokens: maxTokensWithThinking,
2144
2156
  requestModelId: resolveWireModelId(model, reasoning),
2145
2157
  thinkingEnabled: true,
2146
2158
  effort,
@@ -2153,6 +2165,7 @@ function mapOptionsForApi<TApi extends Api>(
2153
2165
  if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
2154
2166
  return castApi<"anthropic-messages">({
2155
2167
  ...base,
2168
+ maxTokens: maxTokensWithThinking,
2156
2169
  requestModelId: resolveWireModelId(model, reasoning),
2157
2170
  thinkingEnabled: true,
2158
2171
  thinkingBudgetTokens: thinkingBudget,
@@ -2212,8 +2225,25 @@ function mapOptionsForApi<TApi extends Api>(
2212
2225
  guardrailTrace: model.guardrailTrace ?? options?.guardrailTrace,
2213
2226
  requestMetadata: options?.requestMetadata,
2214
2227
  };
2215
- // Effort modes send effort directly, no budget_tokens — skip budget inflation.
2216
- if (model.thinking?.mode === "effort" || model.thinking?.mode === "anthropic-adaptive") {
2228
+ // Adaptive Claude shares max_tokens between thinking and the answer, like
2229
+ // the anthropic-messages adaptive path: a caller's cap is the output it
2230
+ // wants, so add the effort's budget on top. Uncapped requests keep the
2231
+ // provider default.
2232
+ if (model.thinking?.mode === "anthropic-adaptive") {
2233
+ const reasoning = bedrockBase.reasoning;
2234
+ const budget = reasoning
2235
+ ? (options?.thinkingBudgets?.[reasoning] ?? BEDROCK_CLAUDE_THINKING[reasoning])
2236
+ : 0;
2237
+ if (!model.reasoning || bedrockBase.maxTokens === undefined || budget <= 0) {
2238
+ return castApi<"bedrock-converse-stream">(bedrockBase);
2239
+ }
2240
+ return castApi<"bedrock-converse-stream">({
2241
+ ...bedrockBase,
2242
+ maxTokens: maxTokensWithThinkingBudget(bedrockBase.maxTokens, model.maxTokens, budget),
2243
+ });
2244
+ }
2245
+ // Effort mode sends effort directly, no budget_tokens — skip budget inflation.
2246
+ if (model.thinking?.mode === "effort") {
2217
2247
  return castApi<"bedrock-converse-stream">(bedrockBase);
2218
2248
  }
2219
2249
  const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
package/src/types.ts CHANGED
@@ -2,6 +2,7 @@ export * from "@oh-my-pi/pi-catalog/effort";
2
2
  export * from "@oh-my-pi/pi-catalog/types";
3
3
 
4
4
  import type { Type } from "@oh-my-pi/omptype";
5
+ import type { AnthropicSlowModeHooks } from "./providers/anthropic-slow-mode";
5
6
  import type {
6
7
  DeleteArgs,
7
8
  DeleteResult,
@@ -529,6 +530,13 @@ export interface StreamOptions {
529
530
  * Providers can use this to persist transport/session state between turns.
530
531
  */
531
532
  providerSessionState?: Map<string, ProviderSessionState>;
533
+ /**
534
+ * Source of user steering a provider may deliver into the response it is
535
+ * streaming (OpenAI Responses `response.steer` over the Codex WebSocket).
536
+ * Providers without mid-response input ignore it; unclaimed steering stays
537
+ * with the caller for its next request.
538
+ */
539
+ liveSteering?: LiveSteering;
532
540
  /** Canonical Codex compaction classification; ignored by other providers. */
533
541
  codexCompaction?: CodexCompactionRequestContext;
534
542
  /** Codex Code Mode tool exposure snapshot emitted as `tool_namespaces_info` turn metadata; ignored by other providers. */
@@ -624,6 +632,42 @@ export interface StreamOptions {
624
632
 
625
633
  /** Cursor exec/MCP tool handlers (cursor-agent only). */
626
634
  execHandlers?: CursorExecHandlers;
635
+ /**
636
+ * Anthropic fallback credit redemption handle from a prior classifier refusal.
637
+ * When present, the Anthropic provider replays the frozen request body and betas with
638
+ * the new model and `fallback_credit_token` to redeem prompt cache credit.
639
+ */
640
+ fallbackCreditRedemption?: AnthropicFallbackCreditHandle;
641
+ /**
642
+ * Anthropic subscription usage-limit state machine (wrap-up allowance and
643
+ * Claude Code's `/low-priority`). Consulted only for first-party OAuth
644
+ * `anthropic` requests: stamps `anthropic-usage-limit: slow` while active,
645
+ * observes limit headers, and decides capacity waits.
646
+ */
647
+ anthropicSlowMode?: AnthropicSlowModeHooks;
648
+ }
649
+
650
+ /**
651
+ * Caller-owned queue of user steering that a provider pulls from while a
652
+ * response streams. See {@link StreamOptions.liveSteering}.
653
+ */
654
+ export interface LiveSteering {
655
+ /** Resolves once steering may be claimable, or when `signal` aborts. Never consumes input. */
656
+ wait(signal: AbortSignal): Promise<void>;
657
+ /** Takes the queued steering as provider messages; `undefined` when none is deliverable now. */
658
+ claim(signal: AbortSignal): Promise<LiveSteerClaim | undefined>;
659
+ }
660
+
661
+ /**
662
+ * Steering taken from a {@link LiveSteering} source. The provider settles it
663
+ * exactly once; later calls are ignored.
664
+ */
665
+ export interface LiveSteerClaim {
666
+ readonly messages: readonly UserMessage[];
667
+ /** The server owns the input: the caller records it right after the current response. */
668
+ accept(): void;
669
+ /** Not delivered: the caller sends the input with its next request. */
670
+ reject(): void;
627
671
  }
628
672
 
629
673
  // Unified options with reasoning passed to streamSimple() and completeSimple()
@@ -988,6 +1032,8 @@ export interface UserMessage {
988
1032
  synthetic?: boolean;
989
1033
  /** True when injected mid-turn as a steer; consumed by the agent's pre-LLM transform to wrap it for emphasis. Never rendered. */
990
1034
  steering?: boolean;
1035
+ /** True when the provider delivered this steer into the response it was streaming (`response.steer`). Display-only; never sent. */
1036
+ liveSteered?: boolean;
991
1037
  /** Timestamp of a client-side history rewrite represented by this message. */
992
1038
  historyRewriteAt?: number;
993
1039
  /** Who initiated this message for billing/attribution semantics. */
@@ -1119,6 +1165,8 @@ export interface AssistantMessage {
1119
1165
  requestControls?: AnthropicRequestControls;
1120
1166
  /** Provider-specific opaque payload used to reconstruct transport-native history. */
1121
1167
  providerPayload?: ProviderPayload;
1168
+ /** In-memory fallback credit handle attached when a refusal response carries a fallback credit token. */
1169
+ fallbackCreditHandle?: AnthropicFallbackCreditHandle;
1122
1170
  timestamp: number; // Unix timestamp in milliseconds
1123
1171
  duration?: number; // Request duration in milliseconds
1124
1172
  ttft?: number; // Time to first token in milliseconds
@@ -1457,3 +1505,14 @@ export type AssistantMessageEvent =
1457
1505
  reason: Extract<StopReason, "aborted" | "error">;
1458
1506
  error: AssistantMessage;
1459
1507
  };
1508
+
1509
+ export interface AnthropicFallbackCreditHandle {
1510
+ token: string;
1511
+ prefillClaim?: boolean | null;
1512
+ params: unknown;
1513
+ betas?: readonly string[];
1514
+ betaHeader?: string;
1515
+ expiresAt: number;
1516
+ /** The refused response's content, in `AssistantMessage` block form. */
1517
+ refusedContent?: AssistantMessage["content"];
1518
+ }
@@ -17,7 +17,7 @@ import { codexRankingStrategy, openaiCodexUsageProvider } from "./openai-codex";
17
17
  import { opencodeGoRankingStrategy, opencodeGoUsageProvider } from "./opencode-go";
18
18
  import { syntheticUsageProvider } from "./synthetic";
19
19
  import { umansUsageProvider } from "./umans";
20
- import { xaiOauthUsageProvider } from "./xai-oauth";
20
+ import { xaiOauthRankingStrategy, xaiOauthUsageProvider } from "./xai-oauth";
21
21
  import { zaiRankingStrategy, zaiUsageProvider } from "./zai";
22
22
 
23
23
  /** Resolves the usage-based ranking strategy for a provider. */
@@ -64,6 +64,7 @@ const DEFAULT_RANKING_STRATEGIES = new Map<Provider, CredentialRankingStrategy>(
64
64
  ["kimi-code", kimiRankingStrategy],
65
65
  ["zai", zaiRankingStrategy],
66
66
  ["opencode-go", opencodeGoRankingStrategy],
67
+ ["xai-oauth", xaiOauthRankingStrategy],
67
68
  ]);
68
69
 
69
70
  /** Built-in ranking strategy for `provider`. */
@@ -10,6 +10,7 @@
10
10
  */
11
11
 
12
12
  import { toNumber } from "@oh-my-pi/pi-catalog/utils";
13
+ import { isUsageLimitExhausted } from "../auth/usage-report";
13
14
  import {
14
15
  buildXAICliBillingUrl,
15
16
  extractXAIAccessTokenSubject,
@@ -17,6 +18,7 @@ import {
17
18
  getXAICliBillingHeaders,
18
19
  } from "../registry/oauth/xai-oauth";
19
20
  import type {
21
+ CredentialRankingStrategy,
20
22
  UsageAmount,
21
23
  UsageFetchContext,
22
24
  UsageFetchParams,
@@ -26,7 +28,7 @@ import type {
26
28
  UsageWindow,
27
29
  } from "../usage";
28
30
  import { isRecord } from "../utils";
29
- import { DAY_MS, parseIsoTimestamp, usageStatus, WEEK_MS } from "./shared";
31
+ import { DAY_MS, HOUR_MS, parseIsoTimestamp, usageStatus, WEEK_MS } from "./shared";
30
32
 
31
33
  const PROVIDER_ID = "xai-oauth";
32
34
  const BILLING_SOURCE = "cli-chat-proxy.grok.com/v1/billing";
@@ -440,3 +442,31 @@ export const xaiOauthUsageProvider: UsageProvider = {
440
442
  };
441
443
  },
442
444
  };
445
+
446
+ /**
447
+ * Ranks SuperGrok accounts by weekly credits (or unified monthly included quota).
448
+ * xAI reports no short window, so the meter maps to `secondary`, which drives drain ranking.
449
+ */
450
+ export const xaiOauthRankingStrategy: CredentialRankingStrategy = {
451
+ scopeLimits(report) {
452
+ // Spent credits/included quota keeps serving on the on-demand cap; only hard-block
453
+ // the credential once no on-demand headroom remains.
454
+ const onDemand = report.limits.find(limit => limit.id === `${PROVIDER_ID}:on-demand`);
455
+ if (onDemand && !isUsageLimitExhausted(onDemand)) return [];
456
+ return report.limits.filter(
457
+ limit => limit.id === `${PROVIDER_ID}:credits:1w` || limit.id === `${PROVIDER_ID}:included:1mo`,
458
+ );
459
+ },
460
+ findWindowLimits(report) {
461
+ const credits = report.limits.find(limit => limit.id === `${PROVIDER_ID}:credits:1w`);
462
+ const included = report.limits.find(limit => limit.id === `${PROVIDER_ID}:included:1mo`);
463
+ return {
464
+ secondary: credits ?? included,
465
+ };
466
+ },
467
+ windowDefaults: {
468
+ // Inert: findWindowLimits never reports a primary window.
469
+ primaryMs: 5 * HOUR_MS,
470
+ secondaryMs: WEEK_MS,
471
+ },
472
+ };
@@ -59,6 +59,25 @@ export function shouldDumpRejectedRequest(error: unknown): boolean {
59
59
  return status === 400 || status === 413;
60
60
  }
61
61
 
62
+ const RAW_HTTP_REQUEST_LINE = "raw-http-request=";
63
+ const RAW_HTTP_REQUEST_SAVE_FAILED_LINE = "raw-http-request-save-failed=";
64
+
65
+ /**
66
+ * Remove the local request-dump lines {@link appendRawHttpRequestDumpFor400} appends,
67
+ * leaving only the provider-facing error text. Hosts that relay provider errors
68
+ * (RPC `prompt_result`) must not leak OMP-local file paths.
69
+ */
70
+ export function stripRawHttpRequestDiagnostics(message: string): string {
71
+ const lines = message.split("\n");
72
+ let end = lines.length;
73
+ while (
74
+ end > 0 &&
75
+ (lines[end - 1].startsWith(RAW_HTTP_REQUEST_LINE) || lines[end - 1].startsWith(RAW_HTTP_REQUEST_SAVE_FAILED_LINE))
76
+ )
77
+ end--;
78
+ return end === lines.length ? message : lines.slice(0, end).join("\n");
79
+ }
80
+
62
81
  export async function appendRawHttpRequestDumpFor400(
63
82
  message: string,
64
83
  error: unknown,
@@ -75,10 +94,10 @@ export async function appendRawHttpRequestDumpFor400(
75
94
 
76
95
  try {
77
96
  await Bun.write(filePath, `${JSON.stringify(payload, null, 2)}\n`);
78
- return `${message}\nraw-http-request=${filePath}`;
97
+ return `${message}\n${RAW_HTTP_REQUEST_LINE}${filePath}`;
79
98
  } catch (writeError) {
80
99
  const writeMessage = writeError instanceof Error ? writeError.message : String(writeError);
81
- return `${message}\nraw-http-request-save-failed=${writeMessage}`;
100
+ return `${message}\n${RAW_HTTP_REQUEST_SAVE_FAILED_LINE}${writeMessage}`;
82
101
  }
83
102
  }
84
103
 
@@ -1,10 +1,10 @@
1
- import { USER_AGENT } from "@oh-my-pi/pi-utils";
1
+ import { APP_NAME, APP_URL, USER_AGENT } from "@oh-my-pi/pi-utils";
2
2
 
3
3
  export function getOpenRouterHeaders(): Record<string, string> {
4
4
  return {
5
5
  "User-Agent": USER_AGENT,
6
- "HTTP-Referer": "https://omp.sh/",
7
- "X-OpenRouter-Title": "omp",
6
+ "HTTP-Referer": APP_URL,
7
+ "X-OpenRouter-Title": APP_NAME,
8
8
  "X-OpenRouter-Categories": "cli-agent",
9
9
  "X-OpenRouter-Cache": "true",
10
10
  "X-OpenRouter-Cache-TTL": "3600",