@oh-my-pi/pi-ai 18.4.2 → 18.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,16 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.3] - 2026-09-28
6
+
7
+ ### Added
8
+
9
+ - Added Command Code usage limits (5-hour, weekly, and credit balance) to /usage and the status line ([#13666](https://github.com/can1357/oh-my-pi/pull/13666) by [@riicodespretty](https://github.com/riicodespretty))
10
+
11
+ ### Changed
12
+
13
+ - Reduced per-token CPU and allocations while streaming: the leaked-thinking scanner used for OpenAI-compatible and custom endpoints no longer allocates per character, chat-completions and Bedrock look up a delta's content block in constant time, Google, Gemini CLI, Codex, and chat-completions streams skip raw SSE line capture unless an `onSseEvent` listener is attached, and event streams drain backlogs without `Array#shift` ([#13650](https://github.com/can1357/oh-my-pi/pull/13650) by [@H4vC](https://github.com/H4vC)).
14
+
5
15
  ## [18.4.2] - 2026-09-28
6
16
 
7
17
  ### Fixed
@@ -2306,28 +2316,4 @@
2306
2316
 
2307
2317
  - Fixed `google-gemini-cli` ignoring `Model.requestModelId` when serializing the request model id
2308
2318
 
2309
- ## [15.11.5] - 2026-06-12
2310
-
2311
- ### Added
2312
-
2313
- - Added `AuthStorage.listUsageHistory` to retrieve historical usage snapshots with optional `provider` and `sinceMs` filtering
2314
- - Added durable usage-history persistence in the sqlite auth store so successful usage reports are recorded as time-series snapshots of limit utilization for later trend inspection
2315
- - Added `AuthStorage.redeemResetCredit` to redeem stored OpenAI Codex saved rate-limit reset credits for a target account by `credentialId`, `accountId`, or `email`
2316
- - Added `listCodexResetCredits` and `consumeCodexResetCredit` exports for OpenAI Codex saved reset-credit listing and redemption
2317
- - Added `resetCredits` with `availableCount` to `UsageReport` so OpenAI Codex usage data now exposes redeemable rate-limit resets
2318
- - Added `openai-codex-reset` exports via package barrel for out-of-band tooling usage
2319
- - Added a one-shot request-debug target that writes the next provider HTTP request JSON to an explicit path.
2320
-
2321
- ### Changed
2322
-
2323
- - Changed `AuthStorage.redeemResetCredit` to invalidate cached usage data after a successful redemption so the next usage report reflects the reset immediately
2324
-
2325
- ### Fixed
2326
-
2327
- - Fixed temporary credential block state so redeemed reset credits immediately make the affected account selectable again after `redeemResetCredit` succeeds
2328
- - Fixed one-shot request-debug path handling so an explicit request log target is consumed after the next request and no longer affects subsequent calls
2329
- - Fixed explicit request-debug path mode to create missing parent directories before writing request logs
2330
- - Fixed explicit request-debug mode to overwrite existing `.res.log` files for the requested path instead of failing when they already exist
2331
- - Fixed OpenAI Responses `previous_response_id` chaining on Zero Data Retention orgs: the in-provider retry classifier missed the ZDR-specific 400 ("Previous response cannot be used for this organization due to Zero Data Retention"), so chained turns kept failing every other request after a brief recovery — the chain was reset but not disabled, so the next successful full-replay turn re-armed it. The ZDR phrasing is now classified categorically: one strike disables chaining for the session (skipping the three-strike circuit breaker) and the in-call retry drops `store: true`/`previous_response_id` and replays the full transcript instead ([#2341](https://github.com/can1357/oh-my-pi/issues/2341)).
2332
-
2333
- Older entries are archived in [packages/ai/CHANGELOG.md@dfbf3cc34eeb](https://github.com/can1357/oh-my-pi/blob/dfbf3cc34eeb5653580f51bfbcae558a9840f697/packages/ai/CHANGELOG.md).
2319
+ Older entries are archived in [packages/ai/CHANGELOG.md@689a3418cb45](https://github.com/can1357/oh-my-pi/blob/689a3418cb45d54a459cde2e1abf3f66f50e47a4/packages/ai/CHANGELOG.md).
@@ -1,3 +1,3 @@
1
- import type { Model } from "@oh-my-pi/pi-catalog/types";
1
+ import { type Model } from "@oh-my-pi/pi-catalog/types";
2
2
  import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
3
  export declare function generateHostedImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,4 @@
1
+ import type { CredentialRankingStrategy, UsageProvider } from "../usage.js";
2
+ export declare const commandCodeUsageProvider: UsageProvider;
3
+ /** Ranks Command Code accounts by the 5-hour and weekly credit windows. */
4
+ export declare const commandCodeRankingStrategy: CredentialRankingStrategy;
@@ -5,6 +5,13 @@ export interface LocalWorkSource {
5
5
  }
6
6
  export declare class EventStream<T, R = T> implements AsyncIterable<T> {
7
7
  #private;
8
+ /**
9
+ * Events pushed while no consumer was waiting. The iterator dequeues by
10
+ * advancing {@link #queueHead} instead of `shift()` — O(remaining) per event,
11
+ * quadratic for a consumer draining a backlog — so while it drains, the
12
+ * slots before the head are consumed (cleared) placeholders. Do not mutate
13
+ * the array while the stream is being iterated.
14
+ */
8
15
  queue: T[];
9
16
  waiting: Array<{
10
17
  resolve: (value: IteratorResult<T>) => void;
@@ -28,8 +28,14 @@ export interface OpenAIStreamRequestInit {
28
28
  fetch?: FetchImpl;
29
29
  /** Optional caller-specific gate composed with shared transport retry exclusions. */
30
30
  shouldRetryResponse?: (response: Response, bodyText: string) => boolean | Promise<boolean>;
31
- /** Raw wire-frame observer (`onSseEvent` debug pipeline). */
31
+ /**
32
+ * Raw wire-frame observer (`onSseEvent` debug pipeline). Leave it unset
33
+ * when no diagnostic listener exists: any observer turns on per-line raw
34
+ * capture for every frame.
35
+ */
32
36
  onSseEvent?: SseEventObserver;
37
+ /** Called when the stream ends on the OpenAI `[DONE]` sentinel; independent of {@link onSseEvent}. */
38
+ onDoneSentinel?: () => void;
33
39
  }
34
40
  export interface OpenAIStreamHandle<TEvent> {
35
41
  /** Decoded `data:` payloads; terminates on `[DONE]` or stream end. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.4.2",
3
+ "version": "18.4.3",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -155,11 +155,11 @@
155
155
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
156
156
  },
157
157
  "dependencies": {
158
- "@oh-my-pi/omptype": "18.4.2",
159
- "@oh-my-pi/pi-catalog": "18.4.2",
160
- "@oh-my-pi/pi-natives": "18.4.2",
161
- "@oh-my-pi/pi-utils": "18.4.2",
162
- "@oh-my-pi/pi-wire": "18.4.2"
158
+ "@oh-my-pi/omptype": "18.4.3",
159
+ "@oh-my-pi/pi-catalog": "18.4.3",
160
+ "@oh-my-pi/pi-natives": "18.4.3",
161
+ "@oh-my-pi/pi-utils": "18.4.3",
162
+ "@oh-my-pi/pi-wire": "18.4.3"
163
163
  },
164
164
  "devDependencies": {
165
165
  "@types/bun": "^1.3.14"
@@ -36,6 +36,18 @@ const TAGS: readonly Tag[] = [
36
36
  const OPENS = TAGS.map(tag => tag.open);
37
37
  const IMPLIED_OPEN_TAGS = TAGS.filter(tag => tag.impliedOpen);
38
38
  const IMPLIED_OPEN_DELIMITERS = [...OPENS, ...IMPLIED_OPEN_TAGS.map(tag => tag.close)];
39
+ /** A hold needs the buffer tail to be a proper prefix of some delimiter, so it must be shorter than this. */
40
+ const MAX_DELIMITER_LENGTH = Math.max(...IMPLIED_OPEN_DELIMITERS.map(delimiter => delimiter.length));
41
+ const BACKTICK = 0x60;
42
+ const BOUNDARY_LEAD_CODES = [...IMPLIED_OPEN_DELIMITERS.map(delimiter => delimiter.charCodeAt(0)), BACKTICK];
43
+ /**
44
+ * `1` at the char code of every character a {@link scanVisible} boundary can
45
+ * start on: the first character of any delimiter (a tag open, an implied close,
46
+ * or a partial of either) and the backtick. Every other character — including
47
+ * any code past the table — is skipped without comparing.
48
+ */
49
+ const BOUNDARY_LEAD = new Uint8Array(Math.max(...BOUNDARY_LEAD_CODES) + 1);
50
+ for (const code of BOUNDARY_LEAD_CODES) BOUNDARY_LEAD[code] = 1;
39
51
 
40
52
  export interface ThinkingInbandScannerOptions {
41
53
  /**
@@ -240,20 +252,26 @@ type VisibleHit =
240
252
  */
241
253
  function scanVisible(buffer: string, final: boolean, impliedOpen: boolean): VisibleHit {
242
254
  const delimiters = impliedOpen ? IMPLIED_OPEN_DELIMITERS : OPENS;
255
+ // Only the tail can be a proper prefix of a delimiter.
256
+ const holdFrom = final ? buffer.length : buffer.length - MAX_DELIMITER_LENGTH + 1;
243
257
  for (let i = 0; i < buffer.length; i++) {
244
- const tag = TAGS.find(candidate => buffer.startsWith(candidate.open, i));
245
- if (tag) return { kind: "tag", index: i, tag };
258
+ const code = buffer.charCodeAt(i);
259
+ if (code >= BOUNDARY_LEAD.length || BOUNDARY_LEAD[code] === 0) continue;
260
+ for (const tag of TAGS) {
261
+ if (buffer.startsWith(tag.open, i)) return { kind: "tag", index: i, tag };
262
+ }
246
263
  if (impliedOpen) {
247
- const closed = IMPLIED_OPEN_TAGS.find(candidate => buffer.startsWith(candidate.close, i));
248
- if (closed) return { kind: "impliedClose", index: i, tag: closed };
264
+ for (const tag of IMPLIED_OPEN_TAGS) {
265
+ if (buffer.startsWith(tag.close, i)) return { kind: "impliedClose", index: i, tag };
266
+ }
249
267
  }
250
- if (!final) {
268
+ if (i >= holdFrom) {
251
269
  const rest = buffer.slice(i);
252
270
  if (delimiters.some(delimiter => delimiter.length > rest.length && delimiter.startsWith(rest))) {
253
271
  return { kind: "hold", index: i };
254
272
  }
255
273
  }
256
- if (buffer[i] === "`") {
274
+ if (code === BACKTICK) {
257
275
  const ticks = backtickRun(buffer, i);
258
276
  if (!final && i + ticks === buffer.length) return { kind: "hold", index: i };
259
277
  return { kind: "code", index: i, ticks };
@@ -1,4 +1,4 @@
1
- import type { Api, Model } from "@oh-my-pi/pi-catalog/types";
1
+ import { type Api, type Model, modelKind } from "@oh-my-pi/pi-catalog/types";
2
2
  import {
3
3
  applyCodexResidencyHeader,
4
4
  CODEX_BASE_URL,
@@ -155,7 +155,10 @@ export async function generateHostedImage(
155
155
  action: content.length > 1 ? "edit" : "generate",
156
156
  output_format: "webp",
157
157
  ...(size ? { size } : {}),
158
- ...(model.api === "openai-responses" ? { model: model.requestModelId ?? model.id } : {}),
158
+ // A chat model generating on its own lets the host pick the image model.
159
+ ...(model.api === "openai-responses" && modelKind(model) === "image"
160
+ ? { model: model.requestModelId ?? model.id }
161
+ : {}),
159
162
  };
160
163
  const body = {
161
164
  model: carrier.requestModelId ?? carrier.id,
@@ -446,6 +446,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
446
446
  };
447
447
 
448
448
  const blocks = output.content as Block[];
449
+ const contentIndexByBlockIndex = new Map<number, number>();
449
450
  let rawRequestDump: RawHttpRequestDump | undefined;
450
451
  const region = resolveBedrockRegion(model.id, options);
451
452
 
@@ -640,7 +641,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
640
641
  if (messageType === "exception") {
641
642
  const exceptionType = message.headers[":exception-type"] || "Exception";
642
643
  const payload = safeParsePayload(message.payload) as { message?: string } | undefined;
643
- const errorMessage = payload?.message || new TextDecoder().decode(message.payload);
644
+ const errorMessage = payload?.message || PAYLOAD_DECODER.decode(message.payload);
644
645
  const text = `${exceptionType}: ${errorMessage}`;
645
646
  throw new AIError.BedrockApiError(text, bedrockStreamExceptionStatus(exceptionType), {
646
647
  code: exceptionType,
@@ -648,7 +649,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
648
649
  }
649
650
  if (messageType === "error") {
650
651
  const code = message.headers[":error-code"] || "UnknownError";
651
- const errorMessage = message.headers[":error-message"] || new TextDecoder().decode(message.payload);
652
+ const errorMessage = message.headers[":error-message"] || PAYLOAD_DECODER.decode(message.payload);
652
653
  throw new AIError.BedrockApiError(`${code}: ${errorMessage}`, bedrockStreamExceptionStatus(code), {
653
654
  code,
654
655
  });
@@ -673,16 +674,35 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
673
674
  }
674
675
  case "contentBlockStart": {
675
676
  if (!firstTokenTime) firstTokenTime = performance.now();
676
- handleContentBlockStart(payload as ContentBlockStartEvent, blocks, output, stream, sentinelInjected);
677
+ handleContentBlockStart(
678
+ payload as ContentBlockStartEvent,
679
+ blocks,
680
+ contentIndexByBlockIndex,
681
+ output,
682
+ stream,
683
+ sentinelInjected,
684
+ );
677
685
  break;
678
686
  }
679
687
  case "contentBlockDelta": {
680
688
  if (!firstTokenTime) firstTokenTime = performance.now();
681
- handleContentBlockDelta(payload as ContentBlockDeltaEvent, blocks, output, stream);
689
+ handleContentBlockDelta(
690
+ payload as ContentBlockDeltaEvent,
691
+ blocks,
692
+ contentIndexByBlockIndex,
693
+ output,
694
+ stream,
695
+ );
682
696
  break;
683
697
  }
684
698
  case "contentBlockStop": {
685
- handleContentBlockStop(payload as ContentBlockStopEvent, blocks, output, stream);
699
+ handleContentBlockStop(
700
+ payload as ContentBlockStopEvent,
701
+ blocks,
702
+ contentIndexByBlockIndex,
703
+ output,
704
+ stream,
705
+ );
686
706
  break;
687
707
  }
688
708
  case "messageStop": {
@@ -782,18 +802,40 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
782
802
  return stream;
783
803
  };
784
804
 
805
+ /** Shared across events: every payload decode is a complete, non-streaming call. */
806
+ const PAYLOAD_DECODER = new TextDecoder();
807
+
785
808
  function safeParsePayload(payload: Uint8Array): unknown {
786
809
  if (payload.length === 0) return {};
787
810
  try {
788
- return JSON.parse(new TextDecoder().decode(payload));
811
+ return JSON.parse(PAYLOAD_DECODER.decode(payload));
789
812
  } catch {
790
813
  return undefined;
791
814
  }
792
815
  }
793
816
 
817
+ /**
818
+ * Append a streamed block and index it by Bedrock's `contentBlockIndex`, so
819
+ * per-delta routing is an O(1) lookup instead of a scan over every block
820
+ * (quadratic over a long turn). The first block registered for an index wins,
821
+ * as the first-match scan did. Returns the block's content index.
822
+ */
823
+ function pushStreamBlock(
824
+ blocks: Block[],
825
+ contentIndexByBlockIndex: Map<number, number>,
826
+ block: Block,
827
+ contentBlockIndex: number,
828
+ ): number {
829
+ const contentIndex = blocks.length;
830
+ blocks.push(block);
831
+ if (!contentIndexByBlockIndex.has(contentBlockIndex)) contentIndexByBlockIndex.set(contentBlockIndex, contentIndex);
832
+ return contentIndex;
833
+ }
834
+
794
835
  function handleContentBlockStart(
795
836
  event: ContentBlockStartEvent,
796
837
  blocks: Block[],
838
+ contentIndexByBlockIndex: Map<number, number>,
797
839
  output: AssistantMessage,
798
840
  stream: AssistantMessageEventStream,
799
841
  sentinelInjected: boolean,
@@ -815,29 +857,29 @@ function handleContentBlockStart(
815
857
  [kStreamingPartialJson]: "",
816
858
  [kStreamingBlockIndex]: index,
817
859
  };
818
- output.content.push(block);
819
- stream.push({ type: "toolcall_start", contentIndex: blocks.length - 1, partial: output });
860
+ const contentIndex = pushStreamBlock(blocks, contentIndexByBlockIndex, block, index);
861
+ stream.push({ type: "toolcall_start", contentIndex, partial: output });
820
862
  }
821
863
  }
822
864
 
823
865
  function handleContentBlockDelta(
824
866
  event: ContentBlockDeltaEvent,
825
867
  blocks: Block[],
868
+ contentIndexByBlockIndex: Map<number, number>,
826
869
  output: AssistantMessage,
827
870
  stream: AssistantMessageEventStream,
828
871
  ): void {
829
872
  const contentBlockIndex = event.contentBlockIndex;
830
873
  const delta = event.delta;
831
- let index = blocks.findIndex(b => b[kStreamingBlockIndex] === contentBlockIndex);
874
+ let index = contentIndexByBlockIndex.get(contentBlockIndex) ?? -1;
832
875
  let block = blocks[index];
833
876
 
834
877
  if (delta?.text !== undefined) {
835
878
  // If no text block exists yet, create one — `handleContentBlockStart` is not sent for text blocks
836
879
  if (!block) {
837
880
  const newBlock: Block = { type: "text", text: "", [kStreamingBlockIndex]: contentBlockIndex };
838
- output.content.push(newBlock);
839
- index = blocks.length - 1;
840
- block = blocks[index];
881
+ index = pushStreamBlock(blocks, contentIndexByBlockIndex, newBlock, contentBlockIndex);
882
+ block = newBlock;
841
883
  stream.push({ type: "text_start", contentIndex: index, partial: output });
842
884
  }
843
885
  if (block.type === "text") {
@@ -863,9 +905,8 @@ function handleContentBlockDelta(
863
905
  thinkingSignature: "",
864
906
  [kStreamingBlockIndex]: contentBlockIndex,
865
907
  };
866
- output.content.push(newBlock);
867
- thinkingIndex = blocks.length - 1;
868
- thinkingBlock = blocks[thinkingIndex];
908
+ thinkingIndex = pushStreamBlock(blocks, contentIndexByBlockIndex, newBlock, contentBlockIndex);
909
+ thinkingBlock = newBlock;
869
910
  stream.push({ type: "thinking_start", contentIndex: thinkingIndex, partial: output });
870
911
  }
871
912
 
@@ -901,10 +942,11 @@ function handleMetadata(event: MetadataEvent, model: Model<"bedrock-converse-str
901
942
  function handleContentBlockStop(
902
943
  event: ContentBlockStopEvent,
903
944
  blocks: Block[],
945
+ contentIndexByBlockIndex: Map<number, number>,
904
946
  output: AssistantMessage,
905
947
  stream: AssistantMessageEventStream,
906
948
  ): void {
907
- const index = blocks.findIndex(b => b[kStreamingBlockIndex] === event.contentBlockIndex);
949
+ const index = contentIndexByBlockIndex.get(event.contentBlockIndex) ?? -1;
908
950
  const block = blocks[index];
909
951
  if (!block) return;
910
952
 
@@ -24,6 +24,8 @@ const PRELUDE_CRC_LEN = 4;
24
24
  const MESSAGE_CRC_LEN = 4;
25
25
  const HEADER_BLOCK_OFFSET = PRELUDE_LEN + PRELUDE_CRC_LEN;
26
26
  const MIN_MESSAGE_LEN = HEADER_BLOCK_OFFSET + MESSAGE_CRC_LEN;
27
+ /** Shared across messages: every header decode is a complete, non-streaming call. */
28
+ const HEADER_DECODER = new TextDecoder();
27
29
 
28
30
  export interface EventStreamMessage {
29
31
  /** Lower-cased copy is *not* applied — Bedrock uses casing like `:event-type` verbatim. */
@@ -64,12 +66,11 @@ export function decodeMessage(frame: Uint8Array): EventStreamMessage {
64
66
  function parseHeaders(buf: Uint8Array): Record<string, string> {
65
67
  const out: Record<string, string> = {};
66
68
  const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength);
67
- const decoder = new TextDecoder();
68
69
  let p = 0;
69
70
  while (p < buf.length) {
70
71
  const nameLen = view.getUint8(p);
71
72
  p += 1;
72
- const name = decoder.decode(buf.subarray(p, p + nameLen));
73
+ const name = HEADER_DECODER.decode(buf.subarray(p, p + nameLen));
73
74
  p += nameLen;
74
75
  const type = view.getUint8(p);
75
76
  p += 1;
@@ -108,7 +109,7 @@ function parseHeaders(buf: Uint8Array): Record<string, string> {
108
109
  // string
109
110
  const len = view.getUint16(p, false);
110
111
  p += 2;
111
- out[name] = decoder.decode(buf.subarray(p, p + len));
112
+ out[name] = HEADER_DECODER.decode(buf.subarray(p, p + len));
112
113
  p += len;
113
114
  break;
114
115
  }
@@ -1,11 +1,10 @@
1
- import { $env } from "@oh-my-pi/pi-utils";
1
+ import { $env, type ServerSentEvent } from "@oh-my-pi/pi-utils";
2
2
  import * as AIError from "../error";
3
3
  import { getEnvApiKey } from "../stream";
4
4
  import type {
5
5
  AssistantMessage,
6
6
  Context,
7
7
  Model,
8
- RawSseEvent,
9
8
  ServiceTier,
10
9
  StreamFunction,
11
10
  StreamOptions,
@@ -104,7 +103,7 @@ const streamAzureOpenAIResponsesOnce = (
104
103
  const { requestAbortController, requestSignal } = abortTracker;
105
104
  const onSseEvent = options?.onSseEvent;
106
105
  const rawSseObserver = onSseEvent
107
- ? (event: RawSseEvent) => {
106
+ ? (event: ServerSentEvent) => {
108
107
  if (!event.event && event.data && event.data !== "[DONE]") {
109
108
  try {
110
109
  const parsed = JSON.parse(event.data);
@@ -120,7 +119,7 @@ const streamAzureOpenAIResponsesOnce = (
120
119
  }
121
120
  } catch {}
122
121
  }
123
- onSseEvent(event, model);
122
+ onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
124
123
  }
125
124
  : undefined;
126
125
 
@@ -751,9 +751,16 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
751
751
  const responseSignal = options?.signal
752
752
  ? AbortSignal.any([options.signal, responseAbortController.signal])
753
753
  : responseAbortController.signal;
754
+ const onSseEvent = options?.onSseEvent;
754
755
  const chunks = iterateWithIdleTimeout(
755
- readSseJson<CloudCodeAssistResponseChunk>(activeResponse.body, responseSignal, event =>
756
- options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
756
+ // Attach the observer only when a diagnostic listener exists: any
757
+ // observer turns on per-line raw capture in `readSseJson`.
758
+ readSseJson<CloudCodeAssistResponseChunk>(
759
+ activeResponse.body,
760
+ responseSignal,
761
+ onSseEvent
762
+ ? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model)
763
+ : undefined,
757
764
  ),
758
765
  {
759
766
  firstItemTimeoutMs: firstEventTimeoutMs,
@@ -4,7 +4,7 @@
4
4
 
5
5
  import { scheduler } from "node:timers/promises";
6
6
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
7
- import { readSseJson } from "@oh-my-pi/pi-utils";
7
+ import { readSseJson, type SseEventObserver } from "@oh-my-pi/pi-utils";
8
8
  import { renderDemotedThinking } from "../dialect/demotion";
9
9
  import { ThinkingFenceStripper } from "../dialect/thinking-fence-strip";
10
10
  import * as AIError from "../error";
@@ -1014,13 +1014,17 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
1014
1014
  let body = await openStream();
1015
1015
  stream.push({ type: "start", partial: output });
1016
1016
 
1017
+ // Attach the observer only when a diagnostic listener exists: any
1018
+ // observer turns on per-line raw capture in `readSseJson`.
1019
+ const onSseEvent = options?.onSseEvent;
1020
+ const sseObserver: SseEventObserver | undefined = onSseEvent
1021
+ ? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model)
1022
+ : undefined;
1017
1023
  // Gemini occasionally finishes with `finishReason: STOP` while emitting only an empty
1018
1024
  // text part and no tool call. Delivered as-is the agent receives a blank message and
1019
1025
  // silently halts mid-task, so retry a bounded number of times before giving up.
1020
1026
  for (let emptyAttempt = 0; ; emptyAttempt++) {
1021
- const googleStream = readSseJson<GenerateContentResponse>(body, options?.signal, event =>
1022
- options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
1023
- );
1027
+ const googleStream = readSseJson<GenerateContentResponse>(body, options?.signal, sseObserver);
1024
1028
  await consumeGoogleStream({
1025
1029
  googleStream,
1026
1030
  output,
@@ -4730,8 +4730,14 @@ async function openCodexSseEventStream(
4730
4730
  if (!response.body) {
4731
4731
  throw new CodexProviderStreamError("No response body", false);
4732
4732
  }
4733
- return readSseJson<Record<string, unknown>>(response.body, signal, event =>
4734
- onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, undefined),
4733
+ // Attach the observer only when a diagnostic listener exists: any observer
4734
+ // turns on per-line raw capture in `readSseJson`.
4735
+ return readSseJson<Record<string, unknown>>(
4736
+ response.body,
4737
+ signal,
4738
+ onSseEvent
4739
+ ? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, undefined)
4740
+ : undefined,
4735
4741
  );
4736
4742
  }
4737
4743
 
@@ -4,7 +4,13 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
4
4
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
5
5
  import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
6
6
  import { clinePassClientHeaders } from "@oh-my-pi/pi-catalog/wire/cline-pass";
7
- import { $env, logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
7
+ import {
8
+ $env,
9
+ logger,
10
+ parseStreamingJson,
11
+ parseStreamingJsonThrottled,
12
+ type ServerSentEvent,
13
+ } from "@oh-my-pi/pi-utils";
8
14
  import { renderDemotedThinking } from "../dialect/demotion";
9
15
  import * as AIError from "../error";
10
16
  import { getKimiCommonHeaders } from "../registry/oauth/kimi";
@@ -16,7 +22,6 @@ import type {
16
22
  MessageAttribution,
17
23
  Model,
18
24
  ProviderSessionState,
19
- RawSseEvent,
20
25
  ServiceTier,
21
26
  StopReason,
22
27
  StreamFunction,
@@ -800,28 +805,29 @@ const streamOpenAICompletionsOnce = (
800
805
  // Track the OpenAI `[DONE]` sentinel independently of `onSseEvent`: it is
801
806
  // the streaming protocol's terminal signal, so a stream that ends with it
802
807
  // completed by server agreement even when no `finish_reason` chunk arrived.
808
+ // It arrives through `onDoneSentinel`, so the diagnostic observer below
809
+ // stays unset (and raw wire-line capture off) when nobody listens.
803
810
  let sawDoneSentinel = false;
804
- const rawSseObserver = (event: RawSseEvent) => {
805
- if (event.data === "[DONE]") sawDoneSentinel = true;
806
- if (onSseEvent) {
807
- if (!event.event && event.data && event.data !== "[DONE]") {
808
- try {
809
- const parsed = JSON.parse(event.data);
810
- const resolvedEvent =
811
- typeof parsed.type === "string"
812
- ? parsed.type
813
- : typeof parsed.object === "string"
814
- ? parsed.object
815
- : null;
816
- if (resolvedEvent) {
817
- event.event = resolvedEvent;
818
- event.raw = [`event: ${resolvedEvent}`, ...event.raw];
819
- }
820
- } catch {}
811
+ const rawSseObserver = onSseEvent
812
+ ? (event: ServerSentEvent) => {
813
+ if (!event.event && event.data && event.data !== "[DONE]") {
814
+ try {
815
+ const parsed = JSON.parse(event.data);
816
+ const resolvedEvent =
817
+ typeof parsed.type === "string"
818
+ ? parsed.type
819
+ : typeof parsed.object === "string"
820
+ ? parsed.object
821
+ : null;
822
+ if (resolvedEvent) {
823
+ event.event = resolvedEvent;
824
+ event.raw = [`event: ${resolvedEvent}`, ...event.raw];
825
+ }
826
+ } catch {}
827
+ }
828
+ onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
821
829
  }
822
- onSseEvent(event, model);
823
- }
824
- };
830
+ : undefined;
825
831
  // Assigned once the block helpers exist (they are scoped to the `try`);
826
832
  // the catch handler uses it to close open blocks before emitting the
827
833
  // terminal error so both exit paths obey the same block lifecycle.
@@ -931,6 +937,9 @@ const streamOpenAICompletionsOnce = (
931
937
  // bounds every attempt and backoff sleep — retries cannot
932
938
  // extend the deadline.
933
939
  onSseEvent: rawSseObserver,
940
+ onDoneSentinel: () => {
941
+ sawDoneSentinel = true;
942
+ },
934
943
  });
935
944
  // Disarm the first-event watchdog as soon as headers arrive — a slow
936
945
  // onResponse callback must not abort an already-connected stream.
@@ -1030,9 +1039,19 @@ const streamOpenAICompletionsOnce = (
1030
1039
  };
1031
1040
  let currentBlock: OpenAIStreamBlock | undefined;
1032
1041
  let messageThoughtSignature: GeminiMessageThoughtSignature | undefined;
1042
+ // Content blocks are append-only for the lifetime of the stream, so each
1043
+ // block's index is stable once pushed. Map block → index to keep the
1044
+ // per-delta contentIndex lookup O(1): a linear `indexOf` per delta turns a
1045
+ // long turn (many blocks × many deltas) quadratic, as openai-shared's
1046
+ // Responses decoder documents (issue #10605).
1047
+ const contentIndexByBlock = new Map<OpenAIStreamBlock, number>();
1048
+ const pushContentBlock = (block: OpenAIStreamBlock): void => {
1049
+ contentIndexByBlock.set(block, output.content.length);
1050
+ output.content.push(block);
1051
+ };
1033
1052
  const blockIndex = (block: OpenAIStreamBlock | undefined): number => {
1034
1053
  if (!block) return Math.max(0, output.content.length - 1);
1035
- return output.content.indexOf(block);
1054
+ return contentIndexByBlock.get(block) ?? output.content.indexOf(block);
1036
1055
  };
1037
1056
  const finishToolCallBlock = (block: ToolCallStreamBlock): void => {
1038
1057
  if (block.partialArgs === undefined) return;
@@ -1087,11 +1106,7 @@ const streamOpenAICompletionsOnce = (
1087
1106
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1088
1107
  finishPendingToolCallBlocks();
1089
1108
  };
1090
- const appendText = (
1091
- message: AssistantMessage,
1092
- eventStream: AssistantMessageEventStream,
1093
- text: string,
1094
- ): void => {
1109
+ const appendText = (text: string): void => {
1095
1110
  if (currentBlock?.type !== "text") {
1096
1111
  // Leave toolCall blocks pending across text transitions: chunks after
1097
1112
  // the first typically carry only `index`, so a finished (de-registered)
@@ -1099,15 +1114,15 @@ const streamOpenAICompletionsOnce = (
1099
1114
  // resume. The stream-end sweep finalizes pending calls.
1100
1115
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1101
1116
  currentBlock = { type: "text", text: "" };
1102
- message.content.push(currentBlock);
1103
- eventStream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: message });
1117
+ pushContentBlock(currentBlock);
1118
+ stream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: output });
1104
1119
  }
1105
1120
  currentBlock.text += text;
1106
- eventStream.push({
1121
+ stream.push({
1107
1122
  type: "text_delta",
1108
1123
  contentIndex: blockIndex(currentBlock),
1109
1124
  delta: text,
1110
- partial: message,
1125
+ partial: output,
1111
1126
  });
1112
1127
  };
1113
1128
  const openThinkingBlock = (signature?: string): ThinkingContent => {
@@ -1116,7 +1131,7 @@ const streamOpenAICompletionsOnce = (
1116
1131
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1117
1132
  const block: ThinkingContent = { type: "thinking", thinking: "", thinkingSignature: signature };
1118
1133
  currentBlock = block;
1119
- output.content.push(block);
1134
+ pushContentBlock(block);
1120
1135
  stream.push({ type: "thinking_start", contentIndex: blockIndex(block), partial: output });
1121
1136
  return block;
1122
1137
  };
@@ -1177,7 +1192,7 @@ const streamOpenAICompletionsOnce = (
1177
1192
  const appendTextDelta = (text: string): void => {
1178
1193
  if (!text) return;
1179
1194
  if (!firstTokenTime) firstTokenTime = performance.now();
1180
- appendText(output, stream, text);
1195
+ appendText(text);
1181
1196
  };
1182
1197
  // Tracks the last full cumulative reasoning snapshot per signature (the
1183
1198
  // reasoning field name) so dedup survives block transitions. Required
@@ -1249,7 +1264,7 @@ const streamOpenAICompletionsOnce = (
1249
1264
  };
1250
1265
  block.arguments = parseStreamingJson(call.arguments);
1251
1266
  currentBlock = block;
1252
- output.content.push(block);
1267
+ pushContentBlock(block);
1253
1268
  stream.push({ type: "toolcall_start", contentIndex: blockIndex(block), partial: output });
1254
1269
  stream.push({
1255
1270
  type: "toolcall_delta",
@@ -1473,7 +1488,7 @@ const streamOpenAICompletionsOnce = (
1473
1488
  if (streamIndex !== undefined) toolCallBlockByIndex.set(streamIndex, block);
1474
1489
  pendingToolCallBlocks.push(block);
1475
1490
  currentBlock = block;
1476
- output.content.push(block);
1491
+ pushContentBlock(block);
1477
1492
  stream.push({
1478
1493
  type: "toolcall_start",
1479
1494
  contentIndex: blockIndex(block),
@@ -1,5 +1,5 @@
1
1
  import { scheduler } from "node:timers/promises";
2
- import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
2
+ import { $flag, logger, type ServerSentEvent, structuredCloneJSON } from "@oh-my-pi/pi-utils";
3
3
  import * as AIError from "../error";
4
4
  import { getEnvApiKey } from "../stream";
5
5
  import type {
@@ -9,7 +9,6 @@ import type {
9
9
  Model,
10
10
  OpenAICompat,
11
11
  ProviderSessionState,
12
- RawSseEvent,
13
12
  ServiceTier,
14
13
  StreamFunction,
15
14
  StreamOptions,
@@ -455,7 +454,7 @@ const streamOpenAIResponsesOnce = (
455
454
  const { requestAbortController, requestSignal } = abortTracker;
456
455
  const onSseEvent = options?.onSseEvent;
457
456
  const rawSseObserver = onSseEvent
458
- ? (event: RawSseEvent) => {
457
+ ? (event: ServerSentEvent) => {
459
458
  if (!event.event && event.data && event.data !== "[DONE]") {
460
459
  try {
461
460
  const parsed = JSON.parse(event.data);
@@ -471,7 +470,7 @@ const streamOpenAIResponsesOnce = (
471
470
  }
472
471
  } catch {}
473
472
  }
474
- onSseEvent(event, model);
473
+ onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
475
474
  }
476
475
  : undefined;
477
476
 
@@ -1667,20 +1667,58 @@ function classifyResponsesBatchItem(item: object): ResponsesBatchItemKind {
1667
1667
  * strict validator. See #8789.
1668
1668
  */
1669
1669
  export function hoistInterleavedResponsesToolBatchMessages<T extends object>(items: readonly T[]): T[] {
1670
- const moved = new Set<number>();
1670
+ const callIdOf = (item: T): string | undefined =>
1671
+ "call_id" in item && typeof item.call_id === "string" ? item.call_id : undefined;
1672
+ // Does a call with `callId` precede `index` within the same contiguous batch body?
1673
+ const hasEarlierBatchCall = (index: number, callId: string): boolean => {
1674
+ for (let probe = index - 1; probe >= 0; probe--) {
1675
+ const kind = classifyResponsesBatchItem(items[probe]);
1676
+ if (kind === "other") return false;
1677
+ if (kind === "call" && callIdOf(items[probe]) === callId) return true;
1678
+ }
1679
+ return false;
1680
+ };
1681
+ const bucketOf = new Map<number, number>();
1671
1682
  const insertBefore = new Map<number, number[]>();
1672
1683
  for (let index = 0; index < items.length; index++) {
1673
1684
  if (classifyResponsesBatchItem(items[index]) !== "output") continue;
1674
1685
  // Only anchor on the first output of a run.
1675
1686
  if (index > 0 && classifyResponsesBatchItem(items[index - 1]) === "output") continue;
1687
+ // Calls the batch still owns further back: the anchor run's outputs, plus
1688
+ // any earlier output crossed on the way.
1689
+ const pending = new Set<string>();
1690
+ for (let probe = index; probe < items.length; probe++) {
1691
+ if (classifyResponsesBatchItem(items[probe]) !== "output") break;
1692
+ const callId = callIdOf(items[probe]);
1693
+ if (callId) pending.add(callId);
1694
+ }
1676
1695
  // Walk back over the batch body (calls interleaved with assistant messages).
1696
+ // An earlier output is crossed only when it and a call the batch still owns
1697
+ // both pair with calls further back — i.e. the output belongs to this same
1698
+ // interrupted batch (#13083). Otherwise it closes a completed prior round,
1699
+ // whose trailing messages stay put.
1677
1700
  let start = index;
1678
1701
  let sawCall = false;
1679
1702
  const messageIndexes: number[] = [];
1680
1703
  while (start > 0) {
1681
- const kind = classifyResponsesBatchItem(items[start - 1]);
1704
+ const item = items[start - 1];
1705
+ const kind = classifyResponsesBatchItem(item);
1682
1706
  if (kind === "call") {
1683
1707
  sawCall = true;
1708
+ const callId = callIdOf(item);
1709
+ if (callId) pending.delete(callId);
1710
+ } else if (kind === "output") {
1711
+ const callId = callIdOf(item);
1712
+ if (!callId || !hasEarlierBatchCall(start - 1, callId)) break;
1713
+ let ownsEarlierCall = false;
1714
+ for (const owned of pending) {
1715
+ if (hasEarlierBatchCall(start - 1, owned)) {
1716
+ ownsEarlierCall = true;
1717
+ break;
1718
+ }
1719
+ }
1720
+ if (!ownsEarlierCall) break;
1721
+ pending.add(callId);
1684
1722
  } else if (kind === "assistant-message") {
1685
1723
  messageIndexes.push(start - 1);
1686
1724
  } else {
@@ -1693,17 +1731,25 @@ export function hoistInterleavedResponsesToolBatchMessages<T extends object>(ite
1693
1731
  messageIndexes.reverse();
1694
1732
  const target = insertBefore.get(start) ?? [];
1695
1733
  for (const messageIndex of messageIndexes) {
1696
- moved.add(messageIndex);
1734
+ // A wider batch can re-collect a message an earlier anchor already
1735
+ // scheduled; move it rather than emitting it twice.
1736
+ const previousStart = bucketOf.get(messageIndex);
1737
+ if (previousStart !== undefined) {
1738
+ const previous = insertBefore.get(previousStart);
1739
+ const slot = previous?.indexOf(messageIndex) ?? -1;
1740
+ if (previous && slot >= 0) previous.splice(slot, 1);
1741
+ }
1742
+ bucketOf.set(messageIndex, start);
1697
1743
  target.push(messageIndex);
1698
1744
  }
1699
1745
  insertBefore.set(start, target);
1700
1746
  }
1701
- if (moved.size === 0) return items.slice();
1747
+ if (bucketOf.size === 0) return items.slice();
1702
1748
  const result: T[] = [];
1703
1749
  for (let index = 0; index < items.length; index++) {
1704
1750
  const pending = insertBefore.get(index);
1705
1751
  if (pending) for (const messageIndex of pending) result.push(items[messageIndex]);
1706
- if (moved.has(index)) continue;
1752
+ if (bucketOf.has(index)) continue;
1707
1753
  result.push(items[index]);
1708
1754
  }
1709
1755
  return result;
@@ -2070,11 +2116,16 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
2070
2116
  },
2071
2117
  );
2072
2118
  const sanitizedHistoryItems = rawSanitizedHistoryItems
2073
- ? adaptResponsesReplayItemsForModel(
2074
- rawSanitizedHistoryItems,
2075
- supportsCustomToolCalls,
2076
- customToolWireNameMap,
2077
- options.model.supportsComputerUse === true,
2119
+ ? ensureRequiredResponsesReasoningReplay(
2120
+ adaptResponsesReplayItemsForModel(
2121
+ rawSanitizedHistoryItems,
2122
+ supportsCustomToolCalls,
2123
+ customToolWireNameMap,
2124
+ options.model.supportsComputerUse === true,
2125
+ ),
2126
+ assistantMsg.stopReason,
2127
+ options.requiresReasoningReplayForAllTurns ?? false,
2128
+ options.requiresReasoningReplayForToolCalls ?? false,
2078
2129
  )
2079
2130
  : undefined;
2080
2131
  if (nativeReplayEnabled && sanitizedHistoryItems) {
@@ -2175,6 +2226,70 @@ function parseResponseReasoningReplayItem(signature: string | undefined): Respon
2175
2226
  */
2176
2227
  export const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
2177
2228
 
2229
+ function createSyntheticResponsesReasoningItem(
2230
+ text = SYNTHETIC_REASONING_REPLAY_PLACEHOLDER,
2231
+ id?: string,
2232
+ ): ResponseReasoningItem {
2233
+ const item = {
2234
+ type: "reasoning",
2235
+ ...(id ? { id } : {}),
2236
+ summary: [],
2237
+ content: [{ type: "reasoning_text", text }],
2238
+ } satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
2239
+ // The vendored SDK type marks `id` required; the wire accepts its absence.
2240
+ return item as ResponseReasoningItem;
2241
+ }
2242
+
2243
+ function isResponsesAssistantTurnBoundary(item: ResponseInput[number]): boolean {
2244
+ if (responsesToolOutputKind(item.type) !== undefined) return true;
2245
+ if (item.type === "compaction") return true;
2246
+ return "role" in item && item.role !== "assistant";
2247
+ }
2248
+
2249
+ function ensureRequiredResponsesReasoningReplay(
2250
+ items: ResponseInput,
2251
+ stopReason: AssistantMessage["stopReason"],
2252
+ requiresAllTurns: boolean,
2253
+ requiresToolCalls: boolean,
2254
+ ): ResponseInput {
2255
+ if (stopReason === "error" || (!requiresAllTurns && !requiresToolCalls)) return items;
2256
+
2257
+ const insertBefore: number[] = [];
2258
+ let turnStart = 0;
2259
+ for (let index = 0; index <= items.length; index++) {
2260
+ if (index < items.length && !isResponsesAssistantTurnBoundary(items[index])) continue;
2261
+
2262
+ let hasContent = false;
2263
+ let hasReasoning = false;
2264
+ let hasToolCall = false;
2265
+ for (let turnIndex = turnStart; turnIndex < index; turnIndex++) {
2266
+ const item = items[turnIndex];
2267
+ if (item.type === "reasoning") {
2268
+ hasReasoning = true;
2269
+ continue;
2270
+ }
2271
+ hasContent = true;
2272
+ if (classifyResponsesBatchItem(item) === "call") hasToolCall = true;
2273
+ }
2274
+ if (hasContent && !hasReasoning && (requiresAllTurns || (requiresToolCalls && hasToolCall))) {
2275
+ insertBefore.push(turnStart);
2276
+ }
2277
+ turnStart = index + 1;
2278
+ }
2279
+ if (insertBefore.length === 0) return items;
2280
+
2281
+ const repaired: ResponseInput = [];
2282
+ let insertionIndex = 0;
2283
+ for (let index = 0; index < items.length; index++) {
2284
+ if (insertBefore[insertionIndex] === index) {
2285
+ repaired.push(createSyntheticResponsesReasoningItem());
2286
+ insertionIndex++;
2287
+ }
2288
+ repaired.push(items[index]);
2289
+ }
2290
+ return repaired;
2291
+ }
2292
+
2178
2293
  export function convertResponsesAssistantMessage<TApi extends Api>(
2179
2294
  assistantMsg: AssistantMessage,
2180
2295
  model: Model<TApi>,
@@ -2345,14 +2460,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
2345
2460
  const carriedReasoningText = carriedReasoningTexts.join("\n");
2346
2461
  const reasoningText =
2347
2462
  carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
2348
- const reasoningItem = {
2349
- type: "reasoning",
2350
- ...(synthesizedReasoningItemId ? { id: synthesizedReasoningItemId } : {}),
2351
- summary: [],
2352
- content: [{ type: "reasoning_text", text: reasoningText }],
2353
- } satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
2354
- // The vendored SDK type marks `id` required; the wire accepts its absence.
2355
- outputItems.unshift(reasoningItem as ResponseReasoningItem);
2463
+ outputItems.unshift(createSyntheticResponsesReasoningItem(reasoningText, synthesizedReasoningItemId));
2356
2464
  }
2357
2465
 
2358
2466
  return outputItems;
@@ -96,9 +96,14 @@ export function createApiKeyLogin(
96
96
  try {
97
97
  await runValidation(rule.validate, label, trimmed, options);
98
98
  } catch (error) {
99
- // An optional probe only rejects on a real auth failure (401/403);
100
- // any other validation-endpoint failure trusts the supplied key.
101
- if (!rule.validate.optional || AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)) {
99
+ // An optional probe only rejects on a real auth failure (401/403, or
100
+ // only 401 with `trustForbidden`); any other validation-endpoint
101
+ // failure trusts the supplied key.
102
+ const trustsForbidden = rule.validate.trustForbidden && AIError.status(error) === 403;
103
+ if (
104
+ !rule.validate.optional ||
105
+ (AIError.is(AIError.classify(error), AIError.Flag.AuthFailed) && !trustsForbidden)
106
+ ) {
102
107
  throw error;
103
108
  }
104
109
  options.onProgress?.(`Skipping ${label} validation endpoint; continuing with provided API key.`);
@@ -0,0 +1,209 @@
1
+ import { ProviderHttpError } from "../error";
2
+ import type {
3
+ CredentialRankingStrategy,
4
+ UsageFetchContext,
5
+ UsageFetchParams,
6
+ UsageLimit,
7
+ UsageProvider,
8
+ UsageReport,
9
+ } from "../usage";
10
+ import { isRecord } from "../utils";
11
+ import { HOUR_MS, parsePositiveTimestamp, usageStatus, WEEK_MS } from "./shared";
12
+
13
+ const PROVIDER = "commandcode";
14
+ const DEFAULT_ORIGIN = "https://api.commandcode.ai";
15
+ // The CLI also sends `?limits=1` for its org spend-limit panel; omp reads no
16
+ // field from that, so the plain identity route is enough.
17
+ const WHOAMI_PATH = "/alpha/whoami";
18
+ const CREDITS_PATH = "/alpha/billing/credits";
19
+
20
+ /**
21
+ * Command Code's inference base carries `/provider` or `/provider/v1`, while the
22
+ * account routes sit at the host root, so only the configured origin is kept.
23
+ * A blank or unparseable override means "not configured".
24
+ */
25
+ function resolveOrigin(baseUrl: string | undefined): string {
26
+ const trimmed = baseUrl?.trim();
27
+ if (!trimmed) return DEFAULT_ORIGIN;
28
+ try {
29
+ return new URL(trimmed).origin;
30
+ } catch {
31
+ return DEFAULT_ORIGIN;
32
+ }
33
+ }
34
+
35
+ function nonEmptyString(value: unknown): string | undefined {
36
+ return typeof value === "string" && value.length > 0 ? value : undefined;
37
+ }
38
+
39
+ function finiteNumber(value: unknown): number | undefined {
40
+ return typeof value === "number" && Number.isFinite(value) ? value : undefined;
41
+ }
42
+
43
+ /**
44
+ * Returns the unwrapped JSON body, or `null` for any failure that should read
45
+ * as "no data", 403 included. Only a 401 throws, so a revoked key purges the
46
+ * cached report instead of re-serving it.
47
+ */
48
+ async function getJson(
49
+ url: string,
50
+ apiKey: string,
51
+ signal: AbortSignal | undefined,
52
+ ctx: UsageFetchContext,
53
+ ): Promise<Record<string, unknown> | null> {
54
+ try {
55
+ const response = await ctx.fetch(url, {
56
+ headers: {
57
+ Authorization: `Bearer ${apiKey}`,
58
+ Accept: "application/json",
59
+ },
60
+ signal,
61
+ });
62
+ if (!response.ok) {
63
+ if (response.status === 401) {
64
+ throw new ProviderHttpError(
65
+ `Command Code usage endpoint returned ${response.status} ${response.statusText}`.trim(),
66
+ response.status,
67
+ );
68
+ }
69
+ ctx.logger?.warn("Command Code usage fetch failed", {
70
+ url,
71
+ status: response.status,
72
+ statusText: response.statusText,
73
+ });
74
+ return null;
75
+ }
76
+ const json: unknown = await response.json();
77
+ if (!isRecord(json)) return null;
78
+ return isRecord(json.data) ? json.data : json;
79
+ } catch (error) {
80
+ if (error instanceof ProviderHttpError) throw error;
81
+ ctx.logger?.warn("Command Code usage fetch error", { url, error: String(error) });
82
+ return null;
83
+ }
84
+ }
85
+
86
+ interface WindowSpec {
87
+ key: "fiveHour" | "weekly";
88
+ id: "5h" | "7d";
89
+ limitLabel: string;
90
+ windowLabel: string;
91
+ durationMs: number;
92
+ }
93
+
94
+ const WINDOWS: readonly WindowSpec[] = [
95
+ { key: "fiveHour", id: "5h", limitLabel: "5-hour limit", windowLabel: "5-hour", durationMs: 5 * HOUR_MS },
96
+ { key: "weekly", id: "7d", limitLabel: "Weekly limit", windowLabel: "Weekly", durationMs: WEEK_MS },
97
+ ];
98
+
99
+ function buildWindowLimit(
100
+ spec: WindowSpec,
101
+ raw: unknown,
102
+ accountId: string,
103
+ orgId: string | undefined,
104
+ ): UsageLimit | undefined {
105
+ if (!isRecord(raw)) return undefined;
106
+ const used = finiteNumber(raw.used);
107
+ if (used === undefined) return undefined;
108
+ const cap = finiteNumber(raw.cap);
109
+ const usedFraction = cap !== undefined && cap > 0 ? used / cap : undefined;
110
+ const resetsAt = parsePositiveTimestamp(raw.resetAt);
111
+ // Absent fields are omitted, not set to `undefined`: the usage wire schema
112
+ // rejects an explicit `undefined` for an optional key.
113
+ return {
114
+ id: `${PROVIDER}:${spec.id}`,
115
+ label: spec.limitLabel,
116
+ scope: { provider: PROVIDER, accountId, ...(orgId ? { orgId } : {}), windowId: spec.id, shared: true },
117
+ window: {
118
+ id: spec.id,
119
+ label: spec.windowLabel,
120
+ durationMs: spec.durationMs,
121
+ ...(resetsAt !== undefined ? { resetsAt } : {}),
122
+ },
123
+ amount: {
124
+ used,
125
+ ...(cap !== undefined ? { limit: cap } : {}),
126
+ ...(usedFraction !== undefined ? { usedFraction } : {}),
127
+ unit: "credits",
128
+ },
129
+ status: raw.exceeded === true ? "exhausted" : usageStatus(usedFraction),
130
+ };
131
+ }
132
+
133
+ /**
134
+ * Reads the account routes that the Command Code CLI calls with the same API
135
+ * key; the Provider API documents no usage endpoint. Pay-as-you-go accounts
136
+ * carry no `windowLimits`, so they report only the credit balance. Credits
137
+ * are queried per org, as the CLI does, so each limit records the org that
138
+ * owns the pool alongside the user.
139
+ */
140
+ async function fetchCommandCodeUsage(params: UsageFetchParams, ctx: UsageFetchContext): Promise<UsageReport | null> {
141
+ if (params.provider !== PROVIDER) return null;
142
+ if (params.credential.type !== "api_key" || !params.credential.apiKey) return null;
143
+ const origin = resolveOrigin(params.baseUrl);
144
+
145
+ const whoami = await getJson(`${origin}${WHOAMI_PATH}`, params.credential.apiKey, params.signal, ctx);
146
+ if (!whoami) return null;
147
+ const user = isRecord(whoami.user) ? whoami.user : undefined;
148
+ const org = isRecord(whoami.org) ? whoami.org : undefined;
149
+ const userId = nonEmptyString(user?.id);
150
+ if (!userId) return null;
151
+ const orgId = nonEmptyString(org?.id);
152
+ const orgLogin = nonEmptyString(org?.login);
153
+
154
+ const creditsUrl = `${origin}${CREDITS_PATH}${orgId ? `?orgId=${encodeURIComponent(orgId)}` : ""}`;
155
+ const creditsBody = await getJson(creditsUrl, params.credential.apiKey, params.signal, ctx);
156
+ if (!creditsBody) return null;
157
+ const credits = isRecord(creditsBody.credits) ? creditsBody.credits : undefined;
158
+ if (!credits) return null;
159
+ const monthly = finiteNumber(credits.monthlyCredits);
160
+ const purchased = finiteNumber(credits.purchasedCredits);
161
+ const free = finiteNumber(credits.freeCredits);
162
+ if (monthly === undefined && purchased === undefined && free === undefined) return null;
163
+
164
+ const limits: UsageLimit[] = [];
165
+ const windowLimits = isRecord(creditsBody.windowLimits) ? creditsBody.windowLimits : undefined;
166
+ for (const spec of WINDOWS) {
167
+ const limit = buildWindowLimit(spec, windowLimits?.[spec.key], userId, orgId);
168
+ if (limit) limits.push(limit);
169
+ }
170
+ limits.push({
171
+ id: `${PROVIDER}:balance`,
172
+ label: "Credit balance",
173
+ scope: { provider: PROVIDER, accountId: userId, ...(orgId ? { orgId } : {}), windowId: "balance", shared: true },
174
+ amount: { remaining: (monthly ?? 0) + (purchased ?? 0) + (free ?? 0), unit: "credits" },
175
+ });
176
+
177
+ return {
178
+ provider: PROVIDER,
179
+ fetchedAt: Date.now(),
180
+ limits,
181
+ metadata: {
182
+ accountId: userId,
183
+ email: nonEmptyString(user?.email),
184
+ orgId,
185
+ orgName: orgLogin,
186
+ planType: nonEmptyString(credits.planId),
187
+ },
188
+ raw: { whoami, credits: creditsBody },
189
+ };
190
+ }
191
+
192
+ export const commandCodeUsageProvider: UsageProvider = {
193
+ id: PROVIDER,
194
+ fetchUsage: fetchCommandCodeUsage,
195
+ supports: params => params.provider === PROVIDER && params.credential.type === "api_key",
196
+ validatesCredentials: true,
197
+ };
198
+
199
+ /** Ranks Command Code accounts by the 5-hour and weekly credit windows. */
200
+ export const commandCodeRankingStrategy: CredentialRankingStrategy = {
201
+ findWindowLimits: report => ({
202
+ primary: report.limits.find(limit => limit.window?.id === "5h"),
203
+ secondary: report.limits.find(limit => limit.window?.id === "7d"),
204
+ }),
205
+ windowDefaults: {
206
+ primaryMs: 5 * HOUR_MS,
207
+ secondaryMs: WEEK_MS,
208
+ },
209
+ };
@@ -4,6 +4,7 @@ import { alibabaTokenPlanRankingStrategy, alibabaTokenPlanUsageProvider } from "
4
4
  import { charmHyperUsageProvider } from "./charm-hyper";
5
5
  import { claudeRankingStrategy, claudeUsageProvider } from "./claude";
6
6
  import { clinePassUsageProvider } from "./cline-pass";
7
+ import { commandCodeRankingStrategy, commandCodeUsageProvider } from "./commandcode";
7
8
  import { cursorRankingStrategy, cursorUsageProvider } from "./cursor";
8
9
  import { devinUsageProvider } from "./devin";
9
10
  import { googleGeminiCliUsageProvider } from "./gemini";
@@ -45,6 +46,7 @@ export const DEFAULT_USAGE_PROVIDERS: readonly UsageProvider[] = [
45
46
  xaiOauthUsageProvider,
46
47
  devinUsageProvider,
47
48
  charmHyperUsageProvider,
49
+ commandCodeUsageProvider,
48
50
  ];
49
51
 
50
52
  const DEFAULT_USAGE_PROVIDER_MAP = new Map<Provider, UsageProvider>(
@@ -66,6 +68,7 @@ const DEFAULT_RANKING_STRATEGIES = new Map<Provider, CredentialRankingStrategy>(
66
68
  ["zai", zaiRankingStrategy],
67
69
  ["opencode-go", opencodeGoRankingStrategy],
68
70
  ["xai-oauth", xaiOauthRankingStrategy],
71
+ ["commandcode", commandCodeRankingStrategy],
69
72
  ]);
70
73
 
71
74
  /** Built-in ranking strategy for `provider`. */
@@ -6,9 +6,21 @@ export interface LocalWorkSource {
6
6
  readonly hasPendingLocalWork: boolean;
7
7
  }
8
8
 
9
+ /** Consumed head slots tolerated before the backlog is compacted (see {@link EventStream.queue}). */
10
+ const QUEUE_COMPACT_MIN_HEAD = 64;
11
+
9
12
  // Generic event stream class for async iteration
10
13
  export class EventStream<T, R = T> implements AsyncIterable<T> {
14
+ /**
15
+ * Events pushed while no consumer was waiting. The iterator dequeues by
16
+ * advancing {@link #queueHead} instead of `shift()` — O(remaining) per event,
17
+ * quadratic for a consumer draining a backlog — so while it drains, the
18
+ * slots before the head are consumed (cleared) placeholders. Do not mutate
19
+ * the array while the stream is being iterated.
20
+ */
11
21
  queue: T[] = [];
22
+ /** Index of the next undelivered event in {@link queue}; 0 whenever the queue is empty. */
23
+ #queueHead = 0;
12
24
  waiting: Array<{ resolve: (value: IteratorResult<T>) => void; reject: (err: unknown) => void }> = [];
13
25
  done = false;
14
26
  /** True once finalResultPromise has been resolved or rejected. */
@@ -115,10 +127,34 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
115
127
  }
116
128
  }
117
129
 
130
+ /**
131
+ * Take the event at the queue head. Clears the consumed slot so it stops
132
+ * retaining the event, resets the queue once drained, and compacts it once
133
+ * consumed slots are at least half of it (amortized O(1) per event).
134
+ */
135
+ #dequeue(): T {
136
+ const queue = this.queue;
137
+ const head = this.#queueHead;
138
+ const event = queue[head];
139
+ if (head + 1 === queue.length) {
140
+ queue.length = 0;
141
+ this.#queueHead = 0;
142
+ return event;
143
+ }
144
+ // The slot is dead once the head moves past it; `undefined` only drops the reference.
145
+ queue[head] = undefined as T;
146
+ this.#queueHead = head + 1;
147
+ if (this.#queueHead >= QUEUE_COMPACT_MIN_HEAD && this.#queueHead * 2 >= queue.length) {
148
+ queue.splice(0, this.#queueHead);
149
+ this.#queueHead = 0;
150
+ }
151
+ return event;
152
+ }
153
+
118
154
  async *[Symbol.asyncIterator](): AsyncIterator<T> {
119
155
  while (true) {
120
- if (this.queue.length > 0) {
121
- yield this.queue.shift()!;
156
+ if (this.#queueHead < this.queue.length) {
157
+ yield this.#dequeue();
122
158
  } else if (this.#failed) {
123
159
  throw this.#error;
124
160
  } else if (this.done) {
@@ -70,8 +70,14 @@ export interface OpenAIStreamRequestInit {
70
70
  fetch?: FetchImpl;
71
71
  /** Optional caller-specific gate composed with shared transport retry exclusions. */
72
72
  shouldRetryResponse?: (response: Response, bodyText: string) => boolean | Promise<boolean>;
73
- /** Raw wire-frame observer (`onSseEvent` debug pipeline). */
73
+ /**
74
+ * Raw wire-frame observer (`onSseEvent` debug pipeline). Leave it unset
75
+ * when no diagnostic listener exists: any observer turns on per-line raw
76
+ * capture for every frame.
77
+ */
74
78
  onSseEvent?: SseEventObserver;
79
+ /** Called when the stream ends on the OpenAI `[DONE]` sentinel; independent of {@link onSseEvent}. */
80
+ onDoneSentinel?: () => void;
75
81
  }
76
82
 
77
83
  export interface OpenAIStreamHandle<TEvent> {
@@ -117,7 +123,7 @@ export async function postOpenAIStream<TEvent>(init: OpenAIStreamRequestInit): P
117
123
  });
118
124
  }
119
125
  return {
120
- events: decodeStream<TEvent>(response.body, init.signal, init.onSseEvent),
126
+ events: decodeStream<TEvent>(response.body, init.signal, init.onSseEvent, init.onDoneSentinel),
121
127
  response,
122
128
  requestId: response.headers.get("x-request-id"),
123
129
  };
@@ -140,8 +146,9 @@ async function* decodeStream<TEvent>(
140
146
  body: ReadableStream<Uint8Array>,
141
147
  signal: AbortSignal | undefined,
142
148
  onSseEvent: SseEventObserver | undefined,
149
+ onDoneSentinel: (() => void) | undefined,
143
150
  ): AsyncGenerator<TEvent> {
144
- for await (const frame of readSseJsonOrText<TEvent>(body, signal, onSseEvent)) {
151
+ for await (const frame of readSseJsonOrText<TEvent>(body, signal, onSseEvent, onDoneSentinel)) {
145
152
  if (typeof frame === "string") {
146
153
  const inBand = AIError.createInBandProviderErrorFromText(frame);
147
154
  if (inBand) throw inBand;