@oh-my-pi/pi-ai 18.2.0 → 18.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +35 -0
  2. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  3. package/dist/types/auth-gateway/index.d.ts +1 -0
  4. package/dist/types/auth-gateway/session-state.d.ts +65 -0
  5. package/dist/types/auth-storage.d.ts +16 -0
  6. package/dist/types/error/body-error.d.ts +15 -0
  7. package/dist/types/error/flags.d.ts +16 -0
  8. package/dist/types/error/index.d.ts +1 -0
  9. package/dist/types/oneshot-retry.d.ts +6 -0
  10. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  11. package/dist/types/providers/openai-shared.d.ts +20 -3
  12. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  13. package/dist/types/registry/oauth/types.d.ts +8 -0
  14. package/dist/types/stream.d.ts +2 -0
  15. package/dist/types/types.d.ts +3 -1
  16. package/dist/types/usage.d.ts +8 -0
  17. package/dist/types/utils/block-symbols.d.ts +36 -0
  18. package/dist/types/utils/openai-http.d.ts +2 -0
  19. package/dist/types/utils/retry-after.d.ts +2 -0
  20. package/dist/types/utils/schema/wire.d.ts +4 -5
  21. package/dist/types/utils.d.ts +9 -0
  22. package/package.json +6 -6
  23. package/src/auth-broker/remote-store.ts +73 -8
  24. package/src/auth-broker/wire-schemas.ts +1 -0
  25. package/src/auth-gateway/index.ts +1 -0
  26. package/src/auth-gateway/server.ts +48 -11
  27. package/src/auth-gateway/session-state.ts +114 -0
  28. package/src/auth-storage.ts +144 -13
  29. package/src/error/body-error.ts +310 -0
  30. package/src/error/flags.ts +63 -13
  31. package/src/error/index.ts +1 -0
  32. package/src/error/retryable.ts +2 -0
  33. package/src/oneshot-retry.ts +13 -3
  34. package/src/providers/anthropic-messages-server.ts +24 -3
  35. package/src/providers/anthropic.ts +101 -15
  36. package/src/providers/cursor.ts +7 -1
  37. package/src/providers/devin.ts +82 -28
  38. package/src/providers/openai-chat-server.ts +4 -0
  39. package/src/providers/openai-codex/request-transformer.ts +36 -0
  40. package/src/providers/openai-codex-responses.ts +35 -12
  41. package/src/providers/openai-completions.ts +43 -12
  42. package/src/providers/openai-reasoning-fallback.ts +6 -6
  43. package/src/providers/openai-responses-server.ts +2 -1
  44. package/src/providers/openai-responses.ts +25 -4
  45. package/src/providers/openai-shared.ts +199 -51
  46. package/src/registry/oauth/perplexity.ts +94 -28
  47. package/src/registry/oauth/types.ts +9 -0
  48. package/src/stream.ts +23 -2
  49. package/src/types.ts +3 -0
  50. package/src/usage/claude.ts +33 -0
  51. package/src/usage/google-antigravity.ts +8 -2
  52. package/src/usage.ts +3 -0
  53. package/src/utils/block-symbols.ts +57 -0
  54. package/src/utils/openai-http.ts +39 -3
  55. package/src/utils/retry-after.ts +12 -0
  56. package/src/utils/schema/normalize.ts +3 -3
  57. package/src/utils/schema/stamps.ts +33 -45
  58. package/src/utils/schema/wire.ts +9 -7
  59. package/src/utils.ts +67 -22
@@ -163,6 +163,28 @@ export const PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN =
163
163
  /peer closed connection without sending complete message body \(incomplete chunked read\)/;
164
164
  /** reqwest body-frame failures forwarded by the Codex HTTP proxy. */
165
165
  export const CODEX_HTTP_BODY_READ_ERROR_PATTERN = /\btransport error reading codex response body\b/i;
166
+
167
+ const RESPONSES_REQUEST_BODY_READ_TIMEOUT_PATTERN = /\btimed out reading request body\b/i;
168
+
169
+ /** Exact HTTP request-body-read timeout diagnostic. */
170
+ export function isRequestBodyReadTimeout(status: number | undefined, message: string | undefined): boolean {
171
+ return status === 408 && RESPONSES_REQUEST_BODY_READ_TIMEOUT_PATTERN.test(message ?? "");
172
+ }
173
+
174
+ /** Exact pre-output Responses 408 that needs a changed-request recovery path. */
175
+ export function isResponsesRequestBodyReadTimeout(message: {
176
+ api?: Api;
177
+ errorStatus?: number;
178
+ errorMessage?: string;
179
+ requestBodyReadTimeoutFullReplay?: boolean;
180
+ }): boolean {
181
+ return (
182
+ message.api === "openai-responses" &&
183
+ message.requestBodyReadTimeoutFullReplay === true &&
184
+ isRequestBodyReadTimeout(message.errorStatus, message.errorMessage)
185
+ );
186
+ }
187
+
166
188
  export const TRANSIENT_TRANSPORT_PATTERN =
167
189
  /\b(?:no[_ -]?capacity|(?:high|peak)[ _-]?demand|(?:at|over|insufficient)[ _-]?capacity|capacity[ _-]?(?:exceeded|exhausted)|peak[ _-]?load)\b|overloaded|provider.?returned.?error|rate.?limit|too many requests|auth-gateway\s+5\d{2}(?=[:\s]|$)|\b(?:429|500|502|503|504)\b|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|unable.?to.?connect\.\s*is the computer able to access the url\?|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|nghttp2_(?:internal_error|refused_stream)|stream closed with error code nghttp2_(?:internal_error|refused_stream)|malformed.?function.?call/i;
168
190
  const AUTH_FAILURE_PATTERN =
@@ -503,21 +525,24 @@ function classifyText(
503
525
  }
504
526
  if (isTimeoutText(errorMessage)) kinds |= Flag.Transient | Flag.Timeout;
505
527
  else if (isTransientErrorText(errorMessage)) kinds |= Flag.Transient;
506
- // A stream truncation or forwarded Codex HTTP body-read failure may not
507
- // match TRANSIENT_TRANSPORT_PATTERN. Flag it explicitly so AIError.retriable and
508
- // the turn-recovery layer treat it as retryable, matching the provider
509
- // retry path (isProviderRetryableError). Separate `if` (not chained onto
510
- // the else-if) so a timeout whose text also reads as a truncation keeps
511
- // Flag.Timeout alongside Flag.Transient. The string arm applies the strict
512
- // STREAM_PARSE_DIAGNOSTIC_PATTERN, per the rationale on isTransientStreamParseError.
513
- // Skip a truncation phrase that rides on a terminal 4xx (e.g. a malformed
514
- // request rejected as "400 unexpected EOF"): that is a deterministic client
515
- // error that replays identically, so keep it terminal. classify() carries
516
- // the outer terminal status down the cause chain so a wrapped truncation
517
- // (ProviderHttpError 400 → cause "unexpected EOF") is caught here too.
528
+ // A stream truncation, transport-level stream drop, or forwarded Codex HTTP
529
+ // body-read failure may not match TRANSIENT_TRANSPORT_PATTERN. Flag it
530
+ // explicitly so AIError.retriable and the turn-recovery layer treat it as
531
+ // retryable, matching the provider retry path (isProviderRetryableError).
532
+ // Separate `if` (not chained onto the else-if) so a timeout whose text also
533
+ // reads as a truncation keeps Flag.Timeout alongside Flag.Transient. The
534
+ // string arm applies the strict STREAM_PARSE_DIAGNOSTIC_PATTERN, per the
535
+ // rationale on isTransientStreamParseError. Skip a phrase that rides on a
536
+ // terminal 4xx (e.g. a malformed request rejected as "400 unexpected EOF"):
537
+ // that is a deterministic client error that replays identically, so keep it
538
+ // terminal. classify() carries the outer terminal status down the cause
539
+ // chain so a wrapped truncation (ProviderHttpError 400 → cause "unexpected
540
+ // EOF") is caught here too.
518
541
  if (
519
542
  !isTerminalClientErrorStatus(statusClean) &&
520
- (isTransientStreamParseError(errorMessage) || CODEX_HTTP_BODY_READ_ERROR_PATTERN.test(errorMessage))
543
+ (isTransientStreamParseError(errorMessage) ||
544
+ isTransientStreamDropError(errorMessage) ||
545
+ CODEX_HTTP_BODY_READ_ERROR_PATTERN.test(errorMessage))
521
546
  ) {
522
547
  kinds |= Flag.Transient;
523
548
  }
@@ -871,6 +896,31 @@ export function isTransientStreamParseError(error: unknown): boolean {
871
896
  return error instanceof Error && STREAM_PARSE_TRUNCATION_PATTERN.test(error.message);
872
897
  }
873
898
 
899
+ /**
900
+ * Transport-level stream drops: the connection or upstream stream ended before a
901
+ * terminal event, with no JSON-parse signal and no retryable status attached.
902
+ *
903
+ * Distinct from {@link STREAM_PARSE_TRUNCATION_PATTERN} (mid-body JSON
904
+ * truncation) — these name the transport itself dropping (proxy/gateway closing
905
+ * the SSE stream, socket dying before the TLS handshake completes). The wording
906
+ * is the statusless twin of a `408 stream disconnected`, which the status path
907
+ * already retries; an identical replay recovers it, so callers under a
908
+ * non-terminal status treat it as transient (#11805).
909
+ */
910
+ const STREAM_DROP_PATTERN =
911
+ /stream disconnected before completion|stream closed before response\.completed|stream was interrupted|stream ended before terminal (?:chunk|completion event)|socket disconnected before secure tls connection/i;
912
+
913
+ /**
914
+ * Transport stream-drop diagnostic (see {@link STREAM_DROP_PATTERN}). Unlike
915
+ * {@link isTransientStreamParseError}, one pattern serves both the live `Error`
916
+ * and the persisted-string forms: the phrasings are high-signal enough to trust
917
+ * detached from a transport `Error`.
918
+ */
919
+ export function isTransientStreamDropError(error: unknown): boolean {
920
+ if (typeof error === "string") return STREAM_DROP_PATTERN.test(error);
921
+ return error instanceof Error && STREAM_DROP_PATTERN.test(error.message);
922
+ }
923
+
874
924
  /** Any malformed stream-envelope error (prefix-tagged or out-of-order events). */
875
925
  export function isStreamEnvelopeError(error: unknown): boolean {
876
926
  return (
@@ -2,6 +2,7 @@ export * from "./abort";
2
2
  export * from "./auth";
3
3
  export * from "./auth-classify";
4
4
  export * from "./aws";
5
+ export * from "./body-error";
5
6
  export * from "./classes";
6
7
  export * from "./finalize";
7
8
  export * from "./flags";
@@ -2,6 +2,7 @@ import { isRetryableError, isUnexpectedSocketCloseMessage } from "@oh-my-pi/pi-u
2
2
  import {
3
3
  CODEX_HTTP_BODY_READ_ERROR_PATTERN,
4
4
  isRetryableStreamEnvelopeError,
5
+ isTransientStreamDropError,
5
6
  isTransientStreamParseError,
6
7
  isUsageLimit,
7
8
  status,
@@ -55,6 +56,7 @@ export function isProviderRetryableError(error: unknown): boolean {
55
56
  CODEX_HTTP_BODY_READ_ERROR_PATTERN.test(msg) ||
56
57
  PROVIDER_TRANSIENT_EXTRA_PATTERN.test(msg) ||
57
58
  isTransientStreamParseError(error) ||
59
+ isTransientStreamDropError(error) ||
58
60
  isRetryableStreamEnvelopeError(error)
59
61
  ) {
60
62
  return true;
@@ -1,7 +1,11 @@
1
- import { extractRetryHint } from "@oh-my-pi/pi-utils";
2
1
  import * as AIError from "./error";
3
2
  import type { AssistantMessage } from "./types";
4
- import { getHeadersFromError, getRetryAfterMsFromHeaders, type HeadersLike } from "./utils/retry-after";
3
+ import {
4
+ extractProviderRetryHint,
5
+ getHeadersFromError,
6
+ getRetryAfterMsFromHeaders,
7
+ type HeadersLike,
8
+ } from "./utils/retry-after";
5
9
 
6
10
  /**
7
11
  * Transient-failure retry for **oneshot** (non-agent-loop) completions.
@@ -66,6 +70,12 @@ export interface OneshotRetryOptions {
66
70
  * Thrown errors need no wiring — headers are recovered from the error itself.
67
71
  */
68
72
  getResponseHeaders?: () => HeadersLike;
73
+ /**
74
+ * Provider id of the model being retried. Selects the catalog-declared
75
+ * timezone for a timezone-naive absolute reset stamp (Z.AI/Zhipu report
76
+ * Beijing time), so an over-cap wait is not misread as UTC and discarded.
77
+ */
78
+ provider?: string;
69
79
  /** Observability hook. Fires immediately before sleeping. */
70
80
  onRetry?: (info: OneshotRetryInfo) => void;
71
81
  }
@@ -195,7 +205,7 @@ export async function retryTransientCompletion(
195
205
  // errors (e.g. AnthropicApiError) carry their own headers.
196
206
  const headers: HeadersLike = thrown !== undefined ? getHeadersFromError(thrown) : options?.getResponseHeaders?.();
197
207
  const headerHintMs = getRetryAfterMsFromHeaders(headers);
198
- const extractedTextHintMs = extractRetryHint(undefined, errorMessage);
208
+ const extractedTextHintMs = extractProviderRetryHint(options?.provider, errorMessage);
199
209
  const suffixValue = RETRY_AFTER_MS_SUFFIX.exec(errorMessage)?.[1];
200
210
  const parsedSuffixMs = suffixValue === undefined ? undefined : Number(suffixValue);
201
211
  const suffixHintMs =
@@ -354,6 +354,23 @@ const REASONING_EFFORT_BY_WIRE: Partial<Record<string, Effort>> = {
354
354
  max: Effort.Max,
355
355
  };
356
356
 
357
+ /**
358
+ * Recover the id of the model this request will actually reach, for labelling
359
+ * replayed assistant turns.
360
+ *
361
+ * `/v1/models` advertises `<provider>/<id>` and nothing else, so that is what
362
+ * clients send, but `resolveModel` resolves a catalog model whose id is the
363
+ * bare half. Labelling a replayed turn with the full id the client sent leaves
364
+ * `transform-messages` reading it as written by some other model.
365
+ *
366
+ * A prefix naming another provider is left intact: it describes a different
367
+ * route, so removing it would invent history rather than recover it.
368
+ */
369
+ function stampedAssistantModelId(wireModelId: string, provider: string): string {
370
+ const prefix = `${provider}/`;
371
+ return wireModelId.startsWith(prefix) ? wireModelId.slice(prefix.length) : wireModelId;
372
+ }
373
+
357
374
  export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
358
375
  const data = anthropicMessagesRequestSchema(body);
359
376
  if (data instanceof type.errors) {
@@ -368,14 +385,18 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
368
385
  } else if (message.role === "system") {
369
386
  messages.push(walkSystemMessage(message, now));
370
387
  } else {
388
+ const content = walkAssistantContent(message.content);
371
389
  const assistant: AssistantMessage = {
372
390
  role: "assistant",
373
- content: walkAssistantContent(message.content),
391
+ content,
374
392
  api: "anthropic-messages",
375
393
  provider: "anthropic",
376
- model: data.model,
394
+ model: stampedAssistantModelId(data.model, "anthropic"),
377
395
  usage: emptyUsage(),
378
- stopReason: "stop",
396
+ // The wire carries no stop reason, but tool calls answered by their
397
+ // `tool_result` blocks did request execution. A constant "stop" reads
398
+ // as an abandoned tool-use turn and strips the turn's signatures.
399
+ stopReason: content.some(block => block.type === "toolCall") ? "toolUse" : "stop",
379
400
  timestamp: now,
380
401
  };
381
402
  messages.push(assistant);
@@ -60,8 +60,10 @@ import {
60
60
  import { createAbortSourceTracker } from "../utils/abort";
61
61
  import {
62
62
  clearStreamingPartialJson,
63
+ copyPerCallContextMessage,
63
64
  type ConversationalUserCarrier,
64
65
  isConversationalUser,
66
+ isPerCallContextMessage,
65
67
  isSyntheticUser,
66
68
  kConversationalUser,
67
69
  kStreamingBlockIndex,
@@ -454,6 +456,15 @@ type AnthropicProviderSessionState = ProviderSessionState & {
454
456
  * `compat.replayUnsignedThinking: false`. Cleared on session close.
455
457
  */
456
458
  replayUnsignedThinkingDisabled: boolean;
459
+ /**
460
+ * Runtime-learned: this endpoint kept rejecting replayed thinking
461
+ * signatures even after unsigned demotion — every surviving block is
462
+ * signed by a foreign signer (e.g. a failover proxy swapped upstreams
463
+ * mid-conversation and minted signatures the restored upstream cannot
464
+ * verify). All subsequent requests drop replayed thinking entirely for
465
+ * this (baseUrl, modelId). Cleared on session close.
466
+ */
467
+ thinkingReplayDisabled: boolean;
457
468
  /** Thinking blocks the API permanently dropped after a prefix mismatch. */
458
469
  prefixDroppedThinkingBlocks: Set<string>;
459
470
  /** Conversation-scoped control baselines, isolated from side requests and advisors. */
@@ -478,12 +489,14 @@ function createAnthropicProviderSessionState(): AnthropicProviderSessionState {
478
489
  strictToolsDisabled: false,
479
490
  fastModeDisabled: false,
480
491
  replayUnsignedThinkingDisabled: false,
492
+ thinkingReplayDisabled: false,
481
493
  prefixDroppedThinkingBlocks: new Set(),
482
494
  controlStates: new Map(),
483
495
  close: () => {
484
496
  state.strictToolsDisabled = false;
485
497
  state.fastModeDisabled = false;
486
498
  state.replayUnsignedThinkingDisabled = false;
499
+ state.thinkingReplayDisabled = false;
487
500
  state.prefixDroppedThinkingBlocks.clear();
488
501
  state.controlStates.clear();
489
502
  },
@@ -2296,7 +2309,8 @@ const streamAnthropicOnce = (
2296
2309
  (providerSessionState?.strictToolsDisabled ?? false) || (model.compat?.disableStrictTools ?? false);
2297
2310
  let dropFastMode = providerSessionState?.fastModeDisabled ?? false;
2298
2311
  let forceDemoteUnsignedThinking = providerSessionState?.replayUnsignedThinkingDisabled ?? false;
2299
- let dropAllThinking = false;
2312
+ let droppedAllThinkingForSignature = providerSessionState?.thinkingReplayDisabled ?? false;
2313
+ let dropAllThinking = droppedAllThinkingForSignature;
2300
2314
  let prefixBindingRetryAttempted = false;
2301
2315
  let prefixMismatchBehavior =
2302
2316
  model.thinking?.prefixBinding && model.compat.supportsThinkingBindingControls
@@ -3367,6 +3381,48 @@ const streamAnthropicOnce = (
3367
3381
  firstTokenTime = undefined;
3368
3382
  continue;
3369
3383
  }
3384
+ if (
3385
+ !dropAllThinking &&
3386
+ firstTokenTime === undefined &&
3387
+ !streamedReplayUnsafeContent &&
3388
+ !isThinkingPrefixBindingError(streamFailureMessage) &&
3389
+ isInvalidThinkingSignatureError(streamFailureMessage)
3390
+ ) {
3391
+ // The unsigned-demotion retry only rewrites UNSIGNED blocks;
3392
+ // when every replayed block carries a signature the signer no
3393
+ // longer accepts (e.g. a failover proxy swapped upstreams
3394
+ // mid-conversation and minted foreign signatures), the retry
3395
+ // resends a byte-identical body and the session 400s forever.
3396
+ // Escalate: drop all replayed thinking — prior-turn reasoning
3397
+ // is optional context — and retry once. Stored history keeps
3398
+ // its thinking blocks; only the wire payload changes.
3399
+ logger.warn(
3400
+ "anthropic: thinking signatures still rejected after unsigned demotion, dropping replayed thinking and retrying",
3401
+ {
3402
+ provider: model.provider,
3403
+ model: model.id,
3404
+ baseUrl,
3405
+ error: streamFailureMessage,
3406
+ },
3407
+ );
3408
+ if (providerSessionState) {
3409
+ providerSessionState.thinkingReplayDisabled = true;
3410
+ }
3411
+ droppedAllThinkingForSignature = true;
3412
+ dropAllThinking = true;
3413
+ params = await prepareParams();
3414
+ providerRetryAttempt = 0;
3415
+ output.content.length = 0;
3416
+ output.model = model.id;
3417
+ output.responseId = undefined;
3418
+ output.errorMessage = undefined;
3419
+ output.inputTransformations = undefined;
3420
+ output.providerPayload = undefined;
3421
+ output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
3422
+ output.stopReason = "stop";
3423
+ firstTokenTime = undefined;
3424
+ continue;
3425
+ }
3370
3426
  if (
3371
3427
  !dropFastMode &&
3372
3428
  model.provider === "anthropic" &&
@@ -3451,6 +3507,9 @@ const streamAnthropicOnce = (
3451
3507
  if (forceDemoteUnsignedThinking && model.compat.replayUnsignedThinking) {
3452
3508
  output.disabledFeatures = [...(output.disabledFeatures ?? []), "unsigned-thinking-replay"];
3453
3509
  }
3510
+ if (droppedAllThinkingForSignature) {
3511
+ output.disabledFeatures = [...(output.disabledFeatures ?? []), "thinking-replay"];
3512
+ }
3454
3513
  stream.push({ type: "done", reason: output.stopReason, message: output });
3455
3514
  stream.end();
3456
3515
  } catch (error) {
@@ -3892,14 +3951,26 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
3892
3951
  params.messages[trailingIndex - 1]?.role === "assistant";
3893
3952
  const messageEnd = hasTrailingAssistantPad ? trailingIndex - 1 : trailingIndex;
3894
3953
 
3954
+ // A breakpoint caches every preceding byte, not only the decorated message.
3955
+ // Once per-call or turn-scoped content appears, no later message can anchor a
3956
+ // prefix reusable by the next request.
3957
+ let stableMessageEnd = messageEnd;
3958
+ for (let index = 0; index <= messageEnd; index++) {
3959
+ const message = params.messages[index];
3960
+ if (message && (message.clear_at === "next_user_message" || isPerCallContextMessage(message))) {
3961
+ stableMessageEnd = index - 1;
3962
+ break;
3963
+ }
3964
+ }
3965
+
3895
3966
  // Decimation counts conversational turns, so it reads the provenance marker
3896
3967
  // `convertAnthropicMessages` records rather than the wire role. A wire `user`
3897
3968
  // can also be a serialized `developer` message, a tool_result run, or an
3898
3969
  // interior `Continue.` pad, none of which advance the user turn ordinal.
3899
3970
  const userIndices: number[] = [];
3900
- for (let index = 0; index <= messageEnd; index++) {
3971
+ for (let index = 0; index <= stableMessageEnd; index++) {
3901
3972
  const message = params.messages[index];
3902
- if (message && message.clear_at !== "next_user_message" && isConversationalUser(message)) {
3973
+ if (message && isConversationalUser(message)) {
3903
3974
  userIndices.push(index);
3904
3975
  }
3905
3976
  }
@@ -3907,11 +3978,11 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
3907
3978
  // Stable historical decimation checkpoint every 15 user turns (15th, 30th, 45th...)
3908
3979
  const decimationIndices = userIndices.filter((_, ordinal) => (ordinal + 1) % ANTHROPIC_DECIMATION_INTERVAL === 0);
3909
3980
 
3910
- // Collect eligible trailing candidates (up to 2 messages walking backward from messageEnd).
3981
+ // Collect up to 2 trailing candidates from the reusable prefix.
3911
3982
  const trailingCandidates: number[] = [];
3912
- for (let index = messageEnd; index >= 0 && trailingCandidates.length < 2; index--) {
3983
+ for (let index = stableMessageEnd; index >= 0 && trailingCandidates.length < 2; index--) {
3913
3984
  const message = params.messages[index];
3914
- if (!message || message.clear_at === "next_user_message") continue;
3985
+ if (!message) continue;
3915
3986
  trailingCandidates.push(index);
3916
3987
  }
3917
3988
 
@@ -4774,7 +4845,12 @@ export function convertAnthropicMessages(
4774
4845
  (msg.role === "user" || msg.role === "developer") &&
4775
4846
  isReplayableAnthropicCompaction(msg.providerPayload, model)
4776
4847
  ) {
4777
- params.push({ role: "assistant", content: [compactionBlockParam(msg.providerPayload)] });
4848
+ const compactionParam: AnthropicMessageParam = {
4849
+ role: "assistant",
4850
+ content: [compactionBlockParam(msg.providerPayload)],
4851
+ };
4852
+ copyPerCallContextMessage(compactionParam, msg);
4853
+ params.push(compactionParam);
4778
4854
  // The block carries the verbatim API summary, so the message text
4779
4855
  // (which holds the harness file lists) would be dropped with it.
4780
4856
  // Queue the file metadata for after the block: it sits past the
@@ -4845,6 +4921,7 @@ export function convertAnthropicMessages(
4845
4921
  if (msg.role === "user" && !agentAuthored && !isSyntheticUser(msg)) {
4846
4922
  param[kConversationalUser] = true;
4847
4923
  }
4924
+ copyPerCallContextMessage(param, msg);
4848
4925
  params.push(param);
4849
4926
  } else if (msg.role === "assistant") {
4850
4927
  const blocks: ContentBlockParam[] = [];
@@ -4982,10 +5059,12 @@ export function convertAnthropicMessages(
4982
5059
  blocks.push(...nonToolUse, ...toolUse);
4983
5060
  }
4984
5061
  if (blocks.length === 0) continue;
4985
- params.push({
5062
+ const assistantParam: AnthropicMessageParam = {
4986
5063
  role: "assistant",
4987
5064
  content: blocks,
4988
- });
5065
+ };
5066
+ copyPerCallContextMessage(assistantParam, msg);
5067
+ params.push(assistantParam);
4989
5068
  // Flush queued file metadata unless this turn left tool calls open:
4990
5069
  // their results must follow the turn contiguously, so the metadata
4991
5070
  // waits for the merged result message (or the end of the list).
@@ -4997,15 +5076,21 @@ export function convertAnthropicMessages(
4997
5076
  const toolResults: ContentBlockParam[] = [];
4998
5077
  // Images stripped out of error tool results, re-attached after the run.
4999
5078
  const hoistedImages: ContentBlockParam[] = [];
5079
+ const toolResultParam: AnthropicMessageParam = {
5080
+ role: "user",
5081
+ content: toolResults,
5082
+ };
5000
5083
 
5001
5084
  // Add the current tool result
5002
5085
  toolResults.push(buildToolResultBlock(model, msg, hoistedImages));
5086
+ copyPerCallContextMessage(toolResultParam, msg);
5003
5087
 
5004
5088
  // Look ahead for consecutive toolResult messages
5005
5089
  let j = i + 1;
5006
5090
  while (j < transformedMessages.length && transformedMessages[j].role === "toolResult") {
5007
5091
  const nextMsg = transformedMessages[j] as ToolResultMessage; // We know it's a toolResult
5008
5092
  toolResults.push(buildToolResultBlock(model, nextMsg, hoistedImages));
5093
+ copyPerCallContextMessage(toolResultParam, nextMsg);
5009
5094
  j++;
5010
5095
  }
5011
5096
 
@@ -5020,10 +5105,7 @@ export function convertAnthropicMessages(
5020
5105
  }
5021
5106
 
5022
5107
  // Add a single user message with all tool results
5023
- params.push({
5024
- role: "user",
5025
- content: toolResults,
5026
- });
5108
+ params.push(toolResultParam);
5027
5109
  // An open tool_use turn's results are whole again; queued file
5028
5110
  // metadata can follow without splitting the pairing.
5029
5111
  flushCompactionFiles();
@@ -5060,20 +5142,24 @@ export function convertAnthropicMessages(
5060
5142
  const controlContent = content.filter(block => block.type !== "text");
5061
5143
  if (scopedContent.length > 0) {
5062
5144
  params[idx] = {
5145
+ ...params[idx],
5063
5146
  role: "system",
5064
5147
  content: scopedContent,
5065
5148
  clear_at: "next_user_message",
5066
5149
  };
5067
- params.splice(idx + 1, 0, {
5150
+ const controlParam: AnthropicMessageParam = {
5068
5151
  role: "system",
5069
5152
  content: controlContent,
5070
5153
  ...(hasEffort ? { output_config: { effort: developer.payload?.effort } } : {}),
5071
- });
5154
+ };
5155
+ copyPerCallContextMessage(controlParam, params[idx]);
5156
+ params.splice(idx + 1, 0, controlParam);
5072
5157
  continue;
5073
5158
  }
5074
5159
  }
5075
5160
 
5076
5161
  params[idx] = {
5162
+ ...params[idx],
5077
5163
  role: "system",
5078
5164
  content,
5079
5165
  ...(turnScoped && !hasEffort && !hasToolChanges ? { clear_at: "next_user_message" } : {}),
@@ -4504,7 +4504,13 @@ export function processInteractionUpdate(
4504
4504
  let persisted: ToolResultMessage | undefined;
4505
4505
  let hostError: string | null = null;
4506
4506
  try {
4507
- persisted = state.onTodoSnapshot?.(snapshot, settled.id, error) ?? undefined;
4507
+ persisted =
4508
+ state.onTodoSnapshot?.(
4509
+ snapshot,
4510
+ settled.id,
4511
+ error,
4512
+ toolCall && selectTodoCalls(toolCall).read ? "read" : "update",
4513
+ ) ?? undefined;
4508
4514
  } catch (callbackError) {
4509
4515
  // A throwing host callback (e.g. session persistence failing on
4510
4516
  // disk error) must not leave the resolved block unpaired: the
@@ -1,5 +1,6 @@
1
1
  import { gunzipSync, gzipSync } from "node:zlib";
2
2
 
3
+ import { classifyModel } from "@oh-my-pi/pi-catalog/compat/taxonomy";
3
4
  import {
4
5
  AssignModelRequestSchema,
5
6
  AssignModelResponseSchema,
@@ -29,8 +30,9 @@ import { create, fromBinary, toBinary } from "@oh-my-pi/pi-catalog/discovery/pro
29
30
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
30
31
  import { DEVIN_DEFAULT_BASE_URL, devinCliMetadata } from "@oh-my-pi/pi-catalog/wire/devin";
31
32
  import { decodeDevinUnaryMessage } from "@oh-my-pi/pi-catalog/wire/devin-proto";
32
- import { logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
33
+ import { isRecord, logger, parseStreamingJson, parseStreamingJsonThrottled, sanitizeText } from "@oh-my-pi/pi-utils";
33
34
  import * as AIError from "../error";
35
+
34
36
  import type {
35
37
  Api,
36
38
  AssistantMessage,
@@ -50,7 +52,7 @@ import { normalizeSystemPrompts } from "../utils";
50
52
  import { isDemotedThinking } from "../utils/block-symbols";
51
53
  import { deterministicUuid } from "../utils/deterministic-id";
52
54
  import { AssistantMessageEventStream } from "../utils/event-stream";
53
- import { toolWireSchema } from "../utils/schema/wire";
55
+ import { normalizeSchemaForGoogle, toolWireSchema } from "../utils/schema";
54
56
  import { transformMessages } from "./transform-messages";
55
57
 
56
58
  /** Base host for Codeium/Windsurf's Cascade chat API (Connect protocol over HTTP/1.1). */
@@ -89,6 +91,58 @@ const MAX_CONNECT_FRAME_PAYLOAD = 16 * 1024 * 1024;
89
91
  * to benefit from the existing context-overflow maintenance path.
90
92
  */
91
93
  const LARGE_HISTORY_RECOVERY_BYTES = 512 * 1024;
94
+ const MAX_DEVIN_ERROR_DETAIL_CHARS = 4096;
95
+ const HTML_ERROR_BODY_PATTERN = /^\s*(?:<!doctype\s+html\b|<html\b)/i;
96
+
97
+ /** Extract a bounded human error without leaking proxy HTML or binary protobuf. */
98
+ function devinErrorDetail(response: Response, payload: Uint8Array): string | undefined {
99
+ let text: string;
100
+ try {
101
+ text = new TextDecoder("utf-8", { fatal: true }).decode(payload).trim();
102
+ } catch {
103
+ return undefined;
104
+ }
105
+ if (response.headers.get("content-type")?.toLowerCase().includes("text/html")) return undefined;
106
+ try {
107
+ const decoded: unknown = JSON.parse(text);
108
+ if (isRecord(decoded)) {
109
+ const error = decoded.error;
110
+ if (isRecord(error) && typeof error.message === "string") text = error.message.trim();
111
+ else if (typeof error === "string") text = error.trim();
112
+ else if (typeof decoded.message === "string") text = decoded.message.trim();
113
+ }
114
+ } catch {}
115
+ // Validate after envelope extraction: JSON escapes (`\u001b`, `<html>` inside a
116
+ // message) materialize bytes the raw-source scan cannot see. Whitespace is
117
+ // collapsed first so benign CRLF does not trip the control detection; after
118
+ // that, any text `sanitizeText` would alter (C0/C1 controls, DEL, malformed
119
+ // Unicode) is untrustworthy diagnostics and suppresses to status-only.
120
+ const normalized = text.replace(/\s+/g, " ").trim();
121
+ if (normalized.length === 0 || HTML_ERROR_BODY_PATTERN.test(normalized) || sanitizeText(normalized) !== normalized) {
122
+ return undefined;
123
+ }
124
+ if (normalized.length <= MAX_DEVIN_ERROR_DETAIL_CHARS) return normalized;
125
+ // Truncate on a code-point boundary: slicing through a surrogate pair would
126
+ // re-introduce the malformed text the sanitizeText gate just ruled out.
127
+ const boundaryUnit = normalized.charCodeAt(MAX_DEVIN_ERROR_DETAIL_CHARS - 1);
128
+ const cut =
129
+ boundaryUnit >= 0xd800 && boundaryUnit <= 0xdbff
130
+ ? MAX_DEVIN_ERROR_DETAIL_CHARS - 1
131
+ : MAX_DEVIN_ERROR_DETAIL_CHARS;
132
+ return normalized.slice(0, cut);
133
+ }
134
+
135
+ function createDevinHttpError(operation: string, response: Response, payload: Uint8Array): AIError.DevinApiError {
136
+ const status = `${response.status}${response.statusText ? ` ${response.statusText}` : ""}`;
137
+ const detail = devinErrorDetail(response, payload);
138
+ return new AIError.DevinApiError(
139
+ `Devin ${operation} error ${status}${detail ? `: ${detail}` : ""}`,
140
+ response.status,
141
+ {
142
+ headers: response.headers,
143
+ },
144
+ );
145
+ }
92
146
 
93
147
  export const streamDevin: StreamFunction<"devin-agent"> = (
94
148
  model: Model<"devin-agent">,
@@ -210,11 +264,8 @@ export const streamDevin: StreamFunction<"devin-agent"> = (
210
264
  });
211
265
 
212
266
  if (!response.ok) {
213
- const text = await response.text();
214
- throw new AIError.DevinApiError(
215
- `Devin API error ${response.status} ${response.statusText}: ${text}`,
216
- response.status,
217
- );
267
+ const payload = new Uint8Array(await response.arrayBuffer());
268
+ throw createDevinHttpError("API", response, payload);
218
269
  }
219
270
  if (!response.body) {
220
271
  throw new AIError.ProviderResponseError("Devin API error: response body is empty", {
@@ -501,12 +552,7 @@ async function fetchDevinAuthMetadata(
501
552
  signal,
502
553
  });
503
554
  const payload = new Uint8Array(await response.arrayBuffer());
504
- if (!response.ok) {
505
- throw new AIError.DevinApiError(
506
- `Devin auth error ${response.status} ${response.statusText}: ${new TextDecoder().decode(payload)}`,
507
- response.status,
508
- );
509
- }
555
+ if (!response.ok) throw createDevinHttpError("auth", response, payload);
510
556
  const decoded = decodeDevinUnaryMessage(GetUserJwtResponseSchema, payload);
511
557
  if (!decoded?.userJwt) {
512
558
  throw new AIError.ProviderResponseError("Devin auth error: GetUserJwt returned an empty user JWT", {
@@ -548,12 +594,7 @@ async function assignDevinModel(
548
594
  signal,
549
595
  });
550
596
  const payload = new Uint8Array(await response.arrayBuffer());
551
- if (!response.ok) {
552
- throw new AIError.DevinApiError(
553
- `Devin AssignModel error ${response.status} ${response.statusText}: ${new TextDecoder().decode(payload)}`,
554
- response.status,
555
- );
556
- }
597
+ if (!response.ok) throw createDevinHttpError("AssignModel", response, payload);
557
598
  const assignment = decodeDevinUnaryMessage(AssignModelResponseSchema, payload)?.assignment;
558
599
  if (!assignment?.assignmentJwt || !assignment.modelUid) {
559
600
  throw new AIError.ProviderResponseError(
@@ -598,11 +639,31 @@ function buildDevinChatRequest(
598
639
  options?.stopSequences && options.stopSequences.length > 0
599
640
  ? [...DEVIN_DEFAULT_STOP_PATTERNS, ...options.stopSequences]
600
641
  : DEVIN_DEFAULT_STOP_PATTERNS;
642
+ const chatModelUid = assignment?.modelUid ?? options?.chatModelUid ?? model.requestModelId ?? model.id;
643
+ // Devin routes multiple provider families through one Cascade envelope. Its
644
+ // Gemini backend applies Google's tool-schema constraints and rejects JSON
645
+ // Schema type arrays (e.g. `["number", "null"]`) as an opaque internal
646
+ // `invalid_argument`; normalize both direct Gemini models and router-assigned
647
+ // enum-style UIDs (`MODEL_GOOGLE_GEMINI_*`) before serializing tools. The UID
648
+ // prefix is the server's own enum namespace, which classifyModel cannot parse.
649
+ const googleToolSchema =
650
+ classifyModel("devin", model.id, { lenient: true }).class === "gemini" ||
651
+ classifyModel("devin", chatModelUid, { lenient: true }).class === "gemini" ||
652
+ chatModelUid.startsWith("MODEL_GOOGLE_GEMINI_");
653
+ const tools = (context.tools ?? []).map((tool: Tool) => {
654
+ const schema = toolWireSchema(tool);
655
+ return create(ChatToolDefinitionSchema, {
656
+ name: tool.name,
657
+ description: tool.description,
658
+ jsonSchemaString: JSON.stringify(googleToolSchema ? normalizeSchemaForGoogle(schema) : schema),
659
+ strict: tool.strict ?? false,
660
+ });
661
+ });
601
662
  return create(GetChatMessageRequestSchema, {
602
663
  metadata: create(MetadataSchema, devinCliMetadata(turn.apiKey, turn.userJwt)),
603
664
  prompt: normalizeSystemPrompts(context.systemPrompt).join("\n\n"),
604
665
  chatMessagePrompts: buildChatMessagePrompts(turn.messages, turn.cascadeId, model),
605
- chatModelUid: assignment?.modelUid ?? options?.chatModelUid ?? model.requestModelId ?? model.id,
666
+ chatModelUid,
606
667
  ...(assignment ? { modelAssignmentJwt: assignment.assignmentJwt } : undefined),
607
668
  requestType: ChatMessageRequestType.CASCADE,
608
669
  plannerMode: ConversationalPlannerMode.DEFAULT,
@@ -622,14 +683,7 @@ function buildDevinChatRequest(
622
683
  stopPatterns,
623
684
  fimEotProbThreshold: 1,
624
685
  }),
625
- tools: (context.tools ?? []).map((tool: Tool) =>
626
- create(ChatToolDefinitionSchema, {
627
- name: tool.name,
628
- description: tool.description,
629
- jsonSchemaString: JSON.stringify(toolWireSchema(tool)),
630
- strict: tool.strict ?? false,
631
- }),
632
- ),
686
+ tools,
633
687
  });
634
688
  }
635
689
 
@@ -30,6 +30,7 @@ import {
30
30
  openaiChatRequestSchema,
31
31
  } from "./openai-chat-server-schema";
32
32
  import { decodeDataUri } from "./openai-data-uri";
33
+ import { coerceNullMessageContentInPlace } from "./openai-shared";
33
34
 
34
35
  export type { ParsedRequest };
35
36
 
@@ -90,6 +91,9 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
90
91
  // for `resolvePromptCacheKey` to pull a cache identity out of inbound
91
92
  // vendor-neutral headers when the body doesn't carry one.
92
93
  rejectUnsupportedExplicitPromptCacheFields(body);
94
+ const request =
95
+ typeof body === "object" && body !== null && !Array.isArray(body) ? (body as Record<string, unknown>) : undefined;
96
+ coerceNullMessageContentInPlace(request?.messages, message => message.role !== "function");
93
97
  const parsed = openaiChatRequestSchema(body);
94
98
  if (parsed instanceof type.errors) {
95
99
  throw new AIError.ValidationError(`openai-chat: ${parsed.summary}`);