@oh-my-pi/pi-ai 18.2.0 → 18.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +53 -0
  2. package/README.md +2 -0
  3. package/dist/types/auth/sqlite-credential-store.d.ts +2 -1
  4. package/dist/types/auth-broker/remote-store.d.ts +17 -0
  5. package/dist/types/auth-gateway/index.d.ts +1 -0
  6. package/dist/types/auth-gateway/session-state.d.ts +118 -0
  7. package/dist/types/auth-storage.d.ts +17 -0
  8. package/dist/types/error/body-error.d.ts +15 -0
  9. package/dist/types/error/flags.d.ts +16 -0
  10. package/dist/types/error/index.d.ts +1 -0
  11. package/dist/types/index.d.ts +1 -0
  12. package/dist/types/oneshot-retry.d.ts +6 -0
  13. package/dist/types/provider-session-state.d.ts +46 -0
  14. package/dist/types/providers/amazon-bedrock.d.ts +3 -0
  15. package/dist/types/providers/aws-sigv4.d.ts +12 -0
  16. package/dist/types/providers/openai-codex/request-transformer.d.ts +27 -0
  17. package/dist/types/providers/openai-responses.d.ts +15 -0
  18. package/dist/types/providers/openai-shared.d.ts +20 -3
  19. package/dist/types/registry/oauth/perplexity.d.ts +1 -7
  20. package/dist/types/registry/oauth/types.d.ts +8 -0
  21. package/dist/types/stream.d.ts +2 -0
  22. package/dist/types/types.d.ts +3 -1
  23. package/dist/types/usage/openai-codex.d.ts +3 -1
  24. package/dist/types/usage.d.ts +11 -1
  25. package/dist/types/utils/block-symbols.d.ts +36 -0
  26. package/dist/types/utils/openai-http.d.ts +2 -0
  27. package/dist/types/utils/retry-after.d.ts +2 -0
  28. package/dist/types/utils/schema/wire.d.ts +4 -5
  29. package/dist/types/utils.d.ts +9 -0
  30. package/package.json +6 -6
  31. package/src/auth/sqlite-credential-store.ts +8 -33
  32. package/src/auth-broker/remote-store.ts +73 -8
  33. package/src/auth-broker/wire-schemas.ts +1 -0
  34. package/src/auth-gateway/index.ts +1 -0
  35. package/src/auth-gateway/server.ts +186 -74
  36. package/src/auth-gateway/session-state.ts +312 -0
  37. package/src/auth-storage.ts +146 -15
  38. package/src/error/body-error.ts +310 -0
  39. package/src/error/flags.ts +63 -13
  40. package/src/error/index.ts +1 -0
  41. package/src/error/retryable.ts +2 -0
  42. package/src/index.ts +1 -0
  43. package/src/oneshot-retry.ts +13 -3
  44. package/src/provider-session-state.ts +56 -0
  45. package/src/providers/amazon-bedrock.ts +20 -3
  46. package/src/providers/anthropic-messages-server.ts +104 -23
  47. package/src/providers/anthropic-signature.ts +5 -2
  48. package/src/providers/anthropic.ts +101 -15
  49. package/src/providers/aws-sigv4.ts +16 -5
  50. package/src/providers/cursor.ts +60 -10
  51. package/src/providers/devin.ts +82 -28
  52. package/src/providers/openai-chat-server.ts +4 -0
  53. package/src/providers/openai-codex/request-transformer.ts +36 -0
  54. package/src/providers/openai-codex-responses.ts +35 -12
  55. package/src/providers/openai-completions.ts +49 -12
  56. package/src/providers/openai-reasoning-fallback.ts +6 -6
  57. package/src/providers/openai-responses-server.ts +2 -1
  58. package/src/providers/openai-responses.ts +52 -4
  59. package/src/providers/openai-shared.ts +199 -51
  60. package/src/registry/oauth/perplexity.ts +94 -28
  61. package/src/registry/oauth/types.ts +9 -0
  62. package/src/stream.ts +23 -2
  63. package/src/types.ts +3 -0
  64. package/src/usage/claude.ts +33 -0
  65. package/src/usage/google-antigravity.ts +8 -2
  66. package/src/usage/openai-codex.ts +94 -11
  67. package/src/usage.ts +8 -1
  68. package/src/utils/block-symbols.ts +57 -0
  69. package/src/utils/http-inspector.ts +20 -0
  70. package/src/utils/openai-http.ts +39 -3
  71. package/src/utils/retry-after.ts +12 -0
  72. package/src/utils/schema/normalize.ts +3 -3
  73. package/src/utils/schema/stamps.ts +33 -45
  74. package/src/utils/schema/wire.ts +9 -7
  75. package/src/utils.ts +67 -22
@@ -19,6 +19,7 @@ import type {
19
19
  ToolResultMessage,
20
20
  UserMessage,
21
21
  } from "../types";
22
+ import { isCursorExecResolved } from "../utils/block-symbols";
22
23
  import {
23
24
  type AnthropicAssistantContentBlock,
24
25
  type AnthropicMessage,
@@ -354,6 +355,23 @@ const REASONING_EFFORT_BY_WIRE: Partial<Record<string, Effort>> = {
354
355
  max: Effort.Max,
355
356
  };
356
357
 
358
+ /**
359
+ * Recover the id of the model this request will actually reach, for labelling
360
+ * replayed assistant turns.
361
+ *
362
+ * `/v1/models` advertises `<provider>/<id>` and nothing else, so that is what
363
+ * clients send, but `resolveModel` resolves a catalog model whose id is the
364
+ * bare half. Labelling a replayed turn with the full id the client sent leaves
365
+ * `transform-messages` reading it as written by some other model.
366
+ *
367
+ * A prefix naming another provider is left intact: it describes a different
368
+ * route, so removing it would invent history rather than recover it.
369
+ */
370
+ function stampedAssistantModelId(wireModelId: string, provider: string): string {
371
+ const prefix = `${provider}/`;
372
+ return wireModelId.startsWith(prefix) ? wireModelId.slice(prefix.length) : wireModelId;
373
+ }
374
+
357
375
  export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
358
376
  const data = anthropicMessagesRequestSchema(body);
359
377
  if (data instanceof type.errors) {
@@ -368,14 +386,18 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
368
386
  } else if (message.role === "system") {
369
387
  messages.push(walkSystemMessage(message, now));
370
388
  } else {
389
+ const content = walkAssistantContent(message.content);
371
390
  const assistant: AssistantMessage = {
372
391
  role: "assistant",
373
- content: walkAssistantContent(message.content),
392
+ content,
374
393
  api: "anthropic-messages",
375
394
  provider: "anthropic",
376
- model: data.model,
395
+ model: stampedAssistantModelId(data.model, "anthropic"),
377
396
  usage: emptyUsage(),
378
- stopReason: "stop",
397
+ // The wire carries no stop reason, but tool calls answered by their
398
+ // `tool_result` blocks did request execution. A constant "stop" reads
399
+ // as an abandoned tool-use turn and strips the turn's signatures.
400
+ stopReason: content.some(block => block.type === "toolCall") ? "toolUse" : "stop",
379
401
  timestamp: now,
380
402
  };
381
403
  messages.push(assistant);
@@ -484,14 +506,33 @@ function randomFallback(): string {
484
506
  return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
485
507
  }
486
508
 
487
- function mapStopReasonOut(reason: StopReason): "end_turn" | "max_tokens" | "tool_use" {
509
+ /**
510
+ * True for a `toolCall` block the client is expected to execute.
511
+ *
512
+ * Cursor's exec channel stamps {@link kCursorExecResolved} on calls it already
513
+ * ran server-side — `todo`, `web_fetch`, `connect_scm`, a native it declined —
514
+ * and those are not handoffs: the client never declared the tool, has no
515
+ * implementation to run, and repeating one would reapply a side effect the
516
+ * server already committed. Only unresolved calls are real external handoffs.
517
+ */
518
+ function isClientToolUse(content: AssistantMessage["content"][number]): content is ToolCall {
519
+ return content.type === "toolCall" && !isCursorExecResolved(content);
520
+ }
521
+
522
+ function mapStopReasonOut(reason: StopReason, hasToolUse: boolean): "end_turn" | "max_tokens" | "tool_use" {
488
523
  switch (reason) {
489
524
  case "length":
490
525
  return "max_tokens";
491
526
  case "toolUse":
492
527
  return "tool_use";
493
528
  default:
494
- return "end_turn";
529
+ // A provider whose protocol has no separate tool-use stop — Cursor
530
+ // ends the turn with `stop` when it hands a client-declared tool
531
+ // back for the caller to execute — still owes the client
532
+ // `tool_use`, or the canonical Anthropic loop (run tools while
533
+ // `stop_reason === "tool_use"`) never runs the tool it asked for.
534
+ // The OpenAI chat wire maps the same case to `tool_calls`.
535
+ return hasToolUse ? "tool_use" : "end_turn";
495
536
  }
496
537
  }
497
538
 
@@ -515,6 +556,9 @@ function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>
515
556
  blocks.push(c.block);
516
557
  break;
517
558
  case "toolCall":
559
+ // Cursor already executed this one; the client must not run it
560
+ // again and cannot answer it. See `isClientToolUse`.
561
+ if (!isClientToolUse(c)) break;
518
562
  blocks.push({ type: "tool_use", id: c.id, name: c.name, input: c.arguments ?? {} });
519
563
  break;
520
564
  }
@@ -547,7 +591,7 @@ export function encodeResponse(message: AssistantMessage, requestedModelId: stri
547
591
  role: "assistant",
548
592
  model: requestedModelId,
549
593
  content: encodeContentBlocks(message),
550
- stop_reason: mapStopReasonOut(message.stopReason),
594
+ stop_reason: mapStopReasonOut(message.stopReason, message.content.some(isClientToolUse)),
551
595
  // TODO: surface the matched stop sequence once pi-ai's
552
596
  // `AssistantMessage.stopReason` carries the matched string. Intentionally
553
597
  // `null` for now (Anthropic schema allows it).
@@ -612,6 +656,23 @@ export function encodeStream(
612
656
  const messageId = newMessageId();
613
657
  let started = false;
614
658
  const open = new Map<number, OpenBlock>();
659
+ // Cursor's exec channel hands back calls it already ran server-side;
660
+ // `isClientToolUse` keeps them off the wire. Anthropic clients (the
661
+ // official SDK included) append every `content_block_start` to their
662
+ // snapshot and then address deltas by `index`, so a hole in the
663
+ // numbering misroutes each later delta. Shift emitted indices down by
664
+ // the number of blocks suppressed before them — identity while
665
+ // nothing is suppressed. A suppressed call always closes the
666
+ // preceding text/thinking block before it opens, so no block that is
667
+ // still open is ever renumbered.
668
+ const suppressed = new Set<number>();
669
+ const wireIndex = (contentIndex: number): number => {
670
+ let shift = 0;
671
+ for (const index of suppressed) {
672
+ if (index < contentIndex) shift++;
673
+ }
674
+ return contentIndex - shift;
675
+ };
615
676
 
616
677
  const ensureStart = (partial: AssistantMessage | undefined) => {
617
678
  if (started) return;
@@ -642,10 +703,11 @@ export function encodeStream(
642
703
  const emitServerToolBlocksBefore = (message: AssistantMessage, beforeIndex: number) => {
643
704
  const limit = Math.min(beforeIndex, message.content.length);
644
705
  while (nextContentIndexToInspect < limit) {
645
- const index = nextContentIndexToInspect++;
646
- const content = message.content[index];
706
+ const contentIndex = nextContentIndexToInspect++;
707
+ const content = message.content[contentIndex];
647
708
  if (content?.type !== "anthropicServerTool") continue;
648
709
  ensureStart(message);
710
+ const index = wireIndex(contentIndex);
649
711
  controller.enqueue(
650
712
  sseFrame("content_block_start", {
651
713
  type: "content_block_start",
@@ -657,10 +719,11 @@ export function encodeStream(
657
719
  }
658
720
  };
659
721
 
660
- const closeBlock = (index: number) => {
661
- if (!open.has(index)) return;
662
- controller.enqueue(sseFrame("content_block_stop", { type: "content_block_stop", index }));
663
- open.delete(index);
722
+ const closeBlock = (contentIndex: number) => {
723
+ const block = open.get(contentIndex);
724
+ if (!block) return;
725
+ controller.enqueue(sseFrame("content_block_stop", { type: "content_block_stop", index: block.index }));
726
+ open.delete(contentIndex);
664
727
  };
665
728
 
666
729
  pingTimer = setInterval(() => {
@@ -690,11 +753,12 @@ export function encodeStream(
690
753
  case "text_start": {
691
754
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
692
755
  ensureStart(ev.partial);
693
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "text" });
756
+ const index = wireIndex(ev.contentIndex);
757
+ open.set(ev.contentIndex, { index, kind: "text" });
694
758
  controller.enqueue(
695
759
  sseFrame("content_block_start", {
696
760
  type: "content_block_start",
697
- index: ev.contentIndex,
761
+ index,
698
762
  content_block: { type: "text", text: "" },
699
763
  }),
700
764
  );
@@ -704,7 +768,7 @@ export function encodeStream(
704
768
  controller.enqueue(
705
769
  sseFrame("content_block_delta", {
706
770
  type: "content_block_delta",
707
- index: ev.contentIndex,
771
+ index: wireIndex(ev.contentIndex),
708
772
  delta: { type: "text_delta", text: ev.delta },
709
773
  }),
710
774
  );
@@ -715,11 +779,12 @@ export function encodeStream(
715
779
  case "thinking_start": {
716
780
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
717
781
  ensureStart(ev.partial);
718
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
782
+ const index = wireIndex(ev.contentIndex);
783
+ open.set(ev.contentIndex, { index, kind: "thinking" });
719
784
  controller.enqueue(
720
785
  sseFrame("content_block_start", {
721
786
  type: "content_block_start",
722
- index: ev.contentIndex,
787
+ index,
723
788
  content_block: { type: "thinking", thinking: "" },
724
789
  }),
725
790
  );
@@ -729,7 +794,7 @@ export function encodeStream(
729
794
  controller.enqueue(
730
795
  sseFrame("content_block_delta", {
731
796
  type: "content_block_delta",
732
- index: ev.contentIndex,
797
+ index: wireIndex(ev.contentIndex),
733
798
  delta: { type: "thinking_delta", thinking: ev.delta },
734
799
  }),
735
800
  );
@@ -740,7 +805,7 @@ export function encodeStream(
740
805
  controller.enqueue(
741
806
  sseFrame("content_block_delta", {
742
807
  type: "content_block_delta",
743
- index: ev.contentIndex,
808
+ index: wireIndex(ev.contentIndex),
744
809
  delta: { type: "signature_delta", signature: c.thinkingSignature },
745
810
  }),
746
811
  );
@@ -752,11 +817,21 @@ export function encodeStream(
752
817
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
753
818
  ensureStart(ev.partial);
754
819
  const tc = ev.partial.content[ev.contentIndex] as ToolCall | undefined;
755
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "tool_use" });
820
+ if (tc && !isClientToolUse(tc)) {
821
+ // Cursor's exec channel already ran this call and
822
+ // buffered its result. Streaming it would invite the
823
+ // client to repeat a committed side effect and answer
824
+ // a tool it never declared, so drop the whole block —
825
+ // start, deltas and stop — from the wire.
826
+ suppressed.add(ev.contentIndex);
827
+ break;
828
+ }
829
+ const index = wireIndex(ev.contentIndex);
830
+ open.set(ev.contentIndex, { index, kind: "tool_use" });
756
831
  controller.enqueue(
757
832
  sseFrame("content_block_start", {
758
833
  type: "content_block_start",
759
- index: ev.contentIndex,
834
+ index,
760
835
  content_block: {
761
836
  type: "tool_use",
762
837
  id: tc?.id ?? "",
@@ -768,10 +843,11 @@ export function encodeStream(
768
843
  break;
769
844
  }
770
845
  case "toolcall_delta":
846
+ if (suppressed.has(ev.contentIndex)) break;
771
847
  controller.enqueue(
772
848
  sseFrame("content_block_delta", {
773
849
  type: "content_block_delta",
774
- index: ev.contentIndex,
850
+ index: wireIndex(ev.contentIndex),
775
851
  delta: { type: "input_json_delta", partial_json: ev.delta },
776
852
  }),
777
853
  );
@@ -788,7 +864,12 @@ export function encodeStream(
788
864
  // TODO: surface matched stop sequence once pi-ai
789
865
  // propagates it on the `done` event.
790
866
  delta: {
791
- stop_reason: mapStopReasonOut(ev.reason),
867
+ // A call Cursor resolved after it opened (an MCP
868
+ // frame answered by a local handler) already
869
+ // streamed; the client still must not be told to
870
+ // run it, so it does not terminate the turn with
871
+ // `tool_use` either.
872
+ stop_reason: mapStopReasonOut(ev.reason, ev.message.content.some(isClientToolUse)),
792
873
  stop_sequence: null,
793
874
  },
794
875
  ...(bindingControlsRequested
@@ -17,7 +17,7 @@ const HEADER_MODEL_FIELD = 6;
17
17
  const MAX_MODEL_ID_LENGTH = 128;
18
18
  const MODEL_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._:/-]*$/;
19
19
 
20
- /** Returns the first length-delimited field `field` in `message`, or undefined. */
20
+ /** Finds a length-delimited field, rejecting tags and lengths that exceed uint32. */
21
21
  function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array | undefined {
22
22
  let offset = 0;
23
23
  while (offset < message.length) {
@@ -27,6 +27,7 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
27
27
  do {
28
28
  if (offset >= message.length) return undefined;
29
29
  byte = message[offset++];
30
+ if (shift === 28 && byte > 0x0f) return undefined;
30
31
  tag |= (byte & 0x7f) << shift;
31
32
  shift += 7;
32
33
  } while (byte & 0x80);
@@ -49,10 +50,12 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
49
50
  do {
50
51
  if (offset >= message.length) return undefined;
51
52
  byte = message[offset++];
53
+ if (shift === 28 && byte > 0x0f) return undefined;
52
54
  length |= (byte & 0x7f) << shift;
53
55
  shift += 7;
54
56
  } while (byte & 0x80);
55
- if (offset + length > message.length) return undefined;
57
+ length >>>= 0;
58
+ if (length > message.length - offset) return undefined;
56
59
  if (fieldNumber === field) return message.subarray(offset, offset + length);
57
60
  offset += length;
58
61
  break;
@@ -60,8 +60,10 @@ import {
60
60
  import { createAbortSourceTracker } from "../utils/abort";
61
61
  import {
62
62
  clearStreamingPartialJson,
63
+ copyPerCallContextMessage,
63
64
  type ConversationalUserCarrier,
64
65
  isConversationalUser,
66
+ isPerCallContextMessage,
65
67
  isSyntheticUser,
66
68
  kConversationalUser,
67
69
  kStreamingBlockIndex,
@@ -454,6 +456,15 @@ type AnthropicProviderSessionState = ProviderSessionState & {
454
456
  * `compat.replayUnsignedThinking: false`. Cleared on session close.
455
457
  */
456
458
  replayUnsignedThinkingDisabled: boolean;
459
+ /**
460
+ * Runtime-learned: this endpoint kept rejecting replayed thinking
461
+ * signatures even after unsigned demotion — every surviving block is
462
+ * signed by a foreign signer (e.g. a failover proxy swapped upstreams
463
+ * mid-conversation and minted signatures the restored upstream cannot
464
+ * verify). All subsequent requests drop replayed thinking entirely for
465
+ * this (baseUrl, modelId). Cleared on session close.
466
+ */
467
+ thinkingReplayDisabled: boolean;
457
468
  /** Thinking blocks the API permanently dropped after a prefix mismatch. */
458
469
  prefixDroppedThinkingBlocks: Set<string>;
459
470
  /** Conversation-scoped control baselines, isolated from side requests and advisors. */
@@ -478,12 +489,14 @@ function createAnthropicProviderSessionState(): AnthropicProviderSessionState {
478
489
  strictToolsDisabled: false,
479
490
  fastModeDisabled: false,
480
491
  replayUnsignedThinkingDisabled: false,
492
+ thinkingReplayDisabled: false,
481
493
  prefixDroppedThinkingBlocks: new Set(),
482
494
  controlStates: new Map(),
483
495
  close: () => {
484
496
  state.strictToolsDisabled = false;
485
497
  state.fastModeDisabled = false;
486
498
  state.replayUnsignedThinkingDisabled = false;
499
+ state.thinkingReplayDisabled = false;
487
500
  state.prefixDroppedThinkingBlocks.clear();
488
501
  state.controlStates.clear();
489
502
  },
@@ -2296,7 +2309,8 @@ const streamAnthropicOnce = (
2296
2309
  (providerSessionState?.strictToolsDisabled ?? false) || (model.compat?.disableStrictTools ?? false);
2297
2310
  let dropFastMode = providerSessionState?.fastModeDisabled ?? false;
2298
2311
  let forceDemoteUnsignedThinking = providerSessionState?.replayUnsignedThinkingDisabled ?? false;
2299
- let dropAllThinking = false;
2312
+ let droppedAllThinkingForSignature = providerSessionState?.thinkingReplayDisabled ?? false;
2313
+ let dropAllThinking = droppedAllThinkingForSignature;
2300
2314
  let prefixBindingRetryAttempted = false;
2301
2315
  let prefixMismatchBehavior =
2302
2316
  model.thinking?.prefixBinding && model.compat.supportsThinkingBindingControls
@@ -3367,6 +3381,48 @@ const streamAnthropicOnce = (
3367
3381
  firstTokenTime = undefined;
3368
3382
  continue;
3369
3383
  }
3384
+ if (
3385
+ !dropAllThinking &&
3386
+ firstTokenTime === undefined &&
3387
+ !streamedReplayUnsafeContent &&
3388
+ !isThinkingPrefixBindingError(streamFailureMessage) &&
3389
+ isInvalidThinkingSignatureError(streamFailureMessage)
3390
+ ) {
3391
+ // The unsigned-demotion retry only rewrites UNSIGNED blocks;
3392
+ // when every replayed block carries a signature the signer no
3393
+ // longer accepts (e.g. a failover proxy swapped upstreams
3394
+ // mid-conversation and minted foreign signatures), the retry
3395
+ // resends a byte-identical body and the session 400s forever.
3396
+ // Escalate: drop all replayed thinking — prior-turn reasoning
3397
+ // is optional context — and retry once. Stored history keeps
3398
+ // its thinking blocks; only the wire payload changes.
3399
+ logger.warn(
3400
+ "anthropic: thinking signatures still rejected after unsigned demotion, dropping replayed thinking and retrying",
3401
+ {
3402
+ provider: model.provider,
3403
+ model: model.id,
3404
+ baseUrl,
3405
+ error: streamFailureMessage,
3406
+ },
3407
+ );
3408
+ if (providerSessionState) {
3409
+ providerSessionState.thinkingReplayDisabled = true;
3410
+ }
3411
+ droppedAllThinkingForSignature = true;
3412
+ dropAllThinking = true;
3413
+ params = await prepareParams();
3414
+ providerRetryAttempt = 0;
3415
+ output.content.length = 0;
3416
+ output.model = model.id;
3417
+ output.responseId = undefined;
3418
+ output.errorMessage = undefined;
3419
+ output.inputTransformations = undefined;
3420
+ output.providerPayload = undefined;
3421
+ output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
3422
+ output.stopReason = "stop";
3423
+ firstTokenTime = undefined;
3424
+ continue;
3425
+ }
3370
3426
  if (
3371
3427
  !dropFastMode &&
3372
3428
  model.provider === "anthropic" &&
@@ -3451,6 +3507,9 @@ const streamAnthropicOnce = (
3451
3507
  if (forceDemoteUnsignedThinking && model.compat.replayUnsignedThinking) {
3452
3508
  output.disabledFeatures = [...(output.disabledFeatures ?? []), "unsigned-thinking-replay"];
3453
3509
  }
3510
+ if (droppedAllThinkingForSignature) {
3511
+ output.disabledFeatures = [...(output.disabledFeatures ?? []), "thinking-replay"];
3512
+ }
3454
3513
  stream.push({ type: "done", reason: output.stopReason, message: output });
3455
3514
  stream.end();
3456
3515
  } catch (error) {
@@ -3892,14 +3951,26 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
3892
3951
  params.messages[trailingIndex - 1]?.role === "assistant";
3893
3952
  const messageEnd = hasTrailingAssistantPad ? trailingIndex - 1 : trailingIndex;
3894
3953
 
3954
+ // A breakpoint caches every preceding byte, not only the decorated message.
3955
+ // Once per-call or turn-scoped content appears, no later message can anchor a
3956
+ // prefix reusable by the next request.
3957
+ let stableMessageEnd = messageEnd;
3958
+ for (let index = 0; index <= messageEnd; index++) {
3959
+ const message = params.messages[index];
3960
+ if (message && (message.clear_at === "next_user_message" || isPerCallContextMessage(message))) {
3961
+ stableMessageEnd = index - 1;
3962
+ break;
3963
+ }
3964
+ }
3965
+
3895
3966
  // Decimation counts conversational turns, so it reads the provenance marker
3896
3967
  // `convertAnthropicMessages` records rather than the wire role. A wire `user`
3897
3968
  // can also be a serialized `developer` message, a tool_result run, or an
3898
3969
  // interior `Continue.` pad, none of which advance the user turn ordinal.
3899
3970
  const userIndices: number[] = [];
3900
- for (let index = 0; index <= messageEnd; index++) {
3971
+ for (let index = 0; index <= stableMessageEnd; index++) {
3901
3972
  const message = params.messages[index];
3902
- if (message && message.clear_at !== "next_user_message" && isConversationalUser(message)) {
3973
+ if (message && isConversationalUser(message)) {
3903
3974
  userIndices.push(index);
3904
3975
  }
3905
3976
  }
@@ -3907,11 +3978,11 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
3907
3978
  // Stable historical decimation checkpoint every 15 user turns (15th, 30th, 45th...)
3908
3979
  const decimationIndices = userIndices.filter((_, ordinal) => (ordinal + 1) % ANTHROPIC_DECIMATION_INTERVAL === 0);
3909
3980
 
3910
- // Collect eligible trailing candidates (up to 2 messages walking backward from messageEnd).
3981
+ // Collect up to 2 trailing candidates from the reusable prefix.
3911
3982
  const trailingCandidates: number[] = [];
3912
- for (let index = messageEnd; index >= 0 && trailingCandidates.length < 2; index--) {
3983
+ for (let index = stableMessageEnd; index >= 0 && trailingCandidates.length < 2; index--) {
3913
3984
  const message = params.messages[index];
3914
- if (!message || message.clear_at === "next_user_message") continue;
3985
+ if (!message) continue;
3915
3986
  trailingCandidates.push(index);
3916
3987
  }
3917
3988
 
@@ -4774,7 +4845,12 @@ export function convertAnthropicMessages(
4774
4845
  (msg.role === "user" || msg.role === "developer") &&
4775
4846
  isReplayableAnthropicCompaction(msg.providerPayload, model)
4776
4847
  ) {
4777
- params.push({ role: "assistant", content: [compactionBlockParam(msg.providerPayload)] });
4848
+ const compactionParam: AnthropicMessageParam = {
4849
+ role: "assistant",
4850
+ content: [compactionBlockParam(msg.providerPayload)],
4851
+ };
4852
+ copyPerCallContextMessage(compactionParam, msg);
4853
+ params.push(compactionParam);
4778
4854
  // The block carries the verbatim API summary, so the message text
4779
4855
  // (which holds the harness file lists) would be dropped with it.
4780
4856
  // Queue the file metadata for after the block: it sits past the
@@ -4845,6 +4921,7 @@ export function convertAnthropicMessages(
4845
4921
  if (msg.role === "user" && !agentAuthored && !isSyntheticUser(msg)) {
4846
4922
  param[kConversationalUser] = true;
4847
4923
  }
4924
+ copyPerCallContextMessage(param, msg);
4848
4925
  params.push(param);
4849
4926
  } else if (msg.role === "assistant") {
4850
4927
  const blocks: ContentBlockParam[] = [];
@@ -4982,10 +5059,12 @@ export function convertAnthropicMessages(
4982
5059
  blocks.push(...nonToolUse, ...toolUse);
4983
5060
  }
4984
5061
  if (blocks.length === 0) continue;
4985
- params.push({
5062
+ const assistantParam: AnthropicMessageParam = {
4986
5063
  role: "assistant",
4987
5064
  content: blocks,
4988
- });
5065
+ };
5066
+ copyPerCallContextMessage(assistantParam, msg);
5067
+ params.push(assistantParam);
4989
5068
  // Flush queued file metadata unless this turn left tool calls open:
4990
5069
  // their results must follow the turn contiguously, so the metadata
4991
5070
  // waits for the merged result message (or the end of the list).
@@ -4997,15 +5076,21 @@ export function convertAnthropicMessages(
4997
5076
  const toolResults: ContentBlockParam[] = [];
4998
5077
  // Images stripped out of error tool results, re-attached after the run.
4999
5078
  const hoistedImages: ContentBlockParam[] = [];
5079
+ const toolResultParam: AnthropicMessageParam = {
5080
+ role: "user",
5081
+ content: toolResults,
5082
+ };
5000
5083
 
5001
5084
  // Add the current tool result
5002
5085
  toolResults.push(buildToolResultBlock(model, msg, hoistedImages));
5086
+ copyPerCallContextMessage(toolResultParam, msg);
5003
5087
 
5004
5088
  // Look ahead for consecutive toolResult messages
5005
5089
  let j = i + 1;
5006
5090
  while (j < transformedMessages.length && transformedMessages[j].role === "toolResult") {
5007
5091
  const nextMsg = transformedMessages[j] as ToolResultMessage; // We know it's a toolResult
5008
5092
  toolResults.push(buildToolResultBlock(model, nextMsg, hoistedImages));
5093
+ copyPerCallContextMessage(toolResultParam, nextMsg);
5009
5094
  j++;
5010
5095
  }
5011
5096
 
@@ -5020,10 +5105,7 @@ export function convertAnthropicMessages(
5020
5105
  }
5021
5106
 
5022
5107
  // Add a single user message with all tool results
5023
- params.push({
5024
- role: "user",
5025
- content: toolResults,
5026
- });
5108
+ params.push(toolResultParam);
5027
5109
  // An open tool_use turn's results are whole again; queued file
5028
5110
  // metadata can follow without splitting the pairing.
5029
5111
  flushCompactionFiles();
@@ -5060,20 +5142,24 @@ export function convertAnthropicMessages(
5060
5142
  const controlContent = content.filter(block => block.type !== "text");
5061
5143
  if (scopedContent.length > 0) {
5062
5144
  params[idx] = {
5145
+ ...params[idx],
5063
5146
  role: "system",
5064
5147
  content: scopedContent,
5065
5148
  clear_at: "next_user_message",
5066
5149
  };
5067
- params.splice(idx + 1, 0, {
5150
+ const controlParam: AnthropicMessageParam = {
5068
5151
  role: "system",
5069
5152
  content: controlContent,
5070
5153
  ...(hasEffort ? { output_config: { effort: developer.payload?.effort } } : {}),
5071
- });
5154
+ };
5155
+ copyPerCallContextMessage(controlParam, params[idx]);
5156
+ params.splice(idx + 1, 0, controlParam);
5072
5157
  continue;
5073
5158
  }
5074
5159
  }
5075
5160
 
5076
5161
  params[idx] = {
5162
+ ...params[idx],
5077
5163
  role: "system",
5078
5164
  content,
5079
5165
  ...(turnScoped && !hasEffort && !hasToolChanges ? { clear_at: "next_user_message" } : {}),
@@ -137,18 +137,29 @@ function encodeRfc3986(str: string): string {
137
137
  return encodeURIComponent(str).replace(/[!'()*]/g, c => `%${c.charCodeAt(0).toString(16).toUpperCase()}`);
138
138
  }
139
139
 
140
- function canonicalQuery(query: string | undefined): string {
140
+ /**
141
+ * AWS's canonical-request spec encodes each name/value first, THEN sorts by
142
+ * the encoded form ("Sort the encoded parameter names by character code" —
143
+ * https://docs.aws.amazon.com/IAM/latest/UserGuide/create-canonical-request.html).
144
+ * Sorting the decoded form instead gives the wrong order whenever encoding
145
+ * changes a character's relative position — e.g. raw key `%7B` (decodes to
146
+ * `{`, 0x7B) vs `x` (0x78): decoded, `x` < `{`; encoded, `%` (0x25) < `x`, so
147
+ * `%7B` sorts first. A gateway that validates SigV4 (or AWS itself) computes
148
+ * the signature over ITS OWN canonicalization and rejects ours if the two
149
+ * disagree on order.
150
+ */
151
+ export function canonicalQuery(query: string | undefined): string {
141
152
  if (!query) return "";
142
153
  const pairs: Array<[string, string]> = [];
143
154
  for (const part of query.split("&")) {
144
155
  if (!part) continue;
145
156
  const eq = part.indexOf("=");
146
- const k = eq === -1 ? part : part.slice(0, eq);
147
- const v = eq === -1 ? "" : part.slice(eq + 1);
148
- pairs.push([decodeURIComponent(k), decodeURIComponent(v)]);
157
+ const rawKey = eq === -1 ? part : part.slice(0, eq);
158
+ const rawValue = eq === -1 ? "" : part.slice(eq + 1);
159
+ pairs.push([encodeRfc3986(decodeURIComponent(rawKey)), encodeRfc3986(decodeURIComponent(rawValue))]);
149
160
  }
150
161
  pairs.sort((a, b) => (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
151
- return pairs.map(([k, v]) => `${encodeRfc3986(k)}=${encodeRfc3986(v)}`).join("&");
162
+ return pairs.map(([k, v]) => `${k}=${v}`).join("&");
152
163
  }
153
164
 
154
165
  export interface SignedHeaders {