@oh-my-pi/pi-ai 18.2.1 → 18.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/README.md +2 -0
- package/dist/types/auth/sqlite-credential-store.d.ts +2 -1
- package/dist/types/auth-gateway/session-state.d.ts +69 -16
- package/dist/types/auth-storage.d.ts +1 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/provider-session-state.d.ts +46 -0
- package/dist/types/providers/amazon-bedrock.d.ts +3 -0
- package/dist/types/providers/aws-sigv4.d.ts +12 -0
- package/dist/types/providers/openai-responses.d.ts +15 -0
- package/dist/types/usage/openai-codex.d.ts +3 -1
- package/dist/types/usage.d.ts +3 -1
- package/package.json +6 -6
- package/src/auth/sqlite-credential-store.ts +8 -33
- package/src/auth-gateway/server.ts +159 -84
- package/src/auth-gateway/session-state.ts +227 -29
- package/src/auth-storage.ts +2 -2
- package/src/index.ts +1 -0
- package/src/provider-session-state.ts +56 -0
- package/src/providers/amazon-bedrock.ts +20 -3
- package/src/providers/anthropic-messages-server.ts +80 -20
- package/src/providers/anthropic-signature.ts +5 -2
- package/src/providers/aws-sigv4.ts +16 -5
- package/src/providers/cursor.ts +53 -9
- package/src/providers/openai-completions.ts +6 -0
- package/src/providers/openai-responses.ts +27 -0
- package/src/usage/openai-codex.ts +94 -11
- package/src/usage.ts +5 -1
- package/src/utils/http-inspector.ts +20 -0
package/src/auth-storage.ts
CHANGED
|
@@ -3783,7 +3783,7 @@ export class AuthStorage {
|
|
|
3783
3783
|
ingestUsageHeaders(
|
|
3784
3784
|
provider: Provider,
|
|
3785
3785
|
headers: Record<string, string>,
|
|
3786
|
-
options?: { sessionId?: string; baseUrl?: string },
|
|
3786
|
+
options?: { sessionId?: string; baseUrl?: string; responseStatus?: number },
|
|
3787
3787
|
): boolean {
|
|
3788
3788
|
if (this.#fetchUsageReportsOverride) return false;
|
|
3789
3789
|
const parseHeaders = this.#resolveUsageProvider(provider)?.parseRateLimitHeaders;
|
|
@@ -3796,7 +3796,7 @@ export class AuthStorage {
|
|
|
3796
3796
|
this.#buildUsageRequestForOauth(provider, credential, options?.baseUrl),
|
|
3797
3797
|
);
|
|
3798
3798
|
const now = Date.now();
|
|
3799
|
-
const parsedReport = parseHeaders(headers, now);
|
|
3799
|
+
const parsedReport = parseHeaders(headers, now, { responseStatus: options?.responseStatus });
|
|
3800
3800
|
if (!parsedReport) return false;
|
|
3801
3801
|
// Throttled to one ingest per interval — except when a window reads
|
|
3802
3802
|
// exhausted: persist that snapshot immediately. A full-backed cache can
|
package/src/index.ts
CHANGED
|
@@ -8,6 +8,7 @@ export * from "./auth-storage";
|
|
|
8
8
|
export * from "./error/rate-limit";
|
|
9
9
|
export * from "./oneshot-retry";
|
|
10
10
|
export * from "./provider-details";
|
|
11
|
+
export * from "./provider-session-state";
|
|
11
12
|
export * from "./providers/anthropic";
|
|
12
13
|
export * from "./providers/anthropic-client";
|
|
13
14
|
export * from "./providers/azure-openai-responses";
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Credential-rotation handling for a retained `providerSessionState` map.
|
|
3
|
+
*
|
|
4
|
+
* A host that keeps one provider-session map per logical conversation (the
|
|
5
|
+
* auth-gateway's server-owned store, an in-process omp session) can outlive the
|
|
6
|
+
* credential that filled it: `AuthStorage.markUsageLimitReached` and the
|
|
7
|
+
* auth-retry resolver both switch a session to a sibling account mid-flight.
|
|
8
|
+
* Most of what a provider learns is a property of the *endpoint*, so rebuilding
|
|
9
|
+
* the whole map on a switch would re-pay every rejected round-trip the map
|
|
10
|
+
* exists to avoid. A minority is a property of the *account*, and keeping that
|
|
11
|
+
* across a switch is a bug.
|
|
12
|
+
*
|
|
13
|
+
* Audit of what the retained records hold, per provider:
|
|
14
|
+
*
|
|
15
|
+
* - **Anthropic** — `fastModeDisabled` is account-scoped: the rejection reads
|
|
16
|
+
* "this model does not support fast mode for your account", i.e. a plan
|
|
17
|
+
* entitlement, so a switch to an entitled sibling must re-probe. Its
|
|
18
|
+
* siblings are endpoint-scoped and stay: `strictToolsDisabled`
|
|
19
|
+
* (grammar-too-large 400 for the model's tool schema),
|
|
20
|
+
* `replayUnsignedThinkingDisabled` / `thinkingReplayDisabled` (the endpoint
|
|
21
|
+
* is a signing proxy), `prefixDroppedThinkingBlocks` (blocks the API itself
|
|
22
|
+
* dropped), `controlStates` (per-conversation control baselines).
|
|
23
|
+
* - **OpenAI Responses** — the `previous_response_id` chain baselines are
|
|
24
|
+
* account-scoped: a stored response belongs to the account that created it.
|
|
25
|
+
* Strict-tools / reasoning-effort fallbacks, replay warmup and the chaining
|
|
26
|
+
* circuit breaker are endpoint-scoped and stay.
|
|
27
|
+
* - **OpenAI Completions** — strict-tools and reasoning-effort fallbacks only;
|
|
28
|
+
* both endpoint-scoped. Nothing to reset.
|
|
29
|
+
* - **Codex** — already sub-keys its WebSocket sessions by account id AND
|
|
30
|
+
* bearer (`getCodexWebSocketSessionKey`), so a switch naturally lands on a
|
|
31
|
+
* fresh transport session while the old one stays reachable for teardown.
|
|
32
|
+
* Resetting from the outside would close a socket a retry may still be on.
|
|
33
|
+
* - **Antigravity** — `lastGoodEndpoint` is endpoint-scoped; the agent /
|
|
34
|
+
* conversation ids are conversation-scoped. Neither depends on the account.
|
|
35
|
+
* - **GitLab Duo** — the active workflow is account-bound, but it is a live
|
|
36
|
+
* server-side workflow plus socket, and the switch happens *inside* the
|
|
37
|
+
* request that may still be resuming it. Tearing it down here would abort the
|
|
38
|
+
* very turn that rotated; it stays on its existing session-close path.
|
|
39
|
+
*/
|
|
40
|
+
|
|
41
|
+
import { clearAnthropicFastModeFallback } from "./providers/anthropic";
|
|
42
|
+
import { resetOpenAIResponsesAccountScopedState } from "./providers/openai-responses";
|
|
43
|
+
import type { ProviderSessionState } from "./types";
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Reset the account-dependent lessons in `states`, keeping everything a
|
|
47
|
+
* provider learned about the endpoint. Call when a retained map is about to be
|
|
48
|
+
* reused for a session whose credential now resolves to a different account.
|
|
49
|
+
*/
|
|
50
|
+
export function resetAccountScopedProviderSessionState(states: Map<string, ProviderSessionState>): void {
|
|
51
|
+
if (states.size === 0) return;
|
|
52
|
+
// Fast mode is the account-scoped half of the Anthropic record; the helper
|
|
53
|
+
// the `/fast on` re-arm path already uses clears exactly that flag.
|
|
54
|
+
clearAnthropicFastModeFallback(states);
|
|
55
|
+
resetOpenAIResponsesAccountScopedState(states);
|
|
56
|
+
}
|
|
@@ -5,6 +5,9 @@
|
|
|
5
5
|
* SigV4 signing and decodes the `application/vnd.amazon.eventstream` response.
|
|
6
6
|
* No `@aws-sdk/*`, no `@smithy/*`, no `proxy-agent`. Proxies are honored via
|
|
7
7
|
* Bun's native `HTTPS_PROXY` support.
|
|
8
|
+
*
|
|
9
|
+
* A `models.yml` `baseUrl` is the request origin verbatim (VPC endpoint, gateway, …);
|
|
10
|
+
* only AWS's own regional host is re-pointed at the resolved region. SigV4 unaffected.
|
|
8
11
|
*/
|
|
9
12
|
|
|
10
13
|
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
|
@@ -145,6 +148,13 @@ const INFERENCE_PROFILE_GEO_DEFAULT_REGION: Record<string, string> = {
|
|
|
145
148
|
jp: "ap-northeast-1",
|
|
146
149
|
};
|
|
147
150
|
|
|
151
|
+
/**
|
|
152
|
+
* AWS's own regional host, which every bundled catalog entry carries as a required
|
|
153
|
+
* placeholder `baseUrl` — no routing info, so its region segment is re-derived.
|
|
154
|
+
* FIPS, VPC-endpoint and gateway hosts don't match and are used as configured.
|
|
155
|
+
*/
|
|
156
|
+
const AWS_REGIONAL_BEDROCK_HOST = /^bedrock-runtime\.[a-z0-9-]+\.amazonaws\.com$/;
|
|
157
|
+
|
|
148
158
|
/** Geo prefix of a cross-region inference-profile id, e.g. `eu.anthropic.…` → `eu`. */
|
|
149
159
|
function inferenceProfileGeo(modelId: string): string | undefined {
|
|
150
160
|
const dot = modelId.indexOf(".");
|
|
@@ -453,9 +463,15 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
453
463
|
// raw dump so the inspector shows exactly what was sent.
|
|
454
464
|
commandInput = { ...commandInput, requestMetadata: sanitizeRequestMetadata(commandInput.requestMetadata) };
|
|
455
465
|
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
466
|
+
// `baseUrl` is the origin verbatim, path prefix (and query, for gateways
|
|
467
|
+
// that authenticate via a query parameter) included, so a gateway mounted
|
|
468
|
+
// under a path works. AWS's own host is re-pointed: the catalog can't know the region.
|
|
469
|
+
const base = new URL(model.baseUrl || `https://bedrock-runtime.${region}.amazonaws.com`);
|
|
470
|
+
if (AWS_REGIONAL_BEDROCK_HOST.test(base.host)) base.host = `bedrock-runtime.${region}.amazonaws.com`;
|
|
471
|
+
const host = base.host;
|
|
472
|
+
const urlPath = `${base.pathname.replace(/\/+$/, "")}/model/${encodeURIComponent(model.id)}/converse-stream`;
|
|
473
|
+
const query = base.search.slice(1) || undefined;
|
|
474
|
+
const url = `${base.origin}${urlPath}${base.search}`;
|
|
459
475
|
rawRequestDump = {
|
|
460
476
|
provider: model.provider,
|
|
461
477
|
api: output.api,
|
|
@@ -527,6 +543,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
527
543
|
method: "POST",
|
|
528
544
|
host,
|
|
529
545
|
path: urlPath,
|
|
546
|
+
query,
|
|
530
547
|
body,
|
|
531
548
|
region,
|
|
532
549
|
service: "bedrock",
|
|
@@ -19,6 +19,7 @@ import type {
|
|
|
19
19
|
ToolResultMessage,
|
|
20
20
|
UserMessage,
|
|
21
21
|
} from "../types";
|
|
22
|
+
import { isCursorExecResolved } from "../utils/block-symbols";
|
|
22
23
|
import {
|
|
23
24
|
type AnthropicAssistantContentBlock,
|
|
24
25
|
type AnthropicMessage,
|
|
@@ -505,14 +506,33 @@ function randomFallback(): string {
|
|
|
505
506
|
return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
|
|
506
507
|
}
|
|
507
508
|
|
|
508
|
-
|
|
509
|
+
/**
|
|
510
|
+
* True for a `toolCall` block the client is expected to execute.
|
|
511
|
+
*
|
|
512
|
+
* Cursor's exec channel stamps {@link kCursorExecResolved} on calls it already
|
|
513
|
+
* ran server-side — `todo`, `web_fetch`, `connect_scm`, a native it declined —
|
|
514
|
+
* and those are not handoffs: the client never declared the tool, has no
|
|
515
|
+
* implementation to run, and repeating one would reapply a side effect the
|
|
516
|
+
* server already committed. Only unresolved calls are real external handoffs.
|
|
517
|
+
*/
|
|
518
|
+
function isClientToolUse(content: AssistantMessage["content"][number]): content is ToolCall {
|
|
519
|
+
return content.type === "toolCall" && !isCursorExecResolved(content);
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
function mapStopReasonOut(reason: StopReason, hasToolUse: boolean): "end_turn" | "max_tokens" | "tool_use" {
|
|
509
523
|
switch (reason) {
|
|
510
524
|
case "length":
|
|
511
525
|
return "max_tokens";
|
|
512
526
|
case "toolUse":
|
|
513
527
|
return "tool_use";
|
|
514
528
|
default:
|
|
515
|
-
|
|
529
|
+
// A provider whose protocol has no separate tool-use stop — Cursor
|
|
530
|
+
// ends the turn with `stop` when it hands a client-declared tool
|
|
531
|
+
// back for the caller to execute — still owes the client
|
|
532
|
+
// `tool_use`, or the canonical Anthropic loop (run tools while
|
|
533
|
+
// `stop_reason === "tool_use"`) never runs the tool it asked for.
|
|
534
|
+
// The OpenAI chat wire maps the same case to `tool_calls`.
|
|
535
|
+
return hasToolUse ? "tool_use" : "end_turn";
|
|
516
536
|
}
|
|
517
537
|
}
|
|
518
538
|
|
|
@@ -536,6 +556,9 @@ function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>
|
|
|
536
556
|
blocks.push(c.block);
|
|
537
557
|
break;
|
|
538
558
|
case "toolCall":
|
|
559
|
+
// Cursor already executed this one; the client must not run it
|
|
560
|
+
// again and cannot answer it. See `isClientToolUse`.
|
|
561
|
+
if (!isClientToolUse(c)) break;
|
|
539
562
|
blocks.push({ type: "tool_use", id: c.id, name: c.name, input: c.arguments ?? {} });
|
|
540
563
|
break;
|
|
541
564
|
}
|
|
@@ -568,7 +591,7 @@ export function encodeResponse(message: AssistantMessage, requestedModelId: stri
|
|
|
568
591
|
role: "assistant",
|
|
569
592
|
model: requestedModelId,
|
|
570
593
|
content: encodeContentBlocks(message),
|
|
571
|
-
stop_reason: mapStopReasonOut(message.stopReason),
|
|
594
|
+
stop_reason: mapStopReasonOut(message.stopReason, message.content.some(isClientToolUse)),
|
|
572
595
|
// TODO: surface the matched stop sequence once pi-ai's
|
|
573
596
|
// `AssistantMessage.stopReason` carries the matched string. Intentionally
|
|
574
597
|
// `null` for now (Anthropic schema allows it).
|
|
@@ -633,6 +656,23 @@ export function encodeStream(
|
|
|
633
656
|
const messageId = newMessageId();
|
|
634
657
|
let started = false;
|
|
635
658
|
const open = new Map<number, OpenBlock>();
|
|
659
|
+
// Cursor's exec channel hands back calls it already ran server-side;
|
|
660
|
+
// `isClientToolUse` keeps them off the wire. Anthropic clients (the
|
|
661
|
+
// official SDK included) append every `content_block_start` to their
|
|
662
|
+
// snapshot and then address deltas by `index`, so a hole in the
|
|
663
|
+
// numbering misroutes each later delta. Shift emitted indices down by
|
|
664
|
+
// the number of blocks suppressed before them — identity while
|
|
665
|
+
// nothing is suppressed. A suppressed call always closes the
|
|
666
|
+
// preceding text/thinking block before it opens, so no block that is
|
|
667
|
+
// still open is ever renumbered.
|
|
668
|
+
const suppressed = new Set<number>();
|
|
669
|
+
const wireIndex = (contentIndex: number): number => {
|
|
670
|
+
let shift = 0;
|
|
671
|
+
for (const index of suppressed) {
|
|
672
|
+
if (index < contentIndex) shift++;
|
|
673
|
+
}
|
|
674
|
+
return contentIndex - shift;
|
|
675
|
+
};
|
|
636
676
|
|
|
637
677
|
const ensureStart = (partial: AssistantMessage | undefined) => {
|
|
638
678
|
if (started) return;
|
|
@@ -663,10 +703,11 @@ export function encodeStream(
|
|
|
663
703
|
const emitServerToolBlocksBefore = (message: AssistantMessage, beforeIndex: number) => {
|
|
664
704
|
const limit = Math.min(beforeIndex, message.content.length);
|
|
665
705
|
while (nextContentIndexToInspect < limit) {
|
|
666
|
-
const
|
|
667
|
-
const content = message.content[
|
|
706
|
+
const contentIndex = nextContentIndexToInspect++;
|
|
707
|
+
const content = message.content[contentIndex];
|
|
668
708
|
if (content?.type !== "anthropicServerTool") continue;
|
|
669
709
|
ensureStart(message);
|
|
710
|
+
const index = wireIndex(contentIndex);
|
|
670
711
|
controller.enqueue(
|
|
671
712
|
sseFrame("content_block_start", {
|
|
672
713
|
type: "content_block_start",
|
|
@@ -678,10 +719,11 @@ export function encodeStream(
|
|
|
678
719
|
}
|
|
679
720
|
};
|
|
680
721
|
|
|
681
|
-
const closeBlock = (
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
722
|
+
const closeBlock = (contentIndex: number) => {
|
|
723
|
+
const block = open.get(contentIndex);
|
|
724
|
+
if (!block) return;
|
|
725
|
+
controller.enqueue(sseFrame("content_block_stop", { type: "content_block_stop", index: block.index }));
|
|
726
|
+
open.delete(contentIndex);
|
|
685
727
|
};
|
|
686
728
|
|
|
687
729
|
pingTimer = setInterval(() => {
|
|
@@ -711,11 +753,12 @@ export function encodeStream(
|
|
|
711
753
|
case "text_start": {
|
|
712
754
|
emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
|
|
713
755
|
ensureStart(ev.partial);
|
|
714
|
-
|
|
756
|
+
const index = wireIndex(ev.contentIndex);
|
|
757
|
+
open.set(ev.contentIndex, { index, kind: "text" });
|
|
715
758
|
controller.enqueue(
|
|
716
759
|
sseFrame("content_block_start", {
|
|
717
760
|
type: "content_block_start",
|
|
718
|
-
index
|
|
761
|
+
index,
|
|
719
762
|
content_block: { type: "text", text: "" },
|
|
720
763
|
}),
|
|
721
764
|
);
|
|
@@ -725,7 +768,7 @@ export function encodeStream(
|
|
|
725
768
|
controller.enqueue(
|
|
726
769
|
sseFrame("content_block_delta", {
|
|
727
770
|
type: "content_block_delta",
|
|
728
|
-
index: ev.contentIndex,
|
|
771
|
+
index: wireIndex(ev.contentIndex),
|
|
729
772
|
delta: { type: "text_delta", text: ev.delta },
|
|
730
773
|
}),
|
|
731
774
|
);
|
|
@@ -736,11 +779,12 @@ export function encodeStream(
|
|
|
736
779
|
case "thinking_start": {
|
|
737
780
|
emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
|
|
738
781
|
ensureStart(ev.partial);
|
|
739
|
-
|
|
782
|
+
const index = wireIndex(ev.contentIndex);
|
|
783
|
+
open.set(ev.contentIndex, { index, kind: "thinking" });
|
|
740
784
|
controller.enqueue(
|
|
741
785
|
sseFrame("content_block_start", {
|
|
742
786
|
type: "content_block_start",
|
|
743
|
-
index
|
|
787
|
+
index,
|
|
744
788
|
content_block: { type: "thinking", thinking: "" },
|
|
745
789
|
}),
|
|
746
790
|
);
|
|
@@ -750,7 +794,7 @@ export function encodeStream(
|
|
|
750
794
|
controller.enqueue(
|
|
751
795
|
sseFrame("content_block_delta", {
|
|
752
796
|
type: "content_block_delta",
|
|
753
|
-
index: ev.contentIndex,
|
|
797
|
+
index: wireIndex(ev.contentIndex),
|
|
754
798
|
delta: { type: "thinking_delta", thinking: ev.delta },
|
|
755
799
|
}),
|
|
756
800
|
);
|
|
@@ -761,7 +805,7 @@ export function encodeStream(
|
|
|
761
805
|
controller.enqueue(
|
|
762
806
|
sseFrame("content_block_delta", {
|
|
763
807
|
type: "content_block_delta",
|
|
764
|
-
index: ev.contentIndex,
|
|
808
|
+
index: wireIndex(ev.contentIndex),
|
|
765
809
|
delta: { type: "signature_delta", signature: c.thinkingSignature },
|
|
766
810
|
}),
|
|
767
811
|
);
|
|
@@ -773,11 +817,21 @@ export function encodeStream(
|
|
|
773
817
|
emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
|
|
774
818
|
ensureStart(ev.partial);
|
|
775
819
|
const tc = ev.partial.content[ev.contentIndex] as ToolCall | undefined;
|
|
776
|
-
|
|
820
|
+
if (tc && !isClientToolUse(tc)) {
|
|
821
|
+
// Cursor's exec channel already ran this call and
|
|
822
|
+
// buffered its result. Streaming it would invite the
|
|
823
|
+
// client to repeat a committed side effect and answer
|
|
824
|
+
// a tool it never declared, so drop the whole block —
|
|
825
|
+
// start, deltas and stop — from the wire.
|
|
826
|
+
suppressed.add(ev.contentIndex);
|
|
827
|
+
break;
|
|
828
|
+
}
|
|
829
|
+
const index = wireIndex(ev.contentIndex);
|
|
830
|
+
open.set(ev.contentIndex, { index, kind: "tool_use" });
|
|
777
831
|
controller.enqueue(
|
|
778
832
|
sseFrame("content_block_start", {
|
|
779
833
|
type: "content_block_start",
|
|
780
|
-
index
|
|
834
|
+
index,
|
|
781
835
|
content_block: {
|
|
782
836
|
type: "tool_use",
|
|
783
837
|
id: tc?.id ?? "",
|
|
@@ -789,10 +843,11 @@ export function encodeStream(
|
|
|
789
843
|
break;
|
|
790
844
|
}
|
|
791
845
|
case "toolcall_delta":
|
|
846
|
+
if (suppressed.has(ev.contentIndex)) break;
|
|
792
847
|
controller.enqueue(
|
|
793
848
|
sseFrame("content_block_delta", {
|
|
794
849
|
type: "content_block_delta",
|
|
795
|
-
index: ev.contentIndex,
|
|
850
|
+
index: wireIndex(ev.contentIndex),
|
|
796
851
|
delta: { type: "input_json_delta", partial_json: ev.delta },
|
|
797
852
|
}),
|
|
798
853
|
);
|
|
@@ -809,7 +864,12 @@ export function encodeStream(
|
|
|
809
864
|
// TODO: surface matched stop sequence once pi-ai
|
|
810
865
|
// propagates it on the `done` event.
|
|
811
866
|
delta: {
|
|
812
|
-
|
|
867
|
+
// A call Cursor resolved after it opened (an MCP
|
|
868
|
+
// frame answered by a local handler) already
|
|
869
|
+
// streamed; the client still must not be told to
|
|
870
|
+
// run it, so it does not terminate the turn with
|
|
871
|
+
// `tool_use` either.
|
|
872
|
+
stop_reason: mapStopReasonOut(ev.reason, ev.message.content.some(isClientToolUse)),
|
|
813
873
|
stop_sequence: null,
|
|
814
874
|
},
|
|
815
875
|
...(bindingControlsRequested
|
|
@@ -17,7 +17,7 @@ const HEADER_MODEL_FIELD = 6;
|
|
|
17
17
|
const MAX_MODEL_ID_LENGTH = 128;
|
|
18
18
|
const MODEL_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._:/-]*$/;
|
|
19
19
|
|
|
20
|
-
/**
|
|
20
|
+
/** Finds a length-delimited field, rejecting tags and lengths that exceed uint32. */
|
|
21
21
|
function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array | undefined {
|
|
22
22
|
let offset = 0;
|
|
23
23
|
while (offset < message.length) {
|
|
@@ -27,6 +27,7 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
|
|
|
27
27
|
do {
|
|
28
28
|
if (offset >= message.length) return undefined;
|
|
29
29
|
byte = message[offset++];
|
|
30
|
+
if (shift === 28 && byte > 0x0f) return undefined;
|
|
30
31
|
tag |= (byte & 0x7f) << shift;
|
|
31
32
|
shift += 7;
|
|
32
33
|
} while (byte & 0x80);
|
|
@@ -49,10 +50,12 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
|
|
|
49
50
|
do {
|
|
50
51
|
if (offset >= message.length) return undefined;
|
|
51
52
|
byte = message[offset++];
|
|
53
|
+
if (shift === 28 && byte > 0x0f) return undefined;
|
|
52
54
|
length |= (byte & 0x7f) << shift;
|
|
53
55
|
shift += 7;
|
|
54
56
|
} while (byte & 0x80);
|
|
55
|
-
|
|
57
|
+
length >>>= 0;
|
|
58
|
+
if (length > message.length - offset) return undefined;
|
|
56
59
|
if (fieldNumber === field) return message.subarray(offset, offset + length);
|
|
57
60
|
offset += length;
|
|
58
61
|
break;
|
|
@@ -137,18 +137,29 @@ function encodeRfc3986(str: string): string {
|
|
|
137
137
|
return encodeURIComponent(str).replace(/[!'()*]/g, c => `%${c.charCodeAt(0).toString(16).toUpperCase()}`);
|
|
138
138
|
}
|
|
139
139
|
|
|
140
|
-
|
|
140
|
+
/**
|
|
141
|
+
* AWS's canonical-request spec encodes each name/value first, THEN sorts by
|
|
142
|
+
* the encoded form ("Sort the encoded parameter names by character code" —
|
|
143
|
+
* https://docs.aws.amazon.com/IAM/latest/UserGuide/create-canonical-request.html).
|
|
144
|
+
* Sorting the decoded form instead gives the wrong order whenever encoding
|
|
145
|
+
* changes a character's relative position — e.g. raw key `%7B` (decodes to
|
|
146
|
+
* `{`, 0x7B) vs `x` (0x78): decoded, `x` < `{`; encoded, `%` (0x25) < `x`, so
|
|
147
|
+
* `%7B` sorts first. A gateway that validates SigV4 (or AWS itself) computes
|
|
148
|
+
* the signature over ITS OWN canonicalization and rejects ours if the two
|
|
149
|
+
* disagree on order.
|
|
150
|
+
*/
|
|
151
|
+
export function canonicalQuery(query: string | undefined): string {
|
|
141
152
|
if (!query) return "";
|
|
142
153
|
const pairs: Array<[string, string]> = [];
|
|
143
154
|
for (const part of query.split("&")) {
|
|
144
155
|
if (!part) continue;
|
|
145
156
|
const eq = part.indexOf("=");
|
|
146
|
-
const
|
|
147
|
-
const
|
|
148
|
-
pairs.push([decodeURIComponent(
|
|
157
|
+
const rawKey = eq === -1 ? part : part.slice(0, eq);
|
|
158
|
+
const rawValue = eq === -1 ? "" : part.slice(eq + 1);
|
|
159
|
+
pairs.push([encodeRfc3986(decodeURIComponent(rawKey)), encodeRfc3986(decodeURIComponent(rawValue))]);
|
|
149
160
|
}
|
|
150
161
|
pairs.sort((a, b) => (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
|
|
151
|
-
return pairs.map(([k, v]) => `${
|
|
162
|
+
return pairs.map(([k, v]) => `${k}=${v}`).join("&");
|
|
152
163
|
}
|
|
153
164
|
|
|
154
165
|
export interface SignedHeaders {
|
package/src/providers/cursor.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { createHash } from "node:crypto";
|
|
2
2
|
import * as fs from "node:fs/promises";
|
|
3
3
|
import http2 from "node:http2";
|
|
4
|
+
import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
|
|
4
5
|
import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
|
|
5
6
|
import type {
|
|
6
7
|
ConversationStep,
|
|
@@ -5234,6 +5235,46 @@ function extractImages(content: (TextContent | ImageContent)[]) {
|
|
|
5234
5235
|
);
|
|
5235
5236
|
}
|
|
5236
5237
|
|
|
5238
|
+
/**
|
|
5239
|
+
* Resolve `max_mode` for the wire id a request actually routes to.
|
|
5240
|
+
*
|
|
5241
|
+
* `GetUsableModels` marks max-mode models per raw row and discovery copies that
|
|
5242
|
+
* onto `cursorMaxMode`, so on a row that puts its own id on the wire the marker
|
|
5243
|
+
* is the authority — Cursor serves the whole Opus `-fast` lane in max mode
|
|
5244
|
+
* (`claude-opus-4-8-high-fast` included) and leaves reasoning tiers such as
|
|
5245
|
+
* `claude-4.6-opus-max` out of it, neither of which the wire slug can tell.
|
|
5246
|
+
*
|
|
5247
|
+
* Collapsing a family ORs the members' markers onto the logical row, so there
|
|
5248
|
+
* `cursorMaxMode: true` only means *some* tier needs max mode; sending it for
|
|
5249
|
+
* every tier is the refused `-low` request of issue #9478. The members' own
|
|
5250
|
+
* markers survive per wire id in `cursorMaxModeRoutes`, so the routed id is
|
|
5251
|
+
* looked up there first.
|
|
5252
|
+
*
|
|
5253
|
+
* A row's own wire id still owns its marker even when it has effort routing
|
|
5254
|
+
* (for example a bare/thinking pair). Logical-only bundled rows and routes
|
|
5255
|
+
* discovery never advertised have no per-id marker; only those use the suffix.
|
|
5256
|
+
* A collapsed row whose `true` no route's suffix can explain keeps it for every
|
|
5257
|
+
* route: the marker came from a member the suffix rule cannot see.
|
|
5258
|
+
*/
|
|
5259
|
+
function resolveCursorMaxMode(model: Model<"cursor-agent">, wireModelId: string): boolean {
|
|
5260
|
+
const discovered = model.cursorMaxModeRoutes?.[wireModelId];
|
|
5261
|
+
if (discovered !== undefined) return discovered;
|
|
5262
|
+
const routing = model.thinking?.effortRouting;
|
|
5263
|
+
if (routing === undefined || wireModelId === model.id) {
|
|
5264
|
+
return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
|
|
5265
|
+
}
|
|
5266
|
+
let routesOwnId = routing.off === model.id;
|
|
5267
|
+
let hasInferredMaxRoute = typeof routing.off === "string" && isCursorMaxModeWireId(routing.off);
|
|
5268
|
+
for (const effort of THINKING_EFFORTS) {
|
|
5269
|
+
const target = routing[effort];
|
|
5270
|
+
if (target === model.id) routesOwnId = true;
|
|
5271
|
+
if (typeof target === "string" && isCursorMaxModeWireId(target)) hasInferredMaxRoute = true;
|
|
5272
|
+
}
|
|
5273
|
+
if (routesOwnId) return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
|
|
5274
|
+
if (model.cursorMaxMode === true && !hasInferredMaxRoute) return true;
|
|
5275
|
+
return isCursorMaxModeWireId(wireModelId);
|
|
5276
|
+
}
|
|
5277
|
+
|
|
5237
5278
|
/**
|
|
5238
5279
|
* Resolve the Cursor Run wire model id and its parameter list.
|
|
5239
5280
|
*
|
|
@@ -5261,9 +5302,11 @@ function resolveCursorWireModel(
|
|
|
5261
5302
|
): {
|
|
5262
5303
|
modelId: string;
|
|
5263
5304
|
parameters: RequestedModel_ModelParameterbytes[];
|
|
5305
|
+
maxMode: boolean;
|
|
5264
5306
|
} {
|
|
5265
5307
|
const wireModelId = requestModelId ?? model.requestModelId ?? model.id;
|
|
5266
|
-
|
|
5308
|
+
const maxMode = resolveCursorMaxMode(model, wireModelId);
|
|
5309
|
+
if (wireMode === "discovered") return { modelId: wireModelId, parameters: [], maxMode };
|
|
5267
5310
|
// `collapseVariantId` keeps the lane in the logical id (`-high-fast` →
|
|
5268
5311
|
// base `-fast`) and decodes the KDL effort (`-none` → `off`).
|
|
5269
5312
|
const collapsed = collapseVariantId("cursor", wireModelId);
|
|
@@ -5271,11 +5314,12 @@ function resolveCursorWireModel(
|
|
|
5271
5314
|
const base = effort !== undefined ? collapsed.logicalId : undefined;
|
|
5272
5315
|
if (effort !== undefined && base && classifyModel("cursor", base).class === "openai") {
|
|
5273
5316
|
if (effort === "off") {
|
|
5274
|
-
return { modelId: base, parameters: [] };
|
|
5317
|
+
return { modelId: base, parameters: [], maxMode };
|
|
5275
5318
|
}
|
|
5276
5319
|
if ((THINKING_EFFORTS as readonly string[]).includes(effort)) {
|
|
5277
5320
|
return {
|
|
5278
5321
|
modelId: base,
|
|
5322
|
+
maxMode,
|
|
5279
5323
|
parameters: [
|
|
5280
5324
|
create(RequestedModel_ModelParameterbytesSchema, { id: "reasoning", value: collapsed.effort }),
|
|
5281
5325
|
],
|
|
@@ -5289,9 +5333,10 @@ function resolveCursorWireModel(
|
|
|
5289
5333
|
return {
|
|
5290
5334
|
modelId: wireModelId,
|
|
5291
5335
|
parameters: [create(RequestedModel_ModelParameterbytesSchema, { id: "fast", value: "false" })],
|
|
5336
|
+
maxMode,
|
|
5292
5337
|
};
|
|
5293
5338
|
}
|
|
5294
|
-
return { modelId: wireModelId, parameters: [] };
|
|
5339
|
+
return { modelId: wireModelId, parameters: [], maxMode };
|
|
5295
5340
|
}
|
|
5296
5341
|
|
|
5297
5342
|
async function buildGrpcRequestForWireMode(
|
|
@@ -5401,12 +5446,11 @@ async function buildGrpcRequestForWireMode(
|
|
|
5401
5446
|
turns,
|
|
5402
5447
|
});
|
|
5403
5448
|
|
|
5404
|
-
const {
|
|
5405
|
-
|
|
5406
|
-
|
|
5407
|
-
|
|
5408
|
-
);
|
|
5409
|
-
const cursorMaxMode = model.cursorMaxMode === true;
|
|
5449
|
+
const {
|
|
5450
|
+
modelId: wireModelId,
|
|
5451
|
+
parameters: wireParameters,
|
|
5452
|
+
maxMode: cursorMaxMode,
|
|
5453
|
+
} = resolveCursorWireModel(model, options?.wireModelId, wireMode);
|
|
5410
5454
|
const modelDetails = create(ModelDetailsSchema, {
|
|
5411
5455
|
modelId: wireModelId,
|
|
5412
5456
|
displayModelId: model.id,
|
|
@@ -1277,6 +1277,12 @@ const streamOpenAICompletionsOnce = (
|
|
|
1277
1277
|
|
|
1278
1278
|
if (choice?.delta?.tool_calls && choice.delta.tool_calls.length > 0) {
|
|
1279
1279
|
const toolCalls = choice.delta.tool_calls;
|
|
1280
|
+
// Pure tool-call responses never emit a text/thinking delta, so
|
|
1281
|
+
// without this stamp TTFT stays undefined for every turn that
|
|
1282
|
+
// begins with a structured call (measured: all toolUse-stop
|
|
1283
|
+
// rows on OpenAI-compatible gateways) and the usage row's
|
|
1284
|
+
// TTFT/tok/s figures silently degrade.
|
|
1285
|
+
if (!firstTokenTime) firstTokenTime = performance.now();
|
|
1280
1286
|
for (let toolCallOffset = 0; toolCallOffset < toolCalls.length; toolCallOffset++) {
|
|
1281
1287
|
const toolCall = toolCalls[toolCallOffset]!;
|
|
1282
1288
|
const streamIndex = typeof toolCall.index === "number" ? toolCall.index : undefined;
|
|
@@ -295,6 +295,33 @@ function resetOpenAIResponsesChainState(state: OpenAIResponsesChainState): void
|
|
|
295
295
|
state.lastPromptCacheBreakpointPolicy = undefined;
|
|
296
296
|
}
|
|
297
297
|
|
|
298
|
+
/**
|
|
299
|
+
* Drop the account-bound half of every retained `openai-responses` record in
|
|
300
|
+
* `states`: the stateful `previous_response_id` chain baselines.
|
|
301
|
+
*
|
|
302
|
+
* Chaining stores the turn server-side under the account that created it, so a
|
|
303
|
+
* baseline minted by one credential is dead weight the moment the session is
|
|
304
|
+
* switched to a sibling account — the next delta request answers
|
|
305
|
+
* `Previous response not found` and burns a turn re-learning that. Everything
|
|
306
|
+
* else this record holds describes the *deployment*, not the account
|
|
307
|
+
* (strict-tools demotion, reasoning-effort fallback, native-history-replay
|
|
308
|
+
* warmup, the chaining circuit breaker), and is deliberately preserved:
|
|
309
|
+
* re-learning an endpoint's limits on every credential switch is the cost this
|
|
310
|
+
* state exists to avoid.
|
|
311
|
+
*/
|
|
312
|
+
export function resetOpenAIResponsesAccountScopedState(states: Map<string, ProviderSessionState>): void {
|
|
313
|
+
for (const [key, value] of states) {
|
|
314
|
+
if (!key.startsWith(OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX)) continue;
|
|
315
|
+
const state = value as OpenAIResponsesProviderSessionState;
|
|
316
|
+
for (const chain of state.chains.values()) {
|
|
317
|
+
resetOpenAIResponsesChainState(chain);
|
|
318
|
+
// The stale-failure counter tallies the previous account's 404s; a
|
|
319
|
+
// fresh account must not inherit a tripped circuit breaker.
|
|
320
|
+
chain.staleFailures = 0;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
|
|
298
325
|
interface OpenAIResponsesChainedParams {
|
|
299
326
|
params: OpenAIResponsesSamplingParams;
|
|
300
327
|
/** Set iff the params carry previous_response_id (delta request). */
|