@code-yeongyu/senpi-ai 2026.9.30 → 2026.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +189 -19
- package/dist/api/anthropic-messages.js +168 -32
- package/dist/api/azure-openai-responses.js +23 -10
- package/dist/api/bedrock-converse-stream.js +12 -4
- package/dist/api/cloudflare-workers-ai-system-one.d.ts +4 -0
- package/dist/api/cloudflare-workers-ai-system-one.js +43 -0
- package/dist/api/cloudflare-workers-ai-system-one.lazy.d.ts +3 -0
- package/dist/api/cloudflare-workers-ai-system-one.lazy.js +4 -0
- package/dist/api/cloudflare.d.ts +2 -0
- package/dist/api/cloudflare.js +2 -0
- package/dist/api/context-room.d.ts +2 -2
- package/dist/api/cursor-agent.js +19 -10
- package/dist/api/devin-agent/request.d.ts +9 -9
- package/dist/api/devin-agent/request.js +14 -10
- package/dist/api/google-generative-ai.js +20 -80
- package/dist/api/google-shared.d.ts +13 -4
- package/dist/api/google-shared.js +54 -4
- package/dist/api/google-vertex.js +19 -62
- package/dist/api/llama-cpp-classify.d.ts +33 -0
- package/dist/api/llama-cpp-classify.js +365 -0
- package/dist/api/llama-cpp-classify.lazy.d.ts +3 -0
- package/dist/api/llama-cpp-classify.lazy.js +4 -0
- package/dist/api/mistral-conversations.d.ts +1 -1
- package/dist/api/mistral-conversations.js +36 -29
- package/dist/api/openai-codex-responses.d.ts +1 -1
- package/dist/api/openai-codex-responses.js +72 -37
- package/dist/api/openai-completions.d.ts +3 -2
- package/dist/api/openai-completions.js +76 -50
- package/dist/api/openai-images-params.d.ts +2 -2
- package/dist/api/openai-images.d.ts +1 -1
- package/dist/api/openai-responses-shared.d.ts +31 -8
- package/dist/api/openai-responses-shared.js +115 -27
- package/dist/api/openai-responses.d.ts +1 -1
- package/dist/api/openai-responses.js +38 -33
- package/dist/api/openrouter-images.d.ts +2 -1
- package/dist/api/openrouter-images.js +1 -0
- package/dist/api/pi-messages.d.ts +3 -3
- package/dist/api/pi-messages.js +3 -2
- package/dist/api/simple-options.d.ts +2 -2
- package/dist/api/simple-options.js +1 -0
- package/dist/api/system-one-shared.d.ts +23 -0
- package/dist/api/system-one-shared.js +183 -0
- package/dist/api/transform-messages.js +5 -2
- package/dist/api/typesafe-system-one.d.ts +4 -0
- package/dist/api/typesafe-system-one.js +19 -0
- package/dist/api/typesafe-system-one.lazy.d.ts +3 -0
- package/dist/api/typesafe-system-one.lazy.js +4 -0
- package/dist/api-registry.d.ts +3 -3
- package/dist/auth/helpers.js +1 -1
- package/dist/auth/oauth/anthropic-callback-listener.js +1 -1
- package/dist/auth/oauth/callback-server.d.ts +55 -0
- package/dist/auth/oauth/callback-server.js +146 -0
- package/dist/auth/oauth/chatgpt-subscription.d.ts +1 -1
- package/dist/auth/oauth/chatgpt-subscription.js +20 -124
- package/dist/auth/oauth/devin-callback.js +1 -1
- package/dist/auth/oauth/load.d.ts +4 -0
- package/dist/auth/oauth/load.js +10 -0
- package/dist/auth/oauth/meta.d.ts +17 -0
- package/dist/auth/oauth/meta.js +190 -0
- package/dist/auth/oauth/openai-chatgpt.d.ts +9 -0
- package/dist/auth/oauth/openai-chatgpt.js +266 -0
- package/dist/auth/oauth/openrouter.d.ts +1 -1
- package/dist/auth/oauth/openrouter.js +19 -138
- package/dist/auth/oauth/radius.d.ts +1 -1
- package/dist/auth/oauth/radius.js +21 -89
- package/dist/auth/resolve.d.ts +3 -8
- package/dist/auth/resolve.js +3 -18
- package/dist/auth/types.d.ts +10 -1
- package/dist/bun-oauth.js +4 -0
- package/dist/cli.js +3 -1
- package/dist/compat.js +17 -14
- package/dist/env-api-keys.js +2 -0
- package/dist/image-models.d.ts +19 -8
- package/dist/image-models.js +14 -13
- package/dist/images-api-registry.d.ts +8 -8
- package/dist/images.d.ts +7 -2
- package/dist/images.js +5 -0
- package/dist/index.d.ts +2 -2
- package/dist/index.js +2 -2
- package/dist/model-catalog.d.ts +27 -9
- package/dist/model-catalog.js +24 -2
- package/dist/model.d.ts +13 -13
- package/dist/models-store.d.ts +3 -2
- package/dist/models.d.ts +106 -36
- package/dist/models.generated.d.ts +142 -42
- package/dist/models.generated.js +142 -42
- package/dist/models.js +174 -36
- package/dist/providers/alibaba-token-plan.models.d.ts +4 -2
- package/dist/providers/alibaba-token-plan.models.js +4 -2
- package/dist/providers/all.d.ts +72 -19
- package/dist/providers/all.js +28 -22
- package/dist/providers/amazon-bedrock.models.d.ts +4 -2
- package/dist/providers/amazon-bedrock.models.js +4 -2
- package/dist/providers/ant-ling.models.d.ts +4 -2
- package/dist/providers/ant-ling.models.js +4 -2
- package/dist/providers/anthropic.models.d.ts +4 -2
- package/dist/providers/anthropic.models.js +4 -2
- package/dist/providers/azure-openai-responses.models.d.ts +4 -2
- package/dist/providers/azure-openai-responses.models.js +4 -2
- package/dist/providers/bai.models.d.ts +4 -2
- package/dist/providers/bai.models.js +4 -2
- package/dist/providers/baseten.models.d.ts +4 -2
- package/dist/providers/baseten.models.js +4 -2
- package/dist/providers/cerebras.models.d.ts +4 -2
- package/dist/providers/cerebras.models.js +4 -2
- package/dist/providers/chatgpt-subscription.models.d.ts +4 -2
- package/dist/providers/chatgpt-subscription.models.js +4 -2
- package/dist/providers/cloudflare-ai-gateway.models.d.ts +4 -2
- package/dist/providers/cloudflare-ai-gateway.models.js +4 -2
- package/dist/providers/cloudflare-stream.d.ts +6 -2
- package/dist/providers/cloudflare-stream.js +6 -0
- package/dist/providers/cloudflare-workers-ai.js +10 -3
- package/dist/providers/cloudflare-workers-ai.models.d.ts +4 -2
- package/dist/providers/cloudflare-workers-ai.models.js +4 -2
- package/dist/providers/data/.manifest.json +1 -1
- package/dist/providers/data/alibaba-token-plan.json +1 -1
- package/dist/providers/data/amazon-bedrock.json +1 -1
- package/dist/providers/data/ant-ling.json +1 -1
- package/dist/providers/data/anthropic.json +1 -1
- package/dist/providers/data/azure-openai-responses.json +1 -1
- package/dist/providers/data/bai.json +1 -1
- package/dist/providers/data/baseten.json +1 -1
- package/dist/providers/data/cerebras.json +1 -1
- package/dist/providers/data/chatgpt-subscription.json +1 -1
- package/dist/providers/data/cloudflare-ai-gateway.json +1 -1
- package/dist/providers/data/cloudflare-workers-ai.json +1 -1
- package/dist/providers/data/deepseek.json +1 -1
- package/dist/providers/data/fireworks.json +1 -1
- package/dist/providers/data/github-copilot.json +1 -1
- package/dist/providers/data/google-vertex.json +1 -1
- package/dist/providers/data/google.json +1 -1
- package/dist/providers/data/groq.json +1 -1
- package/dist/providers/data/huggingface.json +1 -1
- package/dist/providers/data/meta.json +1 -0
- package/dist/providers/data/minimax-cn.json +1 -1
- package/dist/providers/data/minimax.json +1 -1
- package/dist/providers/data/mistral.json +1 -1
- package/dist/providers/data/moonshotai-cn.json +1 -1
- package/dist/providers/data/moonshotai.json +1 -1
- package/dist/providers/data/nvidia.json +1 -1
- package/dist/providers/data/openai.json +1 -1
- package/dist/providers/data/opencode-go.json +1 -1
- package/dist/providers/data/opencode.json +1 -1
- package/dist/providers/data/opengateway.json +1 -1
- package/dist/providers/data/openrouter.json +1 -1
- package/dist/providers/data/qwen-token-plan-cn.json +1 -1
- package/dist/providers/data/qwen-token-plan-individual.json +1 -1
- package/dist/providers/data/qwen-token-plan.json +1 -1
- package/dist/providers/data/radius.json +1 -0
- package/dist/providers/data/together.json +1 -1
- package/dist/providers/data/typesafe.json +1 -0
- package/dist/providers/data/venice.json +1 -1
- package/dist/providers/data/vercel-ai-gateway.json +1 -1
- package/dist/providers/data/xai.json +1 -1
- package/dist/providers/data/xiaomi-token-plan-ams.json +1 -1
- package/dist/providers/data/xiaomi-token-plan-cn.json +1 -1
- package/dist/providers/data/xiaomi-token-plan-sgp.json +1 -1
- package/dist/providers/data/xiaomi.json +1 -1
- package/dist/providers/data/zai-coding-cn.json +1 -1
- package/dist/providers/data/zai.json +1 -1
- package/dist/providers/deepseek.models.d.ts +4 -2
- package/dist/providers/deepseek.models.js +4 -2
- package/dist/providers/faux.d.ts +7 -2
- package/dist/providers/faux.js +31 -22
- package/dist/providers/fireworks.models.d.ts +4 -2
- package/dist/providers/fireworks.models.js +4 -2
- package/dist/providers/github-copilot.models.d.ts +4 -2
- package/dist/providers/github-copilot.models.js +4 -2
- package/dist/providers/google-vertex.models.d.ts +4 -2
- package/dist/providers/google-vertex.models.js +4 -2
- package/dist/providers/google.models.d.ts +4 -2
- package/dist/providers/google.models.js +4 -2
- package/dist/providers/groq.models.d.ts +4 -2
- package/dist/providers/groq.models.js +4 -2
- package/dist/providers/huggingface.models.d.ts +4 -2
- package/dist/providers/huggingface.models.js +4 -2
- package/dist/providers/images/register-builtins.d.ts +2 -2
- package/dist/providers/kimi-coding.models.d.ts +54 -7
- package/dist/providers/kimi-coding.models.js +34 -7
- package/dist/providers/meta.d.ts +3 -0
- package/dist/providers/meta.js +24 -0
- package/dist/providers/meta.models.d.ts +6 -0
- package/dist/providers/meta.models.js +8 -0
- package/dist/providers/minimax-cn.models.d.ts +4 -2
- package/dist/providers/minimax-cn.models.js +4 -2
- package/dist/providers/minimax.models.d.ts +4 -2
- package/dist/providers/minimax.models.js +4 -2
- package/dist/providers/mistral.models.d.ts +4 -2
- package/dist/providers/mistral.models.js +4 -2
- package/dist/providers/moonshotai-cn.models.d.ts +4 -2
- package/dist/providers/moonshotai-cn.models.js +4 -2
- package/dist/providers/moonshotai.models.d.ts +4 -2
- package/dist/providers/moonshotai.models.js +4 -2
- package/dist/providers/nvidia.models.d.ts +4 -2
- package/dist/providers/nvidia.models.js +4 -2
- package/dist/providers/openai.js +4 -2
- package/dist/providers/openai.models.d.ts +4 -2
- package/dist/providers/openai.models.js +4 -2
- package/dist/providers/opencode-go.models.d.ts +4 -2
- package/dist/providers/opencode-go.models.js +4 -2
- package/dist/providers/opencode.d.ts +3 -1
- package/dist/providers/opencode.js +5 -2
- package/dist/providers/opencode.models.d.ts +4 -2
- package/dist/providers/opencode.models.js +4 -2
- package/dist/providers/opengateway.models.d.ts +4 -2
- package/dist/providers/opengateway.models.js +4 -2
- package/dist/providers/openrouter.js +11 -2
- package/dist/providers/openrouter.models.d.ts +4 -2
- package/dist/providers/openrouter.models.js +4 -2
- package/dist/providers/qwen-token-plan-cn.models.d.ts +4 -2
- package/dist/providers/qwen-token-plan-cn.models.js +4 -2
- package/dist/providers/qwen-token-plan-individual.models.d.ts +4 -2
- package/dist/providers/qwen-token-plan-individual.models.js +4 -2
- package/dist/providers/qwen-token-plan.models.d.ts +4 -2
- package/dist/providers/qwen-token-plan.models.js +4 -2
- package/dist/providers/radius.js +19 -5
- package/dist/providers/radius.models.d.ts +6 -0
- package/dist/providers/radius.models.js +8 -0
- package/dist/providers/together.models.d.ts +4 -2
- package/dist/providers/together.models.js +4 -2
- package/dist/providers/typesafe.d.ts +3 -0
- package/dist/providers/typesafe.js +18 -0
- package/dist/providers/typesafe.models.d.ts +6 -0
- package/dist/providers/typesafe.models.js +8 -0
- package/dist/providers/venice.models.d.ts +4 -2
- package/dist/providers/venice.models.js +4 -2
- package/dist/providers/vercel-ai-gateway.js +5 -2
- package/dist/providers/vercel-ai-gateway.models.d.ts +4 -2
- package/dist/providers/vercel-ai-gateway.models.js +4 -2
- package/dist/providers/xai.models.d.ts +4 -2
- package/dist/providers/xai.models.js +4 -2
- package/dist/providers/xiaomi-token-plan-ams.models.d.ts +4 -2
- package/dist/providers/xiaomi-token-plan-ams.models.js +4 -2
- package/dist/providers/xiaomi-token-plan-cn.models.d.ts +4 -2
- package/dist/providers/xiaomi-token-plan-cn.models.js +4 -2
- package/dist/providers/xiaomi-token-plan-sgp.models.d.ts +4 -2
- package/dist/providers/xiaomi-token-plan-sgp.models.js +4 -2
- package/dist/providers/xiaomi.models.d.ts +4 -2
- package/dist/providers/xiaomi.models.js +4 -2
- package/dist/providers/zai-coding-cn.models.d.ts +4 -2
- package/dist/providers/zai-coding-cn.models.js +4 -2
- package/dist/providers/zai.models.d.ts +4 -2
- package/dist/providers/zai.models.js +4 -2
- package/dist/tool-call-middleware/context-transformer.d.ts +8 -5
- package/dist/tool-call-middleware/context-transformer.js +44 -15
- package/dist/types.d.ts +262 -28
- package/dist/utils/diagnostics.d.ts +3 -2
- package/dist/utils/estimate.d.ts +2 -2
- package/dist/utils/estimate.js +22 -30
- package/dist/utils/headers.d.ts +1 -1
- package/dist/utils/headers.js +10 -8
- package/dist/utils/model-operations.d.ts +11 -0
- package/dist/utils/model-operations.js +47 -0
- package/dist/utils/models-error.d.ts +8 -0
- package/dist/utils/models-error.js +18 -0
- package/dist/utils/overflow.d.ts +1 -0
- package/dist/utils/overflow.js +11 -5
- package/dist/utils/prompt-cache-ttl.js +10 -3
- package/dist/utils/retry.js +9 -0
- package/dist/utils/text.d.ts +9 -1
- package/dist/utils/text.js +26 -0
- package/dist/utils/transcript.d.ts +85 -0
- package/dist/utils/transcript.js +205 -0
- package/package.json +3 -4
- package/dist/image-models.generated.d.ts +0 -925
- package/dist/image-models.generated.js +0 -927
- package/dist/images-models.d.ts +0 -95
- package/dist/images-models.js +0 -141
- package/dist/providers/openai-images.d.ts +0 -3
- package/dist/providers/openai-images.js +0 -16
- package/dist/providers/openrouter-images.d.ts +0 -3
- package/dist/providers/openrouter-images.js +0 -22
- /package/dist/{auth/oauth → utils}/oauth-page.d.ts +0 -0
- /package/dist/{auth/oauth → utils}/oauth-page.js +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Api,
|
|
1
|
+
import type { Api, Model, TranscriptContext } from "../types.ts";
|
|
2
2
|
export declare const CONTEXT_SAFETY_TOKENS = 4096;
|
|
3
3
|
/** Tokens always left for the answer when a thinking budget shares the response ceiling. */
|
|
4
4
|
export declare const MIN_ANSWER_TOKENS = 1024;
|
|
@@ -16,5 +16,5 @@ export declare class ContextWindowExhaustedError extends Error {
|
|
|
16
16
|
* {@link MIN_ANSWER_TOKENS}: such a request can only return a truncated tool call or an
|
|
17
17
|
* empty "length" stop while still billing the whole prompt.
|
|
18
18
|
*/
|
|
19
|
-
export declare function clampMaxTokensToContext(model: Model<Api>, context:
|
|
19
|
+
export declare function clampMaxTokensToContext(model: Model<Api>, context: TranscriptContext, maxTokens: number): number;
|
|
20
20
|
//# sourceMappingURL=context-room.d.ts.map
|
package/dist/api/cursor-agent.js
CHANGED
|
@@ -28,6 +28,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream.js";
|
|
|
28
28
|
import { providerHeadersToRecord } from "../utils/headers.js";
|
|
29
29
|
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse.js";
|
|
30
30
|
import { sanitizeSurrogates } from "../utils/sanitize-unicode.js";
|
|
31
|
+
import { getCurrentSystemPrompt, getCurrentTools } from "../utils/transcript.js";
|
|
31
32
|
import { deterministicUuid } from "./cursor-agent/deterministic-id.js";
|
|
32
33
|
import { armExecHeartbeat } from "./cursor-agent/exec-lifecycle.js";
|
|
33
34
|
import { buildMcpStateResult, buildNeutralHookResult, buildPiBashError, buildPiBashResult, buildPiEditError, buildPiEditRejected, buildPiEditResult, buildPiFindError, buildPiFindResult, buildPiGrepError, buildPiGrepResult, buildPiLsError, buildPiLsResult, buildPiReadError, buildPiReadResult, buildPiWriteError, buildPiWriteRejected, buildPiWriteResult, } from "./cursor-agent/exec-modern.js";
|
|
@@ -474,6 +475,7 @@ export function mapH2TransportError(error, baseUrl) {
|
|
|
474
475
|
}
|
|
475
476
|
export const stream = (model, context, options) => {
|
|
476
477
|
const stream = new AssistantMessageEventStream();
|
|
478
|
+
const requestView = toCursorRequestView(context);
|
|
477
479
|
(async () => {
|
|
478
480
|
const output = {
|
|
479
481
|
role: "assistant",
|
|
@@ -591,7 +593,7 @@ export const stream = (model, context, options) => {
|
|
|
591
593
|
attemptPinnedStore = blobStore;
|
|
592
594
|
blobStore.beginRequestPins();
|
|
593
595
|
const cachedState = conversationStateCache.get(conversationId);
|
|
594
|
-
const { requestBytes, conversationState, requestedModel, modelDetails } = await buildGrpcRequest(model,
|
|
596
|
+
const { requestBytes, conversationState, requestedModel, modelDetails } = await buildGrpcRequest(model, requestView, options, {
|
|
595
597
|
conversationId,
|
|
596
598
|
blobStore,
|
|
597
599
|
conversationState: cachedState,
|
|
@@ -605,7 +607,7 @@ export const stream = (model, context, options) => {
|
|
|
605
607
|
// This request's working set is pinned now, so the process ceiling can
|
|
606
608
|
// reclaim from cold conversations without touching what it needs.
|
|
607
609
|
enforceConversationTotalBlobLimit();
|
|
608
|
-
const requestContextTools = buildMcpToolDefinitions(
|
|
610
|
+
const requestContextTools = buildMcpToolDefinitions(requestView.tools);
|
|
609
611
|
const baseUrl = model.baseUrl || CURSOR_API_URL;
|
|
610
612
|
const requestPath = "/agent.v1.AgentService/Run";
|
|
611
613
|
// Caller headers are additive, and are spread FIRST so the protocol
|
|
@@ -2960,7 +2962,7 @@ export function processInteractionUpdate(update, output, stream, state, usageSta
|
|
|
2960
2962
|
type: "toolCall",
|
|
2961
2963
|
id: scmCall?.args?.toolCallId || update.message.value.callId || randomUUID(),
|
|
2962
2964
|
name: "connect_scm",
|
|
2963
|
-
arguments: repository ? { owner: repository.owner, repo: repository.repo } : {},
|
|
2965
|
+
arguments: (repository ? { owner: repository.owner, repo: repository.repo } : {}),
|
|
2964
2966
|
[kStreamingBlockIndex]: output.content.length,
|
|
2965
2967
|
[kStreamingBlockKind]: "connect-scm",
|
|
2966
2968
|
[kStreamingEnvelopeId]: update.message.value.callId || undefined,
|
|
@@ -3656,11 +3658,18 @@ function extractImages(content) {
|
|
|
3656
3658
|
},
|
|
3657
3659
|
}));
|
|
3658
3660
|
}
|
|
3659
|
-
|
|
3661
|
+
function toCursorRequestView(context) {
|
|
3662
|
+
return {
|
|
3663
|
+
systemPrompt: getCurrentSystemPrompt(context.messages),
|
|
3664
|
+
messages: context.messages.filter((message) => message.role !== "system"),
|
|
3665
|
+
tools: getCurrentTools(context.messages),
|
|
3666
|
+
};
|
|
3667
|
+
}
|
|
3668
|
+
async function buildGrpcRequest(model, request, options, state) {
|
|
3660
3669
|
const blobStore = state.blobStore;
|
|
3661
|
-
const systemPromptIds = buildCursorSystemPromptJsons(
|
|
3662
|
-
const activeUserMessageIndex =
|
|
3663
|
-
const activeMessage =
|
|
3670
|
+
const systemPromptIds = buildCursorSystemPromptJsons(request.systemPrompt, model.id).map((json) => storeCursorBlob(blobStore, new TextEncoder().encode(json)));
|
|
3671
|
+
const activeUserMessageIndex = request.messages.length - 1;
|
|
3672
|
+
const activeMessage = request.messages[activeUserMessageIndex];
|
|
3664
3673
|
const activeUserMessage = activeMessage?.role === "user" ? activeMessage : undefined;
|
|
3665
3674
|
let userContent;
|
|
3666
3675
|
let userText = "";
|
|
@@ -3691,11 +3700,11 @@ async function buildGrpcRequest(model, context, options, state) {
|
|
|
3691
3700
|
// Build conversation turns from prior messages, excluding only the active
|
|
3692
3701
|
// user message when the request is sending one. Resume actions must
|
|
3693
3702
|
// preserve trailing tool results.
|
|
3694
|
-
const turns = buildConversationTurns(
|
|
3703
|
+
const turns = buildConversationTurns(request.messages, blobStore, activeUserMessage ? activeUserMessageIndex : -1);
|
|
3695
3704
|
// Cursor's server uses `rootPromptMessagesJson` (not `turns[]`) to build
|
|
3696
3705
|
// the actual model prompt; without it multi-turn conversations lose prior
|
|
3697
3706
|
// context.
|
|
3698
|
-
const rootPromptMessagesJson = buildRootPromptMessagesJson(
|
|
3707
|
+
const rootPromptMessagesJson = buildRootPromptMessagesJson(request.messages, systemPromptIds, blobStore, activeUserMessage ? activeUserMessageIndex : -1);
|
|
3699
3708
|
// Preserve cached non-history state fields (todos, file states, summaries)
|
|
3700
3709
|
// when the system prompt is unchanged; otherwise start fresh.
|
|
3701
3710
|
const cachedPromptHead = state.conversationState?.rootPromptMessagesJson?.slice(0, systemPromptIds.length) ?? [];
|
|
@@ -3752,7 +3761,7 @@ async function buildGrpcRequest(model, context, options, state) {
|
|
|
3752
3761
|
const requestBytes = toBinary(AgentClientMessageSchema, clientMessage);
|
|
3753
3762
|
log("info", "builtRunRequest", {
|
|
3754
3763
|
bytes: requestBytes.length,
|
|
3755
|
-
tools:
|
|
3764
|
+
tools: request.tools.length,
|
|
3756
3765
|
});
|
|
3757
3766
|
return { requestBytes, blobStore, conversationState, requestedModel, modelDetails };
|
|
3758
3767
|
}
|
|
@@ -2,15 +2,15 @@
|
|
|
2
2
|
* Builds the Cascade `GetChatMessage` request from senpi's provider-neutral
|
|
3
3
|
* context.
|
|
4
4
|
*
|
|
5
|
-
* Cascade has no system role: the system prompt
|
|
6
|
-
* `prompt` field, and history is a
|
|
7
|
-
* whose `source` carries the role.
|
|
8
|
-
*
|
|
9
|
-
* retried turn re-sends the same ids
|
|
10
|
-
* transcript, while a native Devin turn is
|
|
11
|
-
* minted for it.
|
|
5
|
+
* Cascade has no system role: the system prompt (replayed from the transcript's
|
|
6
|
+
* system messages) travels in the top-level `prompt` field, and history is a
|
|
7
|
+
* flat list of `ChatMessagePrompt` entries whose `source` carries the role.
|
|
8
|
+
* Message ids must be UUID-shaped; they are derived deterministically from the
|
|
9
|
+
* conversation id and the entry index so a retried turn re-sends the same ids
|
|
10
|
+
* instead of forking the server-side transcript, while a native Devin turn is
|
|
11
|
+
* replayed under the id the server minted for it.
|
|
12
12
|
*/
|
|
13
|
-
import type {
|
|
13
|
+
import type { Message, Model, TranscriptContext } from "../../types.ts";
|
|
14
14
|
import { type ChatMessagePrompt, type GetChatMessageRequest } from "./gen/cascade_pb.ts";
|
|
15
15
|
/** Cascade's own stop vocabulary; the server echoes these as STOP_PATTERN. */
|
|
16
16
|
export declare const DEVIN_DEFAULT_STOP_PATTERNS: readonly ["<|user|>", "<|bot|>", "<|context_request|>", "<|endoftext|>", "<|end_of_turn|>"];
|
|
@@ -20,7 +20,7 @@ export interface DevinModelAssignment {
|
|
|
20
20
|
}
|
|
21
21
|
export interface DevinChatRequestInput {
|
|
22
22
|
model: Model<"devin-agent">;
|
|
23
|
-
context:
|
|
23
|
+
context: TranscriptContext;
|
|
24
24
|
apiKey: string | undefined;
|
|
25
25
|
userJwt?: string;
|
|
26
26
|
cascadeId: string;
|
|
@@ -2,15 +2,16 @@
|
|
|
2
2
|
* Builds the Cascade `GetChatMessage` request from senpi's provider-neutral
|
|
3
3
|
* context.
|
|
4
4
|
*
|
|
5
|
-
* Cascade has no system role: the system prompt
|
|
6
|
-
* `prompt` field, and history is a
|
|
7
|
-
* whose `source` carries the role.
|
|
8
|
-
*
|
|
9
|
-
* retried turn re-sends the same ids
|
|
10
|
-
* transcript, while a native Devin turn is
|
|
11
|
-
* minted for it.
|
|
5
|
+
* Cascade has no system role: the system prompt (replayed from the transcript's
|
|
6
|
+
* system messages) travels in the top-level `prompt` field, and history is a
|
|
7
|
+
* flat list of `ChatMessagePrompt` entries whose `source` carries the role.
|
|
8
|
+
* Message ids must be UUID-shaped; they are derived deterministically from the
|
|
9
|
+
* conversation id and the entry index so a retried turn re-sends the same ids
|
|
10
|
+
* instead of forking the server-side transcript, while a native Devin turn is
|
|
11
|
+
* replayed under the id the server minted for it.
|
|
12
12
|
*/
|
|
13
13
|
import { create } from "@bufbuild/protobuf";
|
|
14
|
+
import { getCurrentSystemPrompt, getCurrentTools } from "../../utils/transcript.js";
|
|
14
15
|
import { deterministicUuid } from "../cursor-agent/deterministic-id.js";
|
|
15
16
|
import { CacheControlType, ChatMessagePromptSchema, ChatMessageRequestType, ChatMessageSource, ChatToolCallSchema, ChatToolDefinitionSchema, CompletionConfigurationSchema, ConversationalPlannerMode, GetChatMessageRequestSchema, ImageDataSchema, PromptCacheOptionsSchema, } from "./gen/cascade_pb.js";
|
|
16
17
|
import { devinCliMetadata } from "./metadata.js";
|
|
@@ -29,17 +30,20 @@ const MIN_TEMPERATURE = 0.0001;
|
|
|
29
30
|
export function buildDevinChatRequest(input) {
|
|
30
31
|
const temperature = Math.max(input.temperature ?? DEFAULT_TEMPERATURE, MIN_TEMPERATURE);
|
|
31
32
|
const stopPatterns = [...DEVIN_DEFAULT_STOP_PATTERNS, ...(input.stopSequences ?? [])];
|
|
33
|
+
const { messages } = input.context;
|
|
34
|
+
// History ids derive from the entry index, so system messages leave the list before mapping.
|
|
35
|
+
const history = messages.filter((message) => message.role !== "system");
|
|
32
36
|
return create(GetChatMessageRequestSchema, {
|
|
33
37
|
metadata: devinCliMetadata(input.apiKey, input.userJwt ?? ""),
|
|
34
|
-
prompt:
|
|
35
|
-
chatMessagePrompts: mapHistory(
|
|
38
|
+
prompt: getCurrentSystemPrompt(messages),
|
|
39
|
+
chatMessagePrompts: mapHistory(history, input.cascadeId, input.model),
|
|
36
40
|
requestType: ChatMessageRequestType.CASCADE,
|
|
37
41
|
plannerMode: ConversationalPlannerMode.DEFAULT,
|
|
38
42
|
chatModelUid: input.assignment?.modelUid ?? input.model.upstreamModelId ?? input.model.id,
|
|
39
43
|
...(input.assignment ? { modelAssignmentJwt: input.assignment.assignmentJwt } : {}),
|
|
40
44
|
cascadeId: input.cascadeId,
|
|
41
45
|
executionId: crypto.randomUUID(),
|
|
42
|
-
tools: (
|
|
46
|
+
tools: getCurrentTools(messages).map(toolDefinition),
|
|
43
47
|
toolChoice: { choice: { case: "optionName", value: "auto" } },
|
|
44
48
|
systemPromptCacheOptions: create(PromptCacheOptionsSchema, { type: CacheControlType.EPHEMERAL }),
|
|
45
49
|
disableParallelToolCalls: input.model.compat?.supportsParallelToolCalls !== true,
|
|
@@ -1,23 +1,19 @@
|
|
|
1
|
-
import { GoogleGenAI,
|
|
1
|
+
import { GoogleGenAI, } from "@google/genai";
|
|
2
2
|
import { calculateCost, clampThinkingLevel } from "../models.js";
|
|
3
3
|
import { formatProviderError, normalizeProviderError } from "../utils/error-body.js";
|
|
4
4
|
import { AssistantMessageEventStream } from "../utils/event-stream.js";
|
|
5
5
|
import { providerHeadersToRecord } from "../utils/headers.js";
|
|
6
6
|
import { getPiUserAgent } from "../utils/pi-user-agent.js";
|
|
7
7
|
import { sanitizeSurrogates } from "../utils/sanitize-unicode.js";
|
|
8
|
-
import {
|
|
8
|
+
import { getSystemMessageText } from "../utils/text.js";
|
|
9
|
+
import { collapseSystemMessages, getCurrentTools, getInitialSystemMessage } from "../utils/transcript.js";
|
|
10
|
+
import { convertMessages, convertTools, getDisabledGoogleThinkingConfig, isThinkingPart, mapStopReason, resolveGoogleFunctionCallingMode, resolveGoogleThinkingLevel, retainThoughtSignature, retryGoogleRequest, supportsGoogleStrictToolSampling, toGoogleSdkThinkingLevel, toGoogleThinkingLevel, toProviderNativeContent, usesGoogleThinkingLevel, } from "./google-shared.js";
|
|
9
11
|
import { applyExtraBody, buildBaseOptions, GOOGLE_RESERVED_BODY_KEYS } from "./simple-options.js";
|
|
10
|
-
const THINKING_LEVEL_MAP = {
|
|
11
|
-
THINKING_LEVEL_UNSPECIFIED: GoogleGenAIThinkingLevel.THINKING_LEVEL_UNSPECIFIED,
|
|
12
|
-
MINIMAL: GoogleGenAIThinkingLevel.MINIMAL,
|
|
13
|
-
LOW: GoogleGenAIThinkingLevel.LOW,
|
|
14
|
-
MEDIUM: GoogleGenAIThinkingLevel.MEDIUM,
|
|
15
|
-
HIGH: GoogleGenAIThinkingLevel.HIGH,
|
|
16
|
-
};
|
|
17
12
|
// Counter for generating unique tool call IDs
|
|
18
13
|
let toolCallCounter = 0;
|
|
19
14
|
export const stream = (model, context, options) => {
|
|
20
15
|
const stream = new AssistantMessageEventStream();
|
|
16
|
+
const normalizedContext = collapseSystemMessages(context);
|
|
21
17
|
(async () => {
|
|
22
18
|
const output = {
|
|
23
19
|
role: "assistant",
|
|
@@ -45,7 +41,7 @@ export const stream = (model, context, options) => {
|
|
|
45
41
|
throw new Error(`No API key for provider: ${model.provider}`);
|
|
46
42
|
}
|
|
47
43
|
const client = createClient(model, apiKey, providerHeadersToRecord(options?.headers));
|
|
48
|
-
let params = buildParams(model,
|
|
44
|
+
let params = buildParams(model, normalizedContext, options);
|
|
49
45
|
const nextParams = await options?.onPayload?.(params, model);
|
|
50
46
|
if (nextParams !== undefined) {
|
|
51
47
|
params = nextParams;
|
|
@@ -58,6 +54,7 @@ export const stream = (model, context, options) => {
|
|
|
58
54
|
const blocks = output.content;
|
|
59
55
|
const blockIndex = () => blocks.length - 1;
|
|
60
56
|
for await (const chunk of googleStream) {
|
|
57
|
+
await options?.onProviderStreamEvent?.(chunk, model);
|
|
61
58
|
// @google/genai documents GenerateContentResponse.responseId as an output-only field
|
|
62
59
|
// used to identify each response. Keep the first non-empty one from the stream.
|
|
63
60
|
output.responseId ||= chunk.responseId;
|
|
@@ -281,13 +278,12 @@ export const streamSimple = (model, context, options) => {
|
|
|
281
278
|
return stream(model, context, { ...base, thinking: { enabled: false } });
|
|
282
279
|
}
|
|
283
280
|
const resolvedLevel = resolveGoogleThinkingLevel(model, clampedReasoning);
|
|
284
|
-
|
|
285
|
-
if (isGemini3ProModel(googleModel) || isGemini3FlashModel(googleModel) || isGemma4Model(googleModel)) {
|
|
281
|
+
if (usesGoogleThinkingLevel(model)) {
|
|
286
282
|
return stream(model, context, {
|
|
287
283
|
...base,
|
|
288
284
|
thinking: {
|
|
289
285
|
enabled: true,
|
|
290
|
-
level:
|
|
286
|
+
level: toGoogleThinkingLevel(resolvedLevel),
|
|
291
287
|
},
|
|
292
288
|
});
|
|
293
289
|
}
|
|
@@ -295,7 +291,7 @@ export const streamSimple = (model, context, options) => {
|
|
|
295
291
|
...base,
|
|
296
292
|
thinking: {
|
|
297
293
|
enabled: true,
|
|
298
|
-
budgetTokens: getGoogleBudget(
|
|
294
|
+
budgetTokens: getGoogleBudget(model, resolvedLevel, options.thinkingBudgets),
|
|
299
295
|
},
|
|
300
296
|
});
|
|
301
297
|
};
|
|
@@ -316,6 +312,8 @@ function createClient(model, apiKey, optionsHeaders) {
|
|
|
316
312
|
}
|
|
317
313
|
function buildParams(model, context, options = {}) {
|
|
318
314
|
const contents = convertMessages(model, context, { preserveThinking: options.thinking?.enabled === true });
|
|
315
|
+
const initialSystemMessage = getInitialSystemMessage(context.messages);
|
|
316
|
+
const currentTools = getCurrentTools(context.messages);
|
|
319
317
|
const generationConfig = {};
|
|
320
318
|
if (options.temperature !== undefined) {
|
|
321
319
|
generationConfig.temperature = options.temperature;
|
|
@@ -324,15 +322,15 @@ function buildParams(model, context, options = {}) {
|
|
|
324
322
|
generationConfig.maxOutputTokens = options.maxTokens;
|
|
325
323
|
}
|
|
326
324
|
const supportsStrictMode = supportsGoogleStrictToolSampling(model.id);
|
|
327
|
-
const functionCallingMode =
|
|
328
|
-
? resolveGoogleFunctionCallingMode(
|
|
325
|
+
const functionCallingMode = currentTools.length > 0
|
|
326
|
+
? resolveGoogleFunctionCallingMode(currentTools, options.toolChoice, supportsStrictMode)
|
|
329
327
|
: undefined;
|
|
328
|
+
const systemInstruction = initialSystemMessage ? getSystemMessageText(initialSystemMessage) : "";
|
|
330
329
|
const config = {
|
|
331
330
|
...(Object.keys(generationConfig).length > 0 && generationConfig),
|
|
332
|
-
...(
|
|
333
|
-
...(
|
|
334
|
-
|
|
335
|
-
tools: convertTools(context.tools, false, supportsStrictMode),
|
|
331
|
+
...(systemInstruction && { systemInstruction: sanitizeSurrogates(systemInstruction) }),
|
|
332
|
+
...(currentTools.length > 0 && {
|
|
333
|
+
tools: convertTools(currentTools, false, supportsStrictMode),
|
|
336
334
|
}),
|
|
337
335
|
...(functionCallingMode !== undefined && {
|
|
338
336
|
toolConfig: { functionCallingConfig: { mode: functionCallingMode } },
|
|
@@ -341,7 +339,7 @@ function buildParams(model, context, options = {}) {
|
|
|
341
339
|
if (options.thinking?.enabled && model.reasoning) {
|
|
342
340
|
const thinkingConfig = { includeThoughts: true };
|
|
343
341
|
if (options.thinking.level !== undefined) {
|
|
344
|
-
thinkingConfig.thinkingLevel =
|
|
342
|
+
thinkingConfig.thinkingLevel = toGoogleSdkThinkingLevel(options.thinking.level);
|
|
345
343
|
}
|
|
346
344
|
else if (options.thinking.budgetTokens !== undefined) {
|
|
347
345
|
thinkingConfig.thinkingBudget = options.thinking.budgetTokens;
|
|
@@ -349,7 +347,7 @@ function buildParams(model, context, options = {}) {
|
|
|
349
347
|
config.thinkingConfig = thinkingConfig;
|
|
350
348
|
}
|
|
351
349
|
else if (model.reasoning && options.thinking && !options.thinking.enabled) {
|
|
352
|
-
config.thinkingConfig =
|
|
350
|
+
config.thinkingConfig = getDisabledGoogleThinkingConfig(model);
|
|
353
351
|
}
|
|
354
352
|
if (options.signal) {
|
|
355
353
|
if (options.signal.aborted) {
|
|
@@ -365,64 +363,6 @@ function buildParams(model, context, options = {}) {
|
|
|
365
363
|
};
|
|
366
364
|
return params;
|
|
367
365
|
}
|
|
368
|
-
function isGemma4Model(model) {
|
|
369
|
-
return /gemma-?4/.test(model.id.toLowerCase());
|
|
370
|
-
}
|
|
371
|
-
function isGemini3ProModel(model) {
|
|
372
|
-
return /gemini-3(?:\.\d+)?-pro/.test(model.id.toLowerCase());
|
|
373
|
-
}
|
|
374
|
-
function isGemini3FlashModel(model) {
|
|
375
|
-
const id = model.id.toLowerCase();
|
|
376
|
-
return /gemini-3(?:\.\d+)?-flash/.test(id) || id === "gemini-flash-latest" || id === "gemini-flash-lite-latest";
|
|
377
|
-
}
|
|
378
|
-
function getDisabledThinkingConfig(model) {
|
|
379
|
-
// Google docs: Gemini 3.1 Pro cannot disable thinking, and Gemini 3 Flash / Flash-Lite
|
|
380
|
-
// do not support full thinking-off either. For Gemini 3 models, use the lowest supported
|
|
381
|
-
// thinkingLevel without includeThoughts so hidden thinking remains invisible to pi.
|
|
382
|
-
if (isGemini3ProModel(model)) {
|
|
383
|
-
return { thinkingLevel: GoogleGenAIThinkingLevel.LOW };
|
|
384
|
-
}
|
|
385
|
-
if (isGemini3FlashModel(model)) {
|
|
386
|
-
return { thinkingLevel: GoogleGenAIThinkingLevel.MINIMAL };
|
|
387
|
-
}
|
|
388
|
-
if (isGemma4Model(model)) {
|
|
389
|
-
return { thinkingLevel: GoogleGenAIThinkingLevel.MINIMAL };
|
|
390
|
-
}
|
|
391
|
-
// Gemini 2.x supports disabling via thinkingBudget = 0.
|
|
392
|
-
return { thinkingBudget: 0 };
|
|
393
|
-
}
|
|
394
|
-
function getThinkingLevel(effort, model) {
|
|
395
|
-
if (isGemini3ProModel(model)) {
|
|
396
|
-
switch (effort) {
|
|
397
|
-
case "minimal":
|
|
398
|
-
case "low":
|
|
399
|
-
return "LOW";
|
|
400
|
-
case "medium":
|
|
401
|
-
case "high":
|
|
402
|
-
return "HIGH";
|
|
403
|
-
}
|
|
404
|
-
}
|
|
405
|
-
if (isGemma4Model(model)) {
|
|
406
|
-
switch (effort) {
|
|
407
|
-
case "minimal":
|
|
408
|
-
case "low":
|
|
409
|
-
return "MINIMAL";
|
|
410
|
-
case "medium":
|
|
411
|
-
case "high":
|
|
412
|
-
return "HIGH";
|
|
413
|
-
}
|
|
414
|
-
}
|
|
415
|
-
switch (effort) {
|
|
416
|
-
case "minimal":
|
|
417
|
-
return "MINIMAL";
|
|
418
|
-
case "low":
|
|
419
|
-
return "LOW";
|
|
420
|
-
case "medium":
|
|
421
|
-
return "MEDIUM";
|
|
422
|
-
case "high":
|
|
423
|
-
return "HIGH";
|
|
424
|
-
}
|
|
425
|
-
}
|
|
426
366
|
function getGoogleBudget(model, level, customBudgets) {
|
|
427
367
|
if (customBudgets?.[level] !== undefined) {
|
|
428
368
|
return customBudgets[level];
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Shared utilities for Google Generative AI and Google Vertex providers.
|
|
3
3
|
*/
|
|
4
|
-
import { type Content, FinishReason, FunctionCallingConfigMode, type Part } from "@google/genai";
|
|
5
|
-
import type {
|
|
4
|
+
import { type Content, FinishReason, FunctionCallingConfigMode, ThinkingLevel as GoogleSdkThinkingLevel, type Part, type ThinkingConfig } from "@google/genai";
|
|
5
|
+
import type { Model, ProviderNativeContent, StopReason, StreamOptions, ThinkingLevel, Tool, TranscriptContext } from "../types.ts";
|
|
6
6
|
type GoogleApiType = "google-generative-ai" | "google-vertex";
|
|
7
7
|
/**
|
|
8
8
|
* Thinking level for Gemini 3 models.
|
|
@@ -11,7 +11,16 @@ type GoogleApiType = "google-generative-ai" | "google-vertex";
|
|
|
11
11
|
export type GoogleApiThinkingLevel = "THINKING_LEVEL_UNSPECIFIED" | "MINIMAL" | "LOW" | "MEDIUM" | "HIGH";
|
|
12
12
|
export type ResolvedGoogleThinkingLevel = Exclude<ThinkingLevel, "xhigh" | "max">;
|
|
13
13
|
/** Resolve a supported pi level or model-specific Google mapping to a standard Google level. */
|
|
14
|
-
export declare function resolveGoogleThinkingLevel<T extends GoogleApiType>(model: Model<T>, level:
|
|
14
|
+
export declare function resolveGoogleThinkingLevel<T extends GoogleApiType>(model: Model<T>, level: ThinkingLevel): ResolvedGoogleThinkingLevel;
|
|
15
|
+
/**
|
|
16
|
+
* Whether this model uses Gemini's discrete `thinkingLevel` control instead of
|
|
17
|
+
* the token-based `thinkingBudget` control. Supported levels come from the
|
|
18
|
+
* model's `thinkingLevelMap`; this only selects the Google wire format.
|
|
19
|
+
*/
|
|
20
|
+
export declare function usesGoogleThinkingLevel<T extends GoogleApiType>(model: Model<T>): boolean;
|
|
21
|
+
export declare function toGoogleThinkingLevel(level: ResolvedGoogleThinkingLevel): GoogleApiThinkingLevel;
|
|
22
|
+
export declare function toGoogleSdkThinkingLevel(level: GoogleApiThinkingLevel): GoogleSdkThinkingLevel;
|
|
23
|
+
export declare function getDisabledGoogleThinkingConfig<T extends GoogleApiType>(model: Model<T>): ThinkingConfig;
|
|
15
24
|
/**
|
|
16
25
|
* Determines whether a streamed Gemini `Part` should be treated as "thinking".
|
|
17
26
|
*
|
|
@@ -45,7 +54,7 @@ export declare function requiresToolCallId(modelId: string): boolean;
|
|
|
45
54
|
/**
|
|
46
55
|
* Convert internal messages to Gemini Content[] format.
|
|
47
56
|
*/
|
|
48
|
-
export declare function convertMessages<T extends GoogleApiType>(model: Model<T>, context:
|
|
57
|
+
export declare function convertMessages<T extends GoogleApiType>(model: Model<T>, context: TranscriptContext, options?: {
|
|
49
58
|
preserveThinking?: boolean;
|
|
50
59
|
}): Content[];
|
|
51
60
|
export declare function toProviderNativeContent(part: Part): ProviderNativeContent;
|
|
@@ -1,16 +1,23 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Shared utilities for Google Generative AI and Google Vertex providers.
|
|
3
3
|
*/
|
|
4
|
-
import { FinishReason, FunctionCallingConfigMode } from "@google/genai";
|
|
4
|
+
import { FinishReason, FunctionCallingConfigMode, ThinkingLevel as GoogleSdkThinkingLevel, } from "@google/genai";
|
|
5
|
+
import { clampThinkingLevel } from "../models.js";
|
|
5
6
|
import { retryProviderRequest } from "../utils/provider-retry.js";
|
|
6
7
|
import { sanitizeSurrogates } from "../utils/sanitize-unicode.js";
|
|
7
8
|
import { normalizeToolCallId } from "../utils/tool-call-id.js";
|
|
9
|
+
import { collapseSystemMessages, withoutInitialSystemMessage } from "../utils/transcript.js";
|
|
8
10
|
import { getJsonSchemaToolParameters, resolveJsonSchemaStrictSampling } from "./constrained-sampling.js";
|
|
9
11
|
import { transformMessages } from "./transform-messages.js";
|
|
12
|
+
const GOOGLE_SDK_THINKING_LEVEL_MAP = {
|
|
13
|
+
THINKING_LEVEL_UNSPECIFIED: GoogleSdkThinkingLevel.THINKING_LEVEL_UNSPECIFIED,
|
|
14
|
+
MINIMAL: GoogleSdkThinkingLevel.MINIMAL,
|
|
15
|
+
LOW: GoogleSdkThinkingLevel.LOW,
|
|
16
|
+
MEDIUM: GoogleSdkThinkingLevel.MEDIUM,
|
|
17
|
+
HIGH: GoogleSdkThinkingLevel.HIGH,
|
|
18
|
+
};
|
|
10
19
|
/** Resolve a supported pi level or model-specific Google mapping to a standard Google level. */
|
|
11
20
|
export function resolveGoogleThinkingLevel(model, level) {
|
|
12
|
-
if (level === "off")
|
|
13
|
-
return "high";
|
|
14
21
|
const mapped = model.thinkingLevelMap?.[level];
|
|
15
22
|
const resolvedLevel = typeof mapped === "string" ? mapped.toLowerCase() : level;
|
|
16
23
|
switch (resolvedLevel) {
|
|
@@ -23,6 +30,47 @@ export function resolveGoogleThinkingLevel(model, level) {
|
|
|
23
30
|
throw new Error(`Unsupported Google thinking level mapping for ${model.provider}/${model.id}: ${level} -> ${String(mapped)}`);
|
|
24
31
|
}
|
|
25
32
|
}
|
|
33
|
+
/**
|
|
34
|
+
* Whether this model uses Gemini's discrete `thinkingLevel` control instead of
|
|
35
|
+
* the token-based `thinkingBudget` control. Supported levels come from the
|
|
36
|
+
* model's `thinkingLevelMap`; this only selects the Google wire format.
|
|
37
|
+
*/
|
|
38
|
+
export function usesGoogleThinkingLevel(model) {
|
|
39
|
+
const id = model.id.toLowerCase();
|
|
40
|
+
return (
|
|
41
|
+
// Match Gemini 3 Pro/Flash IDs with or without a minor version, such as
|
|
42
|
+
// gemini-3-flash-preview, gemini-3.1-pro-preview, and gemini-3.8-flash.
|
|
43
|
+
/gemini-3(?:\.\d+)?-(?:pro|flash)/.test(id) ||
|
|
44
|
+
id === "gemini-flash-latest" ||
|
|
45
|
+
id === "gemini-flash-lite-latest" ||
|
|
46
|
+
// Match both hosted Gemma 4 naming forms: gemma-4-* and gemma4-*.
|
|
47
|
+
/gemma-?4/.test(id));
|
|
48
|
+
}
|
|
49
|
+
export function toGoogleThinkingLevel(level) {
|
|
50
|
+
switch (level) {
|
|
51
|
+
case "minimal":
|
|
52
|
+
return "MINIMAL";
|
|
53
|
+
case "low":
|
|
54
|
+
return "LOW";
|
|
55
|
+
case "medium":
|
|
56
|
+
return "MEDIUM";
|
|
57
|
+
case "high":
|
|
58
|
+
return "HIGH";
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
export function toGoogleSdkThinkingLevel(level) {
|
|
62
|
+
return GOOGLE_SDK_THINKING_LEVEL_MAP[level];
|
|
63
|
+
}
|
|
64
|
+
export function getDisabledGoogleThinkingConfig(model) {
|
|
65
|
+
if (!usesGoogleThinkingLevel(model))
|
|
66
|
+
return { thinkingBudget: 0 };
|
|
67
|
+
const fallback = clampThinkingLevel(model, "off");
|
|
68
|
+
if (fallback === "off")
|
|
69
|
+
return { thinkingBudget: 0 };
|
|
70
|
+
const resolvedLevel = resolveGoogleThinkingLevel(model, fallback);
|
|
71
|
+
const apiLevel = toGoogleThinkingLevel(resolvedLevel);
|
|
72
|
+
return { thinkingLevel: toGoogleSdkThinkingLevel(apiLevel) };
|
|
73
|
+
}
|
|
26
74
|
/**
|
|
27
75
|
* Determines whether a streamed Gemini `Part` should be treated as "thinking".
|
|
28
76
|
*
|
|
@@ -111,13 +159,15 @@ function appendContent(contents, content) {
|
|
|
111
159
|
* Convert internal messages to Gemini Content[] format.
|
|
112
160
|
*/
|
|
113
161
|
export function convertMessages(model, context, options = {}) {
|
|
162
|
+
// Gemini has no mid-conversation system messages; the leading prompt is sent as systemInstruction.
|
|
163
|
+
const conversation = withoutInitialSystemMessage(collapseSystemMessages(context).messages);
|
|
114
164
|
const contents = [];
|
|
115
165
|
const normalizeId = (id) => {
|
|
116
166
|
if (!requiresToolCallId(model.id))
|
|
117
167
|
return id;
|
|
118
168
|
return normalizeToolCallId(id);
|
|
119
169
|
};
|
|
120
|
-
const transformedMessages = transformMessages(
|
|
170
|
+
const transformedMessages = transformMessages(conversation, model, normalizeId, {
|
|
121
171
|
preserveThinking: options.preserveThinking,
|
|
122
172
|
});
|
|
123
173
|
for (const msg of transformedMessages) {
|