jeopi-ai 16.2.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +4347 -0
- package/README.md +1193 -0
- package/dist/types/api-registry.d.ts +30 -0
- package/dist/types/auth-broker/client.d.ts +73 -0
- package/dist/types/auth-broker/discover.d.ts +35 -0
- package/dist/types/auth-broker/index.d.ts +7 -0
- package/dist/types/auth-broker/refresher.d.ts +25 -0
- package/dist/types/auth-broker/remote-store.d.ts +102 -0
- package/dist/types/auth-broker/server.d.ts +43 -0
- package/dist/types/auth-broker/snapshot-cache.d.ts +17 -0
- package/dist/types/auth-broker/types.d.ts +107 -0
- package/dist/types/auth-broker/wire-schemas.d.ts +411 -0
- package/dist/types/auth-gateway/http.d.ts +39 -0
- package/dist/types/auth-gateway/index.d.ts +3 -0
- package/dist/types/auth-gateway/server.d.ts +36 -0
- package/dist/types/auth-gateway/types.d.ts +123 -0
- package/dist/types/auth-retry.d.ts +124 -0
- package/dist/types/auth-storage.d.ts +1026 -0
- package/dist/types/dialect/anthropic.d.ts +15 -0
- package/dist/types/dialect/catalog.d.ts +3 -0
- package/dist/types/dialect/coercion.d.ts +23 -0
- package/dist/types/dialect/deepseek.d.ts +14 -0
- package/dist/types/dialect/demotion.d.ts +23 -0
- package/dist/types/dialect/examples.d.ts +2 -0
- package/dist/types/dialect/factory.d.ts +3 -0
- package/dist/types/dialect/fenced-thinking.d.ts +53 -0
- package/dist/types/dialect/gemini.d.ts +17 -0
- package/dist/types/dialect/gemma.d.ts +15 -0
- package/dist/types/dialect/glm.d.ts +9 -0
- package/dist/types/dialect/harmony.d.ts +8 -0
- package/dist/types/dialect/hermes.d.ts +9 -0
- package/dist/types/dialect/history.d.ts +3 -0
- package/dist/types/dialect/index.d.ts +11 -0
- package/dist/types/dialect/inventory.d.ts +12 -0
- package/dist/types/dialect/kimi.d.ts +14 -0
- package/dist/types/dialect/minimax.d.ts +3 -0
- package/dist/types/dialect/owned-stream.d.ts +4 -0
- package/dist/types/dialect/qwen3.d.ts +9 -0
- package/dist/types/dialect/rendering.d.ts +45 -0
- package/dist/types/dialect/thinking.d.ts +6 -0
- package/dist/types/dialect/types.d.ts +69 -0
- package/dist/types/dialect/xml.d.ts +9 -0
- package/dist/types/error/abort.d.ts +14 -0
- package/dist/types/error/auth-classify.d.ts +16 -0
- package/dist/types/error/auth.d.ts +27 -0
- package/dist/types/error/aws.d.ts +23 -0
- package/dist/types/error/classes.d.ts +102 -0
- package/dist/types/error/finalize.d.ts +39 -0
- package/dist/types/error/flags.d.ts +79 -0
- package/dist/types/error/format.d.ts +20 -0
- package/dist/types/error/gateway.d.ts +20 -0
- package/dist/types/error/index.d.ts +13 -0
- package/dist/types/error/oauth.d.ts +43 -0
- package/dist/types/error/provider.d.ts +42 -0
- package/dist/types/error/rate-limit.d.ts +59 -0
- package/dist/types/error/retryable.d.ts +27 -0
- package/dist/types/error/validation.d.ts +32 -0
- package/dist/types/index.d.ts +49 -0
- package/dist/types/provider-details.d.ts +24 -0
- package/dist/types/providers/__tests__/google-auth.test.d.ts +1 -0
- package/dist/types/providers/__tests__/kimi-code-thinking.test.d.ts +1 -0
- package/dist/types/providers/__tests__/openai-codex-error.test.d.ts +1 -0
- package/dist/types/providers/amazon-bedrock.d.ts +39 -0
- package/dist/types/providers/anthropic-client.d.ts +94 -0
- package/dist/types/providers/anthropic-messages-server-schema.d.ts +497 -0
- package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
- package/dist/types/providers/anthropic-wire.d.ts +318 -0
- package/dist/types/providers/anthropic.d.ts +248 -0
- package/dist/types/providers/aws-credentials.d.ts +53 -0
- package/dist/types/providers/aws-eventstream.d.ts +39 -0
- package/dist/types/providers/aws-sigv4.d.ts +55 -0
- package/dist/types/providers/azure-openai-responses.d.ts +16 -0
- package/dist/types/providers/cursor.d.ts +91 -0
- package/dist/types/providers/devin.d.ts +12 -0
- package/dist/types/providers/error-message.d.ts +27 -0
- package/dist/types/providers/github-copilot-headers.d.ts +40 -0
- package/dist/types/providers/gitlab-duo-workflow.d.ts +254 -0
- package/dist/types/providers/gitlab-duo.d.ts +27 -0
- package/dist/types/providers/google-auth.d.ts +32 -0
- package/dist/types/providers/google-gemini-cli.d.ts +118 -0
- package/dist/types/providers/google-interactions.d.ts +65 -0
- package/dist/types/providers/google-shared.d.ts +203 -0
- package/dist/types/providers/google-types.d.ts +155 -0
- package/dist/types/providers/google-vertex.d.ts +7 -0
- package/dist/types/providers/google.d.ts +4 -0
- package/dist/types/providers/grammar.d.ts +1 -0
- package/dist/types/providers/kimi.d.ts +27 -0
- package/dist/types/providers/mock.d.ts +178 -0
- package/dist/types/providers/ollama.d.ts +7 -0
- package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
- package/dist/types/providers/openai-chat-server-schema.d.ts +695 -0
- package/dist/types/providers/openai-chat-server.d.ts +16 -0
- package/dist/types/providers/openai-chat-wire.d.ts +644 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +54 -0
- package/dist/types/providers/openai-codex/response-handler.d.ts +26 -0
- package/dist/types/providers/openai-codex-responses.d.ts +108 -0
- package/dist/types/providers/openai-completions.d.ts +45 -0
- package/dist/types/providers/openai-reasoning-fallback.d.ts +25 -0
- package/dist/types/providers/openai-responses-server-schema.d.ts +349 -0
- package/dist/types/providers/openai-responses-server.d.ts +17 -0
- package/dist/types/providers/openai-responses-wire.d.ts +6065 -0
- package/dist/types/providers/openai-responses.d.ts +126 -0
- package/dist/types/providers/openai-shared.d.ts +506 -0
- package/dist/types/providers/pi-native-client.d.ts +13 -0
- package/dist/types/providers/pi-native-server.d.ts +69 -0
- package/dist/types/providers/register-builtins.d.ts +32 -0
- package/dist/types/providers/synthetic.d.ts +26 -0
- package/dist/types/providers/transform-messages.d.ts +11 -0
- package/dist/types/providers/vision-guard.d.ts +20 -0
- package/dist/types/registry/aimlapi.d.ts +4 -0
- package/dist/types/registry/alibaba-coding-plan.d.ts +8 -0
- package/dist/types/registry/amazon-bedrock.d.ts +5 -0
- package/dist/types/registry/anthropic.d.ts +10 -0
- package/dist/types/registry/api-key-login.d.ts +42 -0
- package/dist/types/registry/api-key-validation.d.ts +43 -0
- package/dist/types/registry/azure.d.ts +4 -0
- package/dist/types/registry/cerebras.d.ts +7 -0
- package/dist/types/registry/cloudflare-ai-gateway.d.ts +13 -0
- package/dist/types/registry/coreweave.d.ts +7 -0
- package/dist/types/registry/cursor.d.ts +7 -0
- package/dist/types/registry/deepseek.d.ts +8 -0
- package/dist/types/registry/derived.d.ts +5 -0
- package/dist/types/registry/devin.d.ts +8 -0
- package/dist/types/registry/firepass.d.ts +16 -0
- package/dist/types/registry/fireworks.d.ts +7 -0
- package/dist/types/registry/github-copilot.d.ts +7 -0
- package/dist/types/registry/gitlab-duo-workflow.d.ts +10 -0
- package/dist/types/registry/gitlab-duo.d.ts +9 -0
- package/dist/types/registry/google-antigravity.d.ts +9 -0
- package/dist/types/registry/google-gemini-cli.d.ts +9 -0
- package/dist/types/registry/google-vertex.d.ts +5 -0
- package/dist/types/registry/google.d.ts +4 -0
- package/dist/types/registry/groq.d.ts +4 -0
- package/dist/types/registry/huggingface.d.ts +7 -0
- package/dist/types/registry/index.d.ts +4 -0
- package/dist/types/registry/kagi.d.ts +14 -0
- package/dist/types/registry/kilo.d.ts +7 -0
- package/dist/types/registry/kimi-code.d.ts +7 -0
- package/dist/types/registry/litellm.d.ts +13 -0
- package/dist/types/registry/llama-cpp.d.ts +8 -0
- package/dist/types/registry/lm-studio.d.ts +8 -0
- package/dist/types/registry/minimax-code-cn.d.ts +6 -0
- package/dist/types/registry/minimax-code.d.ts +6 -0
- package/dist/types/registry/minimax.d.ts +4 -0
- package/dist/types/registry/mistral.d.ts +4 -0
- package/dist/types/registry/moonshot.d.ts +7 -0
- package/dist/types/registry/nanogpt.d.ts +7 -0
- package/dist/types/registry/nvidia.d.ts +7 -0
- package/dist/types/registry/oauth/__tests__/xai-oauth.test.d.ts +1 -0
- package/dist/types/registry/oauth/anthropic.d.ts +23 -0
- package/dist/types/registry/oauth/callback-server.d.ts +72 -0
- package/dist/types/registry/oauth/cursor.d.ts +15 -0
- package/dist/types/registry/oauth/devin.d.ts +5 -0
- package/dist/types/registry/oauth/github-copilot.d.ts +30 -0
- package/dist/types/registry/oauth/gitlab-duo-workflow.d.ts +6 -0
- package/dist/types/registry/oauth/gitlab-duo.d.ts +3 -0
- package/dist/types/registry/oauth/google-antigravity.d.ts +11 -0
- package/dist/types/registry/oauth/google-gemini-cli.d.ts +10 -0
- package/dist/types/registry/oauth/google-oauth-shared.d.ts +22 -0
- package/dist/types/registry/oauth/index.d.ts +64 -0
- package/dist/types/registry/oauth/kimi.d.ts +21 -0
- package/dist/types/registry/oauth/minimax-code.d.ts +27 -0
- package/dist/types/registry/oauth/openai-codex.d.ts +33 -0
- package/dist/types/registry/oauth/opencode.d.ts +18 -0
- package/dist/types/registry/oauth/perplexity.d.ts +9 -0
- package/dist/types/registry/oauth/pkce.d.ts +8 -0
- package/dist/types/registry/oauth/types.d.ts +56 -0
- package/dist/types/registry/oauth/wafer.d.ts +1 -0
- package/dist/types/registry/oauth/xai-oauth.d.ts +52 -0
- package/dist/types/registry/oauth/xiaomi.d.ts +25 -0
- package/dist/types/registry/ollama-cloud.d.ts +7 -0
- package/dist/types/registry/ollama.d.ts +12 -0
- package/dist/types/registry/openai-codex-device.d.ts +8 -0
- package/dist/types/registry/openai-codex.d.ts +9 -0
- package/dist/types/registry/openai.d.ts +4 -0
- package/dist/types/registry/opencode-go.d.ts +6 -0
- package/dist/types/registry/opencode-zen.d.ts +6 -0
- package/dist/types/registry/openrouter.d.ts +13 -0
- package/dist/types/registry/parallel.d.ts +14 -0
- package/dist/types/registry/perplexity.d.ts +7 -0
- package/dist/types/registry/qianfan.d.ts +7 -0
- package/dist/types/registry/qwen-portal.d.ts +7 -0
- package/dist/types/registry/registry.d.ts +303 -0
- package/dist/types/registry/sakana.d.ts +7 -0
- package/dist/types/registry/synthetic.d.ts +6 -0
- package/dist/types/registry/tavily.d.ts +14 -0
- package/dist/types/registry/together.d.ts +6 -0
- package/dist/types/registry/types.d.ts +51 -0
- package/dist/types/registry/umans.d.ts +7 -0
- package/dist/types/registry/venice.d.ts +13 -0
- package/dist/types/registry/vercel-ai-gateway.d.ts +7 -0
- package/dist/types/registry/vllm.d.ts +7 -0
- package/dist/types/registry/wafer-serverless.d.ts +6 -0
- package/dist/types/registry/xai-oauth.d.ts +7 -0
- package/dist/types/registry/xai.d.ts +4 -0
- package/dist/types/registry/xiaomi-token-plan-ams.d.ts +6 -0
- package/dist/types/registry/xiaomi-token-plan-cn.d.ts +6 -0
- package/dist/types/registry/xiaomi-token-plan-sgp.d.ts +6 -0
- package/dist/types/registry/xiaomi.d.ts +6 -0
- package/dist/types/registry/zai.d.ts +7 -0
- package/dist/types/registry/zenmux.d.ts +7 -0
- package/dist/types/registry/zhipu-coding-plan.d.ts +7 -0
- package/dist/types/stream.d.ts +44 -0
- package/dist/types/types.d.ts +715 -0
- package/dist/types/usage/claude.d.ts +4 -0
- package/dist/types/usage/gemini.d.ts +2 -0
- package/dist/types/usage/github-copilot.d.ts +7 -0
- package/dist/types/usage/google-antigravity.d.ts +15 -0
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/usage/minimax-code.d.ts +2 -0
- package/dist/types/usage/ollama.d.ts +5 -0
- package/dist/types/usage/openai-codex-base-url.d.ts +18 -0
- package/dist/types/usage/openai-codex-reset.d.ts +79 -0
- package/dist/types/usage/openai-codex.d.ts +3 -0
- package/dist/types/usage/opencode-go.d.ts +2 -0
- package/dist/types/usage/shared.d.ts +1 -0
- package/dist/types/usage/zai.d.ts +2 -0
- package/dist/types/usage.d.ts +346 -0
- package/dist/types/utils/abort.d.ts +25 -0
- package/dist/types/utils/anthropic-auth.d.ts +35 -0
- package/dist/types/utils/block-symbols.d.ts +20 -0
- package/dist/types/utils/deterministic-id.d.ts +16 -0
- package/dist/types/utils/empty-completion-retry.d.ts +21 -0
- package/dist/types/utils/event-stream.d.ts +30 -0
- package/dist/types/utils/foundry.d.ts +1 -0
- package/dist/types/utils/google-validation.d.ts +2 -0
- package/dist/types/utils/harmony-leak.d.ts +118 -0
- package/dist/types/utils/http-inspector.d.ts +30 -0
- package/dist/types/utils/idle-iterator.d.ts +137 -0
- package/dist/types/utils/leaked-thinking-stream.d.ts +29 -0
- package/dist/types/utils/openai-http.d.ts +54 -0
- package/dist/types/utils/openrouter-headers.d.ts +1 -0
- package/dist/types/utils/parse-bind.d.ts +23 -0
- package/dist/types/utils/provider-response.d.ts +3 -0
- package/dist/types/utils/proxy.d.ts +29 -0
- package/dist/types/utils/request-debug.d.ts +29 -0
- package/dist/types/utils/retry-after.d.ts +4 -0
- package/dist/types/utils/retry.d.ts +14 -0
- package/dist/types/utils/schema/adapt.d.ts +24 -0
- package/dist/types/utils/schema/compatibility.d.ts +30 -0
- package/dist/types/utils/schema/dereference.d.ts +11 -0
- package/dist/types/utils/schema/draft.d.ts +10 -0
- package/dist/types/utils/schema/equality.d.ts +4 -0
- package/dist/types/utils/schema/fields.d.ts +54 -0
- package/dist/types/utils/schema/index.d.ts +15 -0
- package/dist/types/utils/schema/json-schema-validator.d.ts +20 -0
- package/dist/types/utils/schema/meta-validator.d.ts +2 -0
- package/dist/types/utils/schema/normalize.d.ts +124 -0
- package/dist/types/utils/schema/spill.d.ts +8 -0
- package/dist/types/utils/schema/stamps.d.ts +17 -0
- package/dist/types/utils/schema/strict-tool-validation.d.ts +16 -0
- package/dist/types/utils/schema/types.d.ts +4 -0
- package/dist/types/utils/schema/typescript.d.ts +18 -0
- package/dist/types/utils/schema/wire.d.ts +92 -0
- package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
- package/dist/types/utils/sdk-stream-timeout.d.ts +33 -0
- package/dist/types/utils/sse-debug.d.ts +5 -0
- package/dist/types/utils/stream-markup-healing.d.ts +87 -0
- package/dist/types/utils/thinking-loop.d.ts +102 -0
- package/dist/types/utils/tool-call-loop-guard.d.ts +26 -0
- package/dist/types/utils/tool-choice.d.ts +50 -0
- package/dist/types/utils/validation.d.ts +42 -0
- package/dist/types/utils.d.ts +24 -0
- package/package.json +139 -0
- package/src/api-registry.ts +109 -0
- package/src/auth-broker/client.ts +359 -0
- package/src/auth-broker/discover.ts +222 -0
- package/src/auth-broker/index.ts +7 -0
- package/src/auth-broker/refresher.ts +117 -0
- package/src/auth-broker/remote-store.ts +657 -0
- package/src/auth-broker/server.ts +646 -0
- package/src/auth-broker/snapshot-cache.ts +191 -0
- package/src/auth-broker/types.ts +130 -0
- package/src/auth-broker/wire-schemas.ts +249 -0
- package/src/auth-gateway/http.ts +194 -0
- package/src/auth-gateway/index.ts +3 -0
- package/src/auth-gateway/server.ts +802 -0
- package/src/auth-gateway/types.ts +151 -0
- package/src/auth-retry.ts +250 -0
- package/src/auth-storage.ts +5576 -0
- package/src/dialect/anthropic.md +31 -0
- package/src/dialect/anthropic.ts +608 -0
- package/src/dialect/catalog.ts +29 -0
- package/src/dialect/coercion.ts +136 -0
- package/src/dialect/deepseek.md +24 -0
- package/src/dialect/deepseek.ts +609 -0
- package/src/dialect/demotion.ts +36 -0
- package/src/dialect/examples.ts +33 -0
- package/src/dialect/factory.ts +34 -0
- package/src/dialect/fenced-thinking.ts +184 -0
- package/src/dialect/gemini.md +44 -0
- package/src/dialect/gemini.ts +597 -0
- package/src/dialect/gemma.md +33 -0
- package/src/dialect/gemma.ts +387 -0
- package/src/dialect/glm.md +32 -0
- package/src/dialect/glm.ts +456 -0
- package/src/dialect/harmony.md +31 -0
- package/src/dialect/harmony.ts +346 -0
- package/src/dialect/hermes.md +25 -0
- package/src/dialect/hermes.ts +206 -0
- package/src/dialect/history.ts +81 -0
- package/src/dialect/index.ts +15 -0
- package/src/dialect/inventory.ts +73 -0
- package/src/dialect/kimi.md +24 -0
- package/src/dialect/kimi.ts +340 -0
- package/src/dialect/minimax.md +31 -0
- package/src/dialect/minimax.ts +95 -0
- package/src/dialect/owned-stream.ts +470 -0
- package/src/dialect/prompt-template.md +12 -0
- package/src/dialect/qwen3.md +28 -0
- package/src/dialect/qwen3.ts +240 -0
- package/src/dialect/rendering.ts +249 -0
- package/src/dialect/thinking.ts +122 -0
- package/src/dialect/types.ts +57 -0
- package/src/dialect/xml.md +22 -0
- package/src/dialect/xml.ts +90 -0
- package/src/error/abort.ts +18 -0
- package/src/error/auth-classify.ts +30 -0
- package/src/error/auth.ts +48 -0
- package/src/error/aws.ts +31 -0
- package/src/error/classes.ts +186 -0
- package/src/error/finalize.ts +69 -0
- package/src/error/flags.ts +506 -0
- package/src/error/format.ts +45 -0
- package/src/error/gateway.ts +96 -0
- package/src/error/index.ts +13 -0
- package/src/error/oauth.ts +58 -0
- package/src/error/provider.ts +62 -0
- package/src/error/rate-limit.ts +161 -0
- package/src/error/retryable.ts +70 -0
- package/src/error/validation.ts +44 -0
- package/src/index.ts +49 -0
- package/src/provider-details.ts +90 -0
- package/src/providers/__tests__/google-auth.test.ts +144 -0
- package/src/providers/__tests__/kimi-code-thinking.test.ts +112 -0
- package/src/providers/__tests__/openai-codex-error.test.ts +84 -0
- package/src/providers/amazon-bedrock.ts +1042 -0
- package/src/providers/anthropic-client.ts +295 -0
- package/src/providers/anthropic-messages-server-schema.ts +252 -0
- package/src/providers/anthropic-messages-server.ts +756 -0
- package/src/providers/anthropic-wire.ts +318 -0
- package/src/providers/anthropic.ts +4078 -0
- package/src/providers/aws-credentials.ts +586 -0
- package/src/providers/aws-eventstream.ts +181 -0
- package/src/providers/aws-sigv4.ts +218 -0
- package/src/providers/azure-openai-responses.ts +382 -0
- package/src/providers/cursor/proto/agent.proto +3526 -0
- package/src/providers/cursor/proto/buf.gen.yaml +6 -0
- package/src/providers/cursor/proto/buf.yaml +17 -0
- package/src/providers/cursor.ts +2695 -0
- package/src/providers/devin/proto/buf/validate/validate.proto +468 -0
- package/src/providers/devin/proto/buf.gen.yaml +33 -0
- package/src/providers/devin/proto/buf.yaml +17 -0
- package/src/providers/devin/proto/cel/expr/checked.proto +103 -0
- package/src/providers/devin/proto/cel/expr/eval.proto +38 -0
- package/src/providers/devin/proto/cel/expr/explain.proto +15 -0
- package/src/providers/devin/proto/cel/expr/syntax.proto +113 -0
- package/src/providers/devin/proto/cel/expr/value.proto +41 -0
- package/src/providers/devin/proto/connectext/grpc/status/v1/status.proto +11 -0
- package/src/providers/devin/proto/errorspb/errors.proto +56 -0
- package/src/providers/devin/proto/errorspb/hintdetail.proto +7 -0
- package/src/providers/devin/proto/errorspb/markers.proto +10 -0
- package/src/providers/devin/proto/errorspb/tags.proto +12 -0
- package/src/providers/devin/proto/errorspb/testing.proto +6 -0
- package/src/providers/devin/proto/exa/analytics_pb/analytics.proto +188 -0
- package/src/providers/devin/proto/exa/api_server_pb/api_server.proto +2461 -0
- package/src/providers/devin/proto/exa/auth_pb/auth.proto +19 -0
- package/src/providers/devin/proto/exa/auto_cascade_common_pb/auto_cascade_common.proto +79 -0
- package/src/providers/devin/proto/exa/browser_preview_pb/browser_preview.proto +32 -0
- package/src/providers/devin/proto/exa/bug_checker_pb/bug_checker.proto +22 -0
- package/src/providers/devin/proto/exa/cascade_plugins_pb/cascade_plugins.proto +262 -0
- package/src/providers/devin/proto/exa/chat_client_server_pb/chat_client_server.proto +57 -0
- package/src/providers/devin/proto/exa/chat_pb/chat.proto +449 -0
- package/src/providers/devin/proto/exa/code_edit/code_edit_pb/code_edit.proto +186 -0
- package/src/providers/devin/proto/exa/codeium_common_pb/codeium_common.proto +4157 -0
- package/src/providers/devin/proto/exa/context_module_pb/context_module.proto +175 -0
- package/src/providers/devin/proto/exa/cortex_pb/cortex.proto +3268 -0
- package/src/providers/devin/proto/exa/dev_pb/dev.proto +26 -0
- package/src/providers/devin/proto/exa/diff_action_pb/diff_action.proto +75 -0
- package/src/providers/devin/proto/exa/eval/pr_eval/datasets_pb/datasets.proto +103 -0
- package/src/providers/devin/proto/exa/eval_pb/eval.proto +1315 -0
- package/src/providers/devin/proto/exa/extension_server_pb/extension_server.proto +556 -0
- package/src/providers/devin/proto/exa/file_system_provider_pb/file_system_provider.proto +75 -0
- package/src/providers/devin/proto/exa/index_pb/index.proto +461 -0
- package/src/providers/devin/proto/exa/knowledge_base_pb/knowledge_base.proto +144 -0
- package/src/providers/devin/proto/exa/language_server_pb/language_server.proto +2385 -0
- package/src/providers/devin/proto/exa/model_management_pb/model_management.proto +186 -0
- package/src/providers/devin/proto/exa/opensearch_clients_pb/opensearch_clients.proto +503 -0
- package/src/providers/devin/proto/exa/product_analytics_pb/product_analytics.proto +37 -0
- package/src/providers/devin/proto/exa/prompt_pb/prompt.proto +92 -0
- package/src/providers/devin/proto/exa/reactive_component_pb/reactive_component.proto +96 -0
- package/src/providers/devin/proto/exa/seat_management_pb/seat_management.proto +2680 -0
- package/src/providers/devin/proto/exa/tokenizer_pb/tokenizer.proto +37 -0
- package/src/providers/devin/proto/exa/trainer_pb/config.proto +647 -0
- package/src/providers/devin/proto/exa/tree_sitter/language_data_pb/language_data.proto +14 -0
- package/src/providers/devin/proto/exa/trust_pb/trust.proto +157 -0
- package/src/providers/devin/proto/exa/user_analytics_pb/user_analytics.proto +519 -0
- package/src/providers/devin/proto/google.golang.org/appengine/internal/base/api_base.proto +28 -0
- package/src/providers/devin/proto/google.golang.org/appengine/internal/datastore/datastore_v3.proto +484 -0
- package/src/providers/devin/proto/google.golang.org/appengine/internal/log/log_service.proto +136 -0
- package/src/providers/devin/proto/google.golang.org/appengine/internal/remote_api/remote_api.proto +42 -0
- package/src/providers/devin/proto/google.golang.org/appengine/internal/urlfetch/urlfetch_service.proto +61 -0
- package/src/providers/devin/proto/grpc/binlog/v1/binarylog.proto +84 -0
- package/src/providers/devin/proto/io/prometheus/client/metrics.proto +98 -0
- package/src/providers/devin.ts +577 -0
- package/src/providers/error-message.ts +21 -0
- package/src/providers/github-copilot-headers.ts +141 -0
- package/src/providers/gitlab-duo-workflow-chatml-note.md +1 -0
- package/src/providers/gitlab-duo-workflow.ts +3058 -0
- package/src/providers/gitlab-duo.ts +395 -0
- package/src/providers/google-auth.ts +350 -0
- package/src/providers/google-gemini-cli.ts +1362 -0
- package/src/providers/google-interactions.ts +753 -0
- package/src/providers/google-shared.ts +1103 -0
- package/src/providers/google-types.ts +180 -0
- package/src/providers/google-vertex.ts +183 -0
- package/src/providers/google.ts +87 -0
- package/src/providers/grammar.ts +70 -0
- package/src/providers/kimi.ts +52 -0
- package/src/providers/mock.ts +507 -0
- package/src/providers/ollama.ts +773 -0
- package/src/providers/openai-anthropic-shim.ts +152 -0
- package/src/providers/openai-chat-server-schema.ts +242 -0
- package/src/providers/openai-chat-server.ts +715 -0
- package/src/providers/openai-chat-wire.ts +847 -0
- package/src/providers/openai-codex/request-transformer.ts +295 -0
- package/src/providers/openai-codex/response-handler.ts +102 -0
- package/src/providers/openai-codex-responses.ts +3468 -0
- package/src/providers/openai-completions.ts +2173 -0
- package/src/providers/openai-reasoning-fallback.ts +269 -0
- package/src/providers/openai-responses-reasoning-suppression.md +1 -0
- package/src/providers/openai-responses-server-schema.ts +282 -0
- package/src/providers/openai-responses-server.ts +1280 -0
- package/src/providers/openai-responses-wire.ts +6391 -0
- package/src/providers/openai-responses.ts +1022 -0
- package/src/providers/openai-shared.ts +2648 -0
- package/src/providers/pi-native-client.ts +266 -0
- package/src/providers/pi-native-server.ts +242 -0
- package/src/providers/register-builtins.ts +475 -0
- package/src/providers/synthetic.ts +50 -0
- package/src/providers/transform-messages.ts +787 -0
- package/src/providers/vision-guard.ts +54 -0
- package/src/registry/aimlapi.ts +6 -0
- package/src/registry/alibaba-coding-plan.ts +95 -0
- package/src/registry/amazon-bedrock.ts +22 -0
- package/src/registry/anthropic.ts +26 -0
- package/src/registry/api-key-login.ts +112 -0
- package/src/registry/api-key-validation.ts +161 -0
- package/src/registry/azure.ts +6 -0
- package/src/registry/cerebras.ts +23 -0
- package/src/registry/cloudflare-ai-gateway.ts +45 -0
- package/src/registry/coreweave.ts +40 -0
- package/src/registry/cursor.ts +20 -0
- package/src/registry/deepseek.ts +46 -0
- package/src/registry/derived.ts +9 -0
- package/src/registry/devin.ts +15 -0
- package/src/registry/firepass.ts +32 -0
- package/src/registry/fireworks.ts +28 -0
- package/src/registry/github-copilot.ts +22 -0
- package/src/registry/gitlab-duo-workflow.ts +20 -0
- package/src/registry/gitlab-duo.ts +19 -0
- package/src/registry/google-antigravity.ts +22 -0
- package/src/registry/google-gemini-cli.ts +22 -0
- package/src/registry/google-vertex.ts +38 -0
- package/src/registry/google.ts +6 -0
- package/src/registry/groq.ts +6 -0
- package/src/registry/huggingface.ts +29 -0
- package/src/registry/index.ts +4 -0
- package/src/registry/kagi.ts +46 -0
- package/src/registry/kilo.ts +114 -0
- package/src/registry/kimi-code.ts +17 -0
- package/src/registry/litellm.ts +45 -0
- package/src/registry/llama-cpp.ts +35 -0
- package/src/registry/lm-studio.ts +31 -0
- package/src/registry/minimax-code-cn.ts +12 -0
- package/src/registry/minimax-code.ts +12 -0
- package/src/registry/minimax.ts +6 -0
- package/src/registry/mistral.ts +6 -0
- package/src/registry/moonshot.ts +22 -0
- package/src/registry/nanogpt.ts +22 -0
- package/src/registry/nvidia.ts +61 -0
- package/src/registry/oauth/__tests__/xai-oauth.test.ts +104 -0
- package/src/registry/oauth/anthropic.ts +311 -0
- package/src/registry/oauth/callback-server.ts +315 -0
- package/src/registry/oauth/cursor.ts +171 -0
- package/src/registry/oauth/devin.ts +124 -0
- package/src/registry/oauth/github-copilot.ts +369 -0
- package/src/registry/oauth/gitlab-duo-workflow.ts +146 -0
- package/src/registry/oauth/gitlab-duo.ts +222 -0
- package/src/registry/oauth/google-antigravity.ts +209 -0
- package/src/registry/oauth/google-gemini-cli.ts +273 -0
- package/src/registry/oauth/google-oauth-shared.ts +125 -0
- package/src/registry/oauth/index.ts +269 -0
- package/src/registry/oauth/kimi.ts +289 -0
- package/src/registry/oauth/minimax-code.ts +53 -0
- package/src/registry/oauth/oauth.html +311 -0
- package/src/registry/oauth/openai-codex.ts +364 -0
- package/src/registry/oauth/opencode.ts +50 -0
- package/src/registry/oauth/perplexity.ts +228 -0
- package/src/registry/oauth/pkce.ts +18 -0
- package/src/registry/oauth/types.ts +65 -0
- package/src/registry/oauth/wafer.ts +24 -0
- package/src/registry/oauth/xai-oauth.ts +394 -0
- package/src/registry/oauth/xiaomi.ts +211 -0
- package/src/registry/ollama-cloud.ts +36 -0
- package/src/registry/ollama.ts +43 -0
- package/src/registry/openai-codex-device.ts +18 -0
- package/src/registry/openai-codex.ts +19 -0
- package/src/registry/openai.ts +6 -0
- package/src/registry/opencode-go.ts +12 -0
- package/src/registry/opencode-zen.ts +12 -0
- package/src/registry/openrouter.ts +28 -0
- package/src/registry/parallel.ts +45 -0
- package/src/registry/perplexity.ts +13 -0
- package/src/registry/qianfan.ts +27 -0
- package/src/registry/qwen-portal.ts +50 -0
- package/src/registry/registry.ts +161 -0
- package/src/registry/sakana.ts +22 -0
- package/src/registry/synthetic.ts +21 -0
- package/src/registry/tavily.ts +45 -0
- package/src/registry/together.ts +22 -0
- package/src/registry/types.ts +56 -0
- package/src/registry/umans.ts +23 -0
- package/src/registry/venice.ts +33 -0
- package/src/registry/vercel-ai-gateway.ts +38 -0
- package/src/registry/vllm.ts +34 -0
- package/src/registry/wafer-serverless.ts +12 -0
- package/src/registry/xai-oauth.ts +17 -0
- package/src/registry/xai.ts +6 -0
- package/src/registry/xiaomi-token-plan-ams.ts +12 -0
- package/src/registry/xiaomi-token-plan-cn.ts +12 -0
- package/src/registry/xiaomi-token-plan-sgp.ts +12 -0
- package/src/registry/xiaomi.ts +12 -0
- package/src/registry/zai.ts +27 -0
- package/src/registry/zenmux.ts +22 -0
- package/src/registry/zhipu-coding-plan.ts +27 -0
- package/src/stream.ts +1778 -0
- package/src/types.ts +856 -0
- package/src/usage/claude.ts +485 -0
- package/src/usage/gemini.ts +258 -0
- package/src/usage/github-copilot.ts +424 -0
- package/src/usage/google-antigravity.ts +497 -0
- package/src/usage/kimi.ts +271 -0
- package/src/usage/minimax-code.ts +30 -0
- package/src/usage/ollama.ts +41 -0
- package/src/usage/openai-codex-base-url.ts +35 -0
- package/src/usage/openai-codex-reset.ts +174 -0
- package/src/usage/openai-codex.ts +535 -0
- package/src/usage/opencode-go.ts +89 -0
- package/src/usage/shared.ts +10 -0
- package/src/usage/zai.ts +321 -0
- package/src/usage.ts +333 -0
- package/src/utils/abort.ts +67 -0
- package/src/utils/anthropic-auth.ts +93 -0
- package/src/utils/block-symbols.ts +32 -0
- package/src/utils/deterministic-id.ts +20 -0
- package/src/utils/empty-completion-retry.ts +159 -0
- package/src/utils/event-stream.ts +171 -0
- package/src/utils/foundry.ts +8 -0
- package/src/utils/google-validation.ts +25 -0
- package/src/utils/harmony-leak.ts +456 -0
- package/src/utils/http-inspector.ts +168 -0
- package/src/utils/idle-iterator.ts +473 -0
- package/src/utils/leaked-thinking-stream.ts +294 -0
- package/src/utils/openai-http.ts +122 -0
- package/src/utils/openrouter-headers.ts +12 -0
- package/src/utils/parse-bind.ts +56 -0
- package/src/utils/provider-response.ts +30 -0
- package/src/utils/proxy.ts +240 -0
- package/src/utils/request-debug.ts +351 -0
- package/src/utils/retry-after.ts +110 -0
- package/src/utils/retry.ts +59 -0
- package/src/utils/schema/CONSTRAINTS.md +166 -0
- package/src/utils/schema/adapt.ts +36 -0
- package/src/utils/schema/compatibility.ts +435 -0
- package/src/utils/schema/dereference.ts +98 -0
- package/src/utils/schema/draft.ts +341 -0
- package/src/utils/schema/equality.ts +97 -0
- package/src/utils/schema/fields.ts +207 -0
- package/src/utils/schema/index.ts +15 -0
- package/src/utils/schema/json-schema-validator.ts +595 -0
- package/src/utils/schema/meta-validator.ts +167 -0
- package/src/utils/schema/normalize.ts +1901 -0
- package/src/utils/schema/spill.ts +43 -0
- package/src/utils/schema/stamps.ts +109 -0
- package/src/utils/schema/strict-tool-validation.ts +117 -0
- package/src/utils/schema/types.ts +10 -0
- package/src/utils/schema/typescript.ts +198 -0
- package/src/utils/schema/wire.ts +789 -0
- package/src/utils/schema/zod-decontaminate.ts +331 -0
- package/src/utils/sdk-stream-timeout.ts +43 -0
- package/src/utils/sse-debug.ts +18 -0
- package/src/utils/stream-markup-healing.ts +247 -0
- package/src/utils/thinking-loop.ts +552 -0
- package/src/utils/tool-call-loop-guard.ts +107 -0
- package/src/utils/tool-choice.ts +99 -0
- package/src/utils/validation.ts +1507 -0
- package/src/utils.ts +171 -0
package/src/stream.ts
ADDED
|
@@ -0,0 +1,1778 @@
|
|
|
1
|
+
import * as crypto from "node:crypto";
|
|
2
|
+
import * as fsSync from "node:fs";
|
|
3
|
+
import * as fs from "node:fs/promises";
|
|
4
|
+
import * as path from "node:path";
|
|
5
|
+
import { scheduler } from "node:timers/promises";
|
|
6
|
+
import type { Effort } from "jeopi-catalog/effort";
|
|
7
|
+
import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "jeopi-catalog/hosts";
|
|
8
|
+
import {
|
|
9
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
10
|
+
mapEffortToGoogleThinkingLevel,
|
|
11
|
+
minimumSupportedEffort,
|
|
12
|
+
requireSupportedEffort,
|
|
13
|
+
resolveWireModelId,
|
|
14
|
+
} from "jeopi-catalog/model-thinking";
|
|
15
|
+
import { CATALOG_PROVIDERS, type ProviderCatalogEntry } from "jeopi-catalog/provider-models";
|
|
16
|
+
import { $env, $pickenv, getConfigRootDir, isEnoent, logger, withExtraCaFetch } from "jeopi-utils";
|
|
17
|
+
import { getCustomApi } from "./api-registry";
|
|
18
|
+
import { AUTH_RETRY_STEPS, isApiKeyResolver, resolveRetryKey } from "./auth-retry";
|
|
19
|
+
import * as AIError from "./error";
|
|
20
|
+
import { ProviderHttpError } from "./error";
|
|
21
|
+
import { isUsageLimitOutcome } from "./error/rate-limit";
|
|
22
|
+
import type { BedrockOptions } from "./providers/amazon-bedrock";
|
|
23
|
+
import type { AnthropicOptions } from "./providers/anthropic";
|
|
24
|
+
import type { CursorOptions } from "./providers/cursor";
|
|
25
|
+
import type { DevinOptions } from "./providers/devin";
|
|
26
|
+
import { isGitLabDuoModel, streamGitLabDuo } from "./providers/gitlab-duo";
|
|
27
|
+
import { type GitLabDuoWorkflowOptions, streamGitLabDuoWorkflow } from "./providers/gitlab-duo-workflow";
|
|
28
|
+
import type { GoogleOptions } from "./providers/google";
|
|
29
|
+
import { getVertexAccessToken } from "./providers/google-auth";
|
|
30
|
+
import type { GoogleGeminiCliOptions } from "./providers/google-gemini-cli";
|
|
31
|
+
import type { GoogleVertexOptions } from "./providers/google-vertex";
|
|
32
|
+
import { isKimiModel, streamKimi } from "./providers/kimi";
|
|
33
|
+
import type { OllamaChatOptions } from "./providers/ollama";
|
|
34
|
+
import type { OpenAICompletionsOptions } from "./providers/openai-completions";
|
|
35
|
+
import { streamPiNative } from "./providers/pi-native-client";
|
|
36
|
+
// Heavy provider stream functions are imported lazily via register-builtins,
|
|
37
|
+
// which wraps each provider module in a dynamic import. This keeps the
|
|
38
|
+
// AWS SDK, google-auth-library, @google/genai, @bufbuild/protobuf, and
|
|
39
|
+
// other provider SDKs out of the CLI startup parse graph. The
|
|
40
|
+
// gitlab-duo / kimi / synthetic providers stay eager because their modules
|
|
41
|
+
// export routing predicates (isGitLabDuoModel, isKimiModel, isSyntheticModel)
|
|
42
|
+
// that must be callable synchronously before streaming begins, and their
|
|
43
|
+
// modules are thin wrappers with no heavy SDK dependencies.
|
|
44
|
+
import {
|
|
45
|
+
streamAnthropic,
|
|
46
|
+
streamAzureOpenAIResponses,
|
|
47
|
+
streamBedrock,
|
|
48
|
+
streamCursor,
|
|
49
|
+
streamDevin,
|
|
50
|
+
streamGoogle,
|
|
51
|
+
streamGoogleGeminiCli,
|
|
52
|
+
streamGoogleVertex,
|
|
53
|
+
streamOllama,
|
|
54
|
+
streamOpenAICodexResponses,
|
|
55
|
+
streamOpenAICompletions,
|
|
56
|
+
streamOpenAIResponses,
|
|
57
|
+
} from "./providers/register-builtins";
|
|
58
|
+
import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
|
|
59
|
+
import { PROVIDER_REGISTRY } from "./registry";
|
|
60
|
+
import type {
|
|
61
|
+
Api,
|
|
62
|
+
AssistantMessage,
|
|
63
|
+
AssistantMessageEvent,
|
|
64
|
+
Context,
|
|
65
|
+
FetchImpl,
|
|
66
|
+
Model,
|
|
67
|
+
OptionsForApi,
|
|
68
|
+
SimpleStreamOptions,
|
|
69
|
+
StreamOptions,
|
|
70
|
+
ThinkingBudgets,
|
|
71
|
+
ToolChoice,
|
|
72
|
+
} from "./types";
|
|
73
|
+
import { AssistantMessageEventStream } from "./utils/event-stream";
|
|
74
|
+
import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream";
|
|
75
|
+
import { wrapFetchForProxy } from "./utils/proxy";
|
|
76
|
+
import { withRequestDebugFetch } from "./utils/request-debug";
|
|
77
|
+
import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop";
|
|
78
|
+
|
|
79
|
+
function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
|
|
80
|
+
return (
|
|
81
|
+
model.provider === "google-vertex" &&
|
|
82
|
+
((model.api === "openai-completions" && isVertexExpressOpenAIUrl(model.baseUrl)) ||
|
|
83
|
+
(model.api === "anthropic-messages" && isVertexRawPredictUrl(model.baseUrl)))
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
type ProviderInFlightLease = {
|
|
88
|
+
path: string;
|
|
89
|
+
heartbeat: NodeJS.Timeout;
|
|
90
|
+
flushHeartbeat: () => Promise<void>;
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
type ProviderInFlightLeaseInfo = {
|
|
94
|
+
pid: number;
|
|
95
|
+
timestamp: number;
|
|
96
|
+
token: string;
|
|
97
|
+
};
|
|
98
|
+
type ProviderInFlightStaleLock = { token: string } | { mtimeMs: number };
|
|
99
|
+
type ProviderInFlightLockIdentity = { dev: number; ino: number; birthtimeMs: number };
|
|
100
|
+
|
|
101
|
+
const PROVIDER_INFLIGHT_LOCK_STALE_MS = 10_000;
|
|
102
|
+
const PROVIDER_INFLIGHT_LEASE_STALE_MS = 30_000;
|
|
103
|
+
const PROVIDER_INFLIGHT_HEARTBEAT_MS = 5_000;
|
|
104
|
+
const PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS = 250;
|
|
105
|
+
|
|
106
|
+
let configuredProviderMaxInFlightRequests: Record<string, number> = {};
|
|
107
|
+
let providerInFlightRootOverride: string | undefined;
|
|
108
|
+
|
|
109
|
+
export function configureProviderMaxInFlightRequests(limits: Record<string, number> | undefined): void {
|
|
110
|
+
configuredProviderMaxInFlightRequests = limits ?? {};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function resolveProviderInFlightLimit(
|
|
114
|
+
provider: string,
|
|
115
|
+
options?: Pick<StreamOptions, "maxInFlightRequests">,
|
|
116
|
+
): number | undefined {
|
|
117
|
+
const limits = options?.maxInFlightRequests ?? configuredProviderMaxInFlightRequests;
|
|
118
|
+
const value = limits[provider];
|
|
119
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return undefined;
|
|
120
|
+
return Math.max(1, Math.floor(value));
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function providerInFlightRoot(): string {
|
|
124
|
+
if (providerInFlightRootOverride) return providerInFlightRootOverride;
|
|
125
|
+
return path.join(getConfigRootDir(), "run", "provider-inflight");
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function providerInFlightSegment(provider: string): string {
|
|
129
|
+
return crypto.createHash("sha256").update(provider).digest("base64url");
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function providerInFlightDir(provider: string): string {
|
|
133
|
+
return path.join(providerInFlightRoot(), providerInFlightSegment(provider));
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function providerInFlightSignalPath(provider: string): string {
|
|
137
|
+
return path.join(providerInFlightDir(provider), ".wakeup");
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
function providerInFlightLockDir(provider: string): string {
|
|
141
|
+
return `${providerInFlightDir(provider)}.lock`;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// `process.kill(pid, 0)` may throw for permission/sandbox reasons even when a
|
|
145
|
+
// process exists. Treat non-ESRCH failures as alive; timestamp expiry still
|
|
146
|
+
// reaps leases whose heartbeat stopped.
|
|
147
|
+
function isProcessAlive(pid: number): boolean {
|
|
148
|
+
try {
|
|
149
|
+
process.kill(pid, 0);
|
|
150
|
+
return true;
|
|
151
|
+
} catch (error) {
|
|
152
|
+
return (error as NodeJS.ErrnoException).code !== "ESRCH";
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
async function readProviderInFlightInfo(infoPath: string): Promise<ProviderInFlightLeaseInfo | null> {
|
|
157
|
+
try {
|
|
158
|
+
const content = await fs.readFile(infoPath, "utf-8");
|
|
159
|
+
const parsed = JSON.parse(content) as Partial<ProviderInFlightLeaseInfo>;
|
|
160
|
+
if (typeof parsed.pid !== "number" || typeof parsed.timestamp !== "number" || typeof parsed.token !== "string") {
|
|
161
|
+
return null;
|
|
162
|
+
}
|
|
163
|
+
return { pid: parsed.pid, timestamp: parsed.timestamp, token: parsed.token };
|
|
164
|
+
} catch {
|
|
165
|
+
return null;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
async function writeProviderInFlightInfo(dir: string, token: string): Promise<void> {
|
|
170
|
+
const info: ProviderInFlightLeaseInfo = { pid: process.pid, timestamp: Date.now(), token };
|
|
171
|
+
const infoPath = path.join(dir, "info.json");
|
|
172
|
+
const tempPath = path.join(dir, `.info-${process.pid}-${crypto.randomUUID()}.tmp`);
|
|
173
|
+
try {
|
|
174
|
+
await Bun.write(tempPath, JSON.stringify(info));
|
|
175
|
+
await fs.rename(tempPath, infoPath);
|
|
176
|
+
} catch (error) {
|
|
177
|
+
await fs.rm(tempPath, { force: true }).catch(() => {});
|
|
178
|
+
throw error;
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
async function isProviderInFlightDirStale(dir: string, staleMs: number): Promise<boolean> {
|
|
183
|
+
const info = await readProviderInFlightInfo(path.join(dir, "info.json"));
|
|
184
|
+
if (info) {
|
|
185
|
+
if (!isProcessAlive(info.pid)) return true;
|
|
186
|
+
return Date.now() - info.timestamp > staleMs;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
try {
|
|
190
|
+
const stat = await fs.stat(path.join(dir, "info.json"));
|
|
191
|
+
return Date.now() - stat.mtimeMs > staleMs;
|
|
192
|
+
} catch (error) {
|
|
193
|
+
if (!isEnoent(error)) throw error;
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
try {
|
|
197
|
+
const stat = await fs.stat(dir);
|
|
198
|
+
return Date.now() - stat.mtimeMs > staleMs;
|
|
199
|
+
} catch (error) {
|
|
200
|
+
if (isEnoent(error)) return false;
|
|
201
|
+
throw error;
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
async function readProviderInFlightStaleLock(lockDir: string): Promise<ProviderInFlightStaleLock | null> {
|
|
206
|
+
const infoPath = path.join(lockDir, "info.json");
|
|
207
|
+
const info = await readProviderInFlightInfo(infoPath);
|
|
208
|
+
if (info) return isProcessAlive(info.pid) ? null : { token: info.token };
|
|
209
|
+
|
|
210
|
+
try {
|
|
211
|
+
const stat = await fs.stat(lockDir);
|
|
212
|
+
return Date.now() - stat.mtimeMs > PROVIDER_INFLIGHT_LOCK_STALE_MS ? { mtimeMs: stat.mtimeMs } : null;
|
|
213
|
+
} catch (error) {
|
|
214
|
+
if (isEnoent(error)) return null;
|
|
215
|
+
throw error;
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
async function readProviderInFlightLockIdentity(lockDir: string): Promise<ProviderInFlightLockIdentity> {
|
|
220
|
+
const stat = await fs.stat(lockDir);
|
|
221
|
+
return { dev: stat.dev, ino: stat.ino, birthtimeMs: stat.birthtimeMs };
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function isSameProviderInFlightLock(
|
|
225
|
+
current: ProviderInFlightLockIdentity,
|
|
226
|
+
expected: ProviderInFlightLockIdentity,
|
|
227
|
+
): boolean {
|
|
228
|
+
if (current.dev !== expected.dev) return false;
|
|
229
|
+
if (current.ino !== 0 || expected.ino !== 0) return current.ino === expected.ino;
|
|
230
|
+
return current.birthtimeMs === expected.birthtimeMs;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
async function releaseProviderInFlightStaleLock(lockDir: string, stale: ProviderInFlightStaleLock): Promise<void> {
|
|
234
|
+
if ("token" in stale) {
|
|
235
|
+
await releaseProviderInFlightLock(lockDir, stale.token);
|
|
236
|
+
return;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
const infoPath = path.join(lockDir, "info.json");
|
|
240
|
+
if (await readProviderInFlightInfo(infoPath)) return;
|
|
241
|
+
try {
|
|
242
|
+
const stat = await fs.stat(lockDir);
|
|
243
|
+
if (stat.mtimeMs !== stale.mtimeMs || Date.now() - stat.mtimeMs <= PROVIDER_INFLIGHT_LOCK_STALE_MS) return;
|
|
244
|
+
await fs.rm(lockDir, { recursive: true, force: true });
|
|
245
|
+
} catch {}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
// Best-effort token-checked release. A token mismatch means another process has
|
|
249
|
+
// already replaced the lock, so the fresh lock must be left intact.
|
|
250
|
+
async function releaseProviderInFlightLock(lockDir: string, token: string): Promise<void> {
|
|
251
|
+
try {
|
|
252
|
+
const info = await readProviderInFlightInfo(path.join(lockDir, "info.json"));
|
|
253
|
+
if (!info || info.token !== token) return;
|
|
254
|
+
await fs.rm(lockDir, { recursive: true, force: true });
|
|
255
|
+
} catch {}
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
async function releaseProviderInFlightLockDirIfSame(
|
|
259
|
+
lockDir: string,
|
|
260
|
+
identity: ProviderInFlightLockIdentity,
|
|
261
|
+
): Promise<void> {
|
|
262
|
+
try {
|
|
263
|
+
if (await readProviderInFlightInfo(path.join(lockDir, "info.json"))) return;
|
|
264
|
+
const current = await readProviderInFlightLockIdentity(lockDir);
|
|
265
|
+
if (!isSameProviderInFlightLock(current, identity)) return;
|
|
266
|
+
await fs.rm(lockDir, { recursive: true, force: true });
|
|
267
|
+
} catch {}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
async function acquireProviderInFlightLock(provider: string, signal?: AbortSignal): Promise<() => Promise<void>> {
|
|
271
|
+
const lockDir = providerInFlightLockDir(provider);
|
|
272
|
+
await fs.mkdir(path.dirname(lockDir), { recursive: true });
|
|
273
|
+
|
|
274
|
+
while (true) {
|
|
275
|
+
if (signal?.aborted) throw signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
|
|
276
|
+
try {
|
|
277
|
+
await fs.mkdir(lockDir);
|
|
278
|
+
const lockIdentity = await readProviderInFlightLockIdentity(lockDir);
|
|
279
|
+
const token = crypto.randomUUID();
|
|
280
|
+
try {
|
|
281
|
+
await writeProviderInFlightInfo(lockDir, token);
|
|
282
|
+
} catch (error) {
|
|
283
|
+
await releaseProviderInFlightLockDirIfSame(lockDir, lockIdentity);
|
|
284
|
+
throw error;
|
|
285
|
+
}
|
|
286
|
+
return async () => {
|
|
287
|
+
await releaseProviderInFlightLock(lockDir, token);
|
|
288
|
+
};
|
|
289
|
+
} catch (error) {
|
|
290
|
+
if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
const staleLock = await readProviderInFlightStaleLock(lockDir);
|
|
294
|
+
if (staleLock) {
|
|
295
|
+
await releaseProviderInFlightStaleLock(lockDir, staleLock);
|
|
296
|
+
await signalProviderInFlightWaiters(provider);
|
|
297
|
+
continue;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
await waitForProviderInFlightSignal(provider, signal);
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
async function cleanupProviderInFlightLeases(providerDir: string): Promise<number> {
|
|
305
|
+
let active = 0;
|
|
306
|
+
let entries: string[];
|
|
307
|
+
try {
|
|
308
|
+
entries = await fs.readdir(providerDir);
|
|
309
|
+
} catch (error) {
|
|
310
|
+
if (isEnoent(error)) return 0;
|
|
311
|
+
throw error;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
for (const entry of entries) {
|
|
315
|
+
const leaseDir = path.join(providerDir, entry);
|
|
316
|
+
let isDirectory = false;
|
|
317
|
+
try {
|
|
318
|
+
isDirectory = (await fs.stat(leaseDir)).isDirectory();
|
|
319
|
+
} catch (error) {
|
|
320
|
+
if (isEnoent(error)) continue;
|
|
321
|
+
throw error;
|
|
322
|
+
}
|
|
323
|
+
if (!isDirectory) continue;
|
|
324
|
+
if (await isProviderInFlightDirStale(leaseDir, PROVIDER_INFLIGHT_LEASE_STALE_MS)) {
|
|
325
|
+
await fs.rm(leaseDir, { recursive: true, force: true });
|
|
326
|
+
continue;
|
|
327
|
+
}
|
|
328
|
+
active++;
|
|
329
|
+
}
|
|
330
|
+
return active;
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
async function tryAcquireProviderInFlightLease(
|
|
334
|
+
provider: string,
|
|
335
|
+
limit: number,
|
|
336
|
+
signal?: AbortSignal,
|
|
337
|
+
): Promise<ProviderInFlightLease | null> {
|
|
338
|
+
const releaseLock = await acquireProviderInFlightLock(provider, signal);
|
|
339
|
+
try {
|
|
340
|
+
const dir = providerInFlightDir(provider);
|
|
341
|
+
await fs.mkdir(dir, { recursive: true });
|
|
342
|
+
const active = await cleanupProviderInFlightLeases(dir);
|
|
343
|
+
if (active >= limit) return null;
|
|
344
|
+
|
|
345
|
+
const leaseDir = path.join(dir, `${process.pid}-${Date.now()}-${crypto.randomUUID()}`);
|
|
346
|
+
const token = crypto.randomUUID();
|
|
347
|
+
try {
|
|
348
|
+
await fs.mkdir(leaseDir);
|
|
349
|
+
await writeProviderInFlightInfo(leaseDir, token);
|
|
350
|
+
} catch (error) {
|
|
351
|
+
await removeProviderInFlightLeaseDir(leaseDir).catch(() => {});
|
|
352
|
+
throw error;
|
|
353
|
+
}
|
|
354
|
+
let heartbeatFlush = Promise.resolve();
|
|
355
|
+
const touchHeartbeat = () => {
|
|
356
|
+
heartbeatFlush = heartbeatFlush
|
|
357
|
+
.then(
|
|
358
|
+
() => writeProviderInFlightInfo(leaseDir, token),
|
|
359
|
+
() => writeProviderInFlightInfo(leaseDir, token),
|
|
360
|
+
)
|
|
361
|
+
.catch(() => {});
|
|
362
|
+
};
|
|
363
|
+
const heartbeat = setInterval(touchHeartbeat, PROVIDER_INFLIGHT_HEARTBEAT_MS);
|
|
364
|
+
heartbeat.unref?.();
|
|
365
|
+
return { path: leaseDir, heartbeat, flushHeartbeat: () => heartbeatFlush };
|
|
366
|
+
} finally {
|
|
367
|
+
await releaseLock();
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
async function signalProviderInFlightWaitersInDir(dir: string): Promise<void> {
|
|
372
|
+
try {
|
|
373
|
+
await fs.mkdir(dir, { recursive: true });
|
|
374
|
+
await Bun.write(path.join(dir, ".wakeup"), String(Date.now()));
|
|
375
|
+
} catch {}
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
async function signalProviderInFlightWaiters(provider: string): Promise<void> {
|
|
379
|
+
await signalProviderInFlightWaitersInDir(providerInFlightDir(provider));
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
function waitForProviderInFlightSignal(provider: string, signal?: AbortSignal): Promise<void> {
|
|
383
|
+
if (signal?.aborted)
|
|
384
|
+
return Promise.reject(signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch"));
|
|
385
|
+
const signalPath = providerInFlightSignalPath(provider);
|
|
386
|
+
const waitStarted = Date.now();
|
|
387
|
+
const { promise, resolve, reject } = Promise.withResolvers<void>();
|
|
388
|
+
let settled = false;
|
|
389
|
+
let watcher: fsSync.FSWatcher | undefined;
|
|
390
|
+
const timer = setTimeout(() => finish(resolve), PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS);
|
|
391
|
+
const finish = (settle: () => void) => {
|
|
392
|
+
if (settled) return;
|
|
393
|
+
settled = true;
|
|
394
|
+
clearTimeout(timer);
|
|
395
|
+
watcher?.close();
|
|
396
|
+
signal?.removeEventListener("abort", onAbort);
|
|
397
|
+
settle();
|
|
398
|
+
};
|
|
399
|
+
const onAbort = () => {
|
|
400
|
+
finish(() => reject(signal?.reason ?? new AIError.AbortError("Provider request aborted before dispatch")));
|
|
401
|
+
};
|
|
402
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
403
|
+
try {
|
|
404
|
+
watcher = fsSync.watch(providerInFlightDir(provider), (_event, filename) => {
|
|
405
|
+
if (filename === ".wakeup" || filename === null) {
|
|
406
|
+
finish(resolve);
|
|
407
|
+
}
|
|
408
|
+
});
|
|
409
|
+
void fs.stat(signalPath).then(
|
|
410
|
+
stat => {
|
|
411
|
+
if (stat.mtimeMs >= waitStarted) finish(resolve);
|
|
412
|
+
},
|
|
413
|
+
error => {
|
|
414
|
+
if (!isEnoent(error)) finish(resolve);
|
|
415
|
+
},
|
|
416
|
+
);
|
|
417
|
+
} catch {
|
|
418
|
+
// Filesystem notifications are best-effort across platforms; the fallback
|
|
419
|
+
// timer keeps stale-lock/lease cleanup progressing if an event is dropped.
|
|
420
|
+
}
|
|
421
|
+
return promise;
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
async function removeProviderInFlightLeaseDir(leasePath: string): Promise<void> {
|
|
425
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
426
|
+
try {
|
|
427
|
+
await fs.rm(leasePath, { recursive: true, force: true });
|
|
428
|
+
return;
|
|
429
|
+
} catch (error) {
|
|
430
|
+
if (isEnoent(error)) return;
|
|
431
|
+
const code = (error as NodeJS.ErrnoException).code;
|
|
432
|
+
if (attempt < 2 && (code === "EBUSY" || code === "ENOTEMPTY" || code === "EPERM")) {
|
|
433
|
+
await Bun.sleep(25);
|
|
434
|
+
continue;
|
|
435
|
+
}
|
|
436
|
+
throw error;
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// Signal into the lease's OWN provider directory (derived from `lease.path`)
|
|
442
|
+
// rather than recomputing it from the current root. A release that lands after
|
|
443
|
+
// the in-flight root has been repointed (only the test seam does that) must not
|
|
444
|
+
// write `.wakeup` into an unrelated provider directory.
|
|
445
|
+
async function releaseProviderInFlightLease(lease: ProviderInFlightLease): Promise<void> {
|
|
446
|
+
clearInterval(lease.heartbeat);
|
|
447
|
+
await lease.flushHeartbeat();
|
|
448
|
+
await removeProviderInFlightLeaseDir(lease.path);
|
|
449
|
+
await signalProviderInFlightWaitersInDir(path.dirname(lease.path));
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
async function acquireProviderInFlightSlot(
|
|
453
|
+
provider: string,
|
|
454
|
+
limit: number | undefined,
|
|
455
|
+
signal?: AbortSignal,
|
|
456
|
+
): Promise<() => Promise<void>> {
|
|
457
|
+
if (limit === undefined) return async () => {};
|
|
458
|
+
let loggedWait = false;
|
|
459
|
+
while (true) {
|
|
460
|
+
if (signal?.aborted) throw signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
|
|
461
|
+
const lease = await tryAcquireProviderInFlightLease(provider, limit, signal);
|
|
462
|
+
if (lease) return () => releaseProviderInFlightLease(lease);
|
|
463
|
+
if (!loggedWait) {
|
|
464
|
+
loggedWait = true;
|
|
465
|
+
logger.debug("Provider in-flight limit blocked request", { provider, limit });
|
|
466
|
+
}
|
|
467
|
+
await waitForProviderInFlightSignal(provider, signal);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
export const __providerInFlightForTesting = {
|
|
472
|
+
setRoot(root: string | undefined): void {
|
|
473
|
+
providerInFlightRootOverride = root;
|
|
474
|
+
},
|
|
475
|
+
providerDir(provider: string): string {
|
|
476
|
+
return providerInFlightDir(provider);
|
|
477
|
+
},
|
|
478
|
+
lockDir(provider: string): string {
|
|
479
|
+
return providerInFlightLockDir(provider);
|
|
480
|
+
},
|
|
481
|
+
async captureStaleLockRelease(provider: string): Promise<(() => Promise<void>) | null> {
|
|
482
|
+
const lockDir = providerInFlightLockDir(provider);
|
|
483
|
+
const stale = await readProviderInFlightStaleLock(lockDir);
|
|
484
|
+
if (!stale) return null;
|
|
485
|
+
return () => releaseProviderInFlightStaleLock(lockDir, stale);
|
|
486
|
+
},
|
|
487
|
+
async captureLockDirRelease(provider: string): Promise<(() => Promise<void>) | null> {
|
|
488
|
+
const lockDir = providerInFlightLockDir(provider);
|
|
489
|
+
try {
|
|
490
|
+
const identity = await readProviderInFlightLockIdentity(lockDir);
|
|
491
|
+
return () => releaseProviderInFlightLockDirIfSame(lockDir, identity);
|
|
492
|
+
} catch {
|
|
493
|
+
return null;
|
|
494
|
+
}
|
|
495
|
+
},
|
|
496
|
+
};
|
|
497
|
+
|
|
498
|
+
function withProviderInFlightLimit<TOptions extends Pick<StreamOptions, "signal" | "maxInFlightRequests">>(
|
|
499
|
+
model: Model<Api>,
|
|
500
|
+
options: TOptions | undefined,
|
|
501
|
+
dispatch: () => AssistantMessageEventStream,
|
|
502
|
+
): AssistantMessageEventStream {
|
|
503
|
+
// Leaked-thinking healing folds in here — the one shared provider-dispatch
|
|
504
|
+
// chokepoint — so the loop guard (which wraps this) sees healed events and all
|
|
505
|
+
// six provider exits are covered by one wrap. Healing is idempotent.
|
|
506
|
+
const limit = resolveProviderInFlightLimit(model.provider, options);
|
|
507
|
+
if (limit === undefined) return wrapLeakedThinkingStream(dispatch());
|
|
508
|
+
|
|
509
|
+
const outer = new AssistantMessageEventStream();
|
|
510
|
+
void (async () => {
|
|
511
|
+
let release: (() => Promise<void>) | undefined;
|
|
512
|
+
let released = false;
|
|
513
|
+
const releaseOnce = async () => {
|
|
514
|
+
if (!release || released) return;
|
|
515
|
+
released = true;
|
|
516
|
+
await release();
|
|
517
|
+
};
|
|
518
|
+
try {
|
|
519
|
+
const startedWaitingAt = Date.now();
|
|
520
|
+
release = await acquireProviderInFlightSlot(model.provider, limit, options?.signal);
|
|
521
|
+
if (Date.now() - startedWaitingAt >= PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS) {
|
|
522
|
+
logger.debug("Provider in-flight limit wait completed", { provider: model.provider, limit });
|
|
523
|
+
}
|
|
524
|
+
if (options?.signal?.aborted) {
|
|
525
|
+
throw options.signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
|
|
526
|
+
}
|
|
527
|
+
const inner = wrapLeakedThinkingStream(dispatch());
|
|
528
|
+
try {
|
|
529
|
+
for await (const event of inner) {
|
|
530
|
+
outer.push(event);
|
|
531
|
+
if (outer.done) return;
|
|
532
|
+
}
|
|
533
|
+
if (!outer.done) outer.end(await inner.result());
|
|
534
|
+
} finally {
|
|
535
|
+
await releaseOnce();
|
|
536
|
+
}
|
|
537
|
+
} catch (error) {
|
|
538
|
+
await releaseOnce();
|
|
539
|
+
if (!outer.done) outer.fail(error);
|
|
540
|
+
}
|
|
541
|
+
})();
|
|
542
|
+
return outer;
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
function createVertexAuthenticatedFetch(options: StreamOptions | undefined): FetchImpl {
|
|
546
|
+
const baseFetch = options?.fetch ?? fetch;
|
|
547
|
+
const vertexFetch = async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
|
548
|
+
const token = await getVertexAccessToken({ signal: options?.signal, fetch: baseFetch });
|
|
549
|
+
const headers = new Headers(init?.headers);
|
|
550
|
+
headers.set("Authorization", `Bearer ${token}`);
|
|
551
|
+
const rewritten = resolveVertexRequest(input);
|
|
552
|
+
const url = rewritten instanceof Request ? rewritten.url : rewritten.toString();
|
|
553
|
+
if (isVertexRawPredictUrl(url)) {
|
|
554
|
+
const bodyText = await readVertexRequestBody(rewritten, init);
|
|
555
|
+
const transformed = transformVertexAnthropicBody(bodyText);
|
|
556
|
+
return baseFetch(url, {
|
|
557
|
+
...init,
|
|
558
|
+
method: init?.method ?? (rewritten instanceof Request ? rewritten.method : "POST"),
|
|
559
|
+
headers,
|
|
560
|
+
body: transformed,
|
|
561
|
+
});
|
|
562
|
+
}
|
|
563
|
+
return baseFetch(rewritten, { ...init, headers });
|
|
564
|
+
};
|
|
565
|
+
return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {});
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
async function readVertexRequestBody(input: string | URL | Request, init: RequestInit | undefined): Promise<string> {
|
|
569
|
+
if (input instanceof Request) return input.clone().text();
|
|
570
|
+
const body = init?.body;
|
|
571
|
+
if (typeof body === "string") return body;
|
|
572
|
+
if (body instanceof Uint8Array) return new TextDecoder().decode(body);
|
|
573
|
+
if (body instanceof ArrayBuffer) return new TextDecoder().decode(body);
|
|
574
|
+
return "";
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
// Vertex Claude rejects the standard Anthropic body shape: the `model` field
|
|
578
|
+
// is encoded in the URL path and `anthropic_version: "vertex-2023-10-16"` is
|
|
579
|
+
// required in the JSON body instead of the `anthropic-version` HTTP header.
|
|
580
|
+
function transformVertexAnthropicBody(bodyText: string): string {
|
|
581
|
+
if (!bodyText) return bodyText;
|
|
582
|
+
try {
|
|
583
|
+
const payload = JSON.parse(bodyText) as Record<string, unknown>;
|
|
584
|
+
delete payload.model;
|
|
585
|
+
payload.anthropic_version = "vertex-2023-10-16";
|
|
586
|
+
return JSON.stringify(payload);
|
|
587
|
+
} catch {
|
|
588
|
+
return bodyText;
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
function resolveVertexRequest(input: string | URL | Request): string | URL | Request {
|
|
593
|
+
const project = $env.GOOGLE_CLOUD_PROJECT || $env.GCP_PROJECT || $env.GCLOUD_PROJECT;
|
|
594
|
+
const location = $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION;
|
|
595
|
+
if (!project || !location) return input;
|
|
596
|
+
|
|
597
|
+
const rewriteUrl = (url: string): string => {
|
|
598
|
+
const hasPlaceholder =
|
|
599
|
+
url.includes("{project}") ||
|
|
600
|
+
url.includes("{location}") ||
|
|
601
|
+
url.includes("%7Bproject%7D") ||
|
|
602
|
+
url.includes("%7Blocation%7D");
|
|
603
|
+
const host = location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`;
|
|
604
|
+
const rewritten = hasPlaceholder
|
|
605
|
+
? url
|
|
606
|
+
.replace("https://{location}-aiplatform.googleapis.com", `https://${host}`)
|
|
607
|
+
.replace("https://%7Blocation%7D-aiplatform.googleapis.com", `https://${host}`)
|
|
608
|
+
.replaceAll("{project}", encodeURIComponent(project))
|
|
609
|
+
.replaceAll("%7Bproject%7D", encodeURIComponent(project))
|
|
610
|
+
.replaceAll("{location}", encodeURIComponent(location))
|
|
611
|
+
.replaceAll("%7Blocation%7D", encodeURIComponent(location))
|
|
612
|
+
: url;
|
|
613
|
+
return rewritten.replace(":streamRawPredict/v1/messages", ":streamRawPredict");
|
|
614
|
+
};
|
|
615
|
+
|
|
616
|
+
if (input instanceof Request) {
|
|
617
|
+
const rewrittenUrl = rewriteUrl(input.url);
|
|
618
|
+
return rewrittenUrl === input.url ? input : new Request(rewrittenUrl, input);
|
|
619
|
+
}
|
|
620
|
+
if (input instanceof URL) {
|
|
621
|
+
const rewrittenUrl = rewriteUrl(input.toString());
|
|
622
|
+
return rewrittenUrl === input.toString() ? input : new URL(rewrittenUrl);
|
|
623
|
+
}
|
|
624
|
+
return rewriteUrl(input);
|
|
625
|
+
}
|
|
626
|
+
|
|
627
|
+
type KeyResolver = string | (() => string | undefined);
|
|
628
|
+
|
|
629
|
+
const LEGACY_ENV_KEYS: Record<string, KeyResolver> = {
|
|
630
|
+
// Non-provider / search-tool keys and API-name keys not modeled as registry provider defs.
|
|
631
|
+
"azure-openai-responses": "AZURE_OPENAI_API_KEY",
|
|
632
|
+
exa: "EXA_API_KEY",
|
|
633
|
+
jina: "JINA_API_KEY",
|
|
634
|
+
brave: "BRAVE_API_KEY",
|
|
635
|
+
tinyfish: "TINYFISH_API_KEY",
|
|
636
|
+
firecrawl: "FIRECRAWL_API_KEY",
|
|
637
|
+
};
|
|
638
|
+
|
|
639
|
+
/**
|
|
640
|
+
* Env fallbacks derived from the catalog table — the single source for plain
|
|
641
|
+
* provider env-var names. Registry defs override with computed resolvers
|
|
642
|
+
* (Foundry/ADC/Bedrock probes); legacy non-provider keys merge last.
|
|
643
|
+
*/
|
|
644
|
+
const CATALOG_ENTRY_ENV_KEYS = (CATALOG_PROVIDERS as readonly ProviderCatalogEntry[]).flatMap(provider => {
|
|
645
|
+
const envVars = provider.envVars;
|
|
646
|
+
if (!envVars || envVars.length === 0) return [];
|
|
647
|
+
const resolver: KeyResolver = envVars.length === 1 ? envVars[0] : () => $pickenv(...envVars);
|
|
648
|
+
return [[provider.id, resolver] as [string, KeyResolver]];
|
|
649
|
+
});
|
|
650
|
+
|
|
651
|
+
const serviceProviderMap: Record<string, KeyResolver> = {
|
|
652
|
+
...Object.fromEntries(CATALOG_ENTRY_ENV_KEYS),
|
|
653
|
+
...Object.fromEntries(
|
|
654
|
+
PROVIDER_REGISTRY.flatMap(provider =>
|
|
655
|
+
provider.envKeys != null ? [[provider.id, provider.envKeys] as [string, KeyResolver]] : [],
|
|
656
|
+
),
|
|
657
|
+
),
|
|
658
|
+
...LEGACY_ENV_KEYS,
|
|
659
|
+
};
|
|
660
|
+
|
|
661
|
+
/**
|
|
662
|
+
* Get API key for provider from known environment variables, e.g. OPENAI_API_KEY.
|
|
663
|
+
*
|
|
664
|
+
* Will not return API keys for providers that require OAuth tokens.
|
|
665
|
+
* Checks Bun.env, then cwd/.env, then ~/.env.
|
|
666
|
+
*/
|
|
667
|
+
export function getEnvApiKey(provider: string): string | undefined {
|
|
668
|
+
const resolver = serviceProviderMap[provider];
|
|
669
|
+
if (typeof resolver === "string") {
|
|
670
|
+
return $env[resolver];
|
|
671
|
+
}
|
|
672
|
+
return resolver?.();
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
/**
|
|
676
|
+
* Name of the environment variable that backs `getEnvApiKey` for a provider,
|
|
677
|
+
* when that provider maps to a single named variable (e.g. `github-copilot` →
|
|
678
|
+
* `COPILOT_GITHUB_TOKEN`). Returns undefined for providers whose env fallback
|
|
679
|
+
* is computed (multi-var pickers, Vertex ADC / Bedrock probes, …) since no
|
|
680
|
+
* single variable name describes the source.
|
|
681
|
+
*/
|
|
682
|
+
export function getEnvApiKeyName(provider: string): string | undefined {
|
|
683
|
+
const resolver = serviceProviderMap[provider];
|
|
684
|
+
return typeof resolver === "string" ? resolver : undefined;
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
/**
|
|
688
|
+
* Enumerate every provider that has an env-var fallback for `getEnvApiKey`.
|
|
689
|
+
* Used by `omp auth-broker migrate --include-env` to discover env-sourced keys
|
|
690
|
+
* that should be uploaded to the broker.
|
|
691
|
+
*/
|
|
692
|
+
export function listProvidersWithEnvKey(): string[] {
|
|
693
|
+
return Object.keys(serviceProviderMap);
|
|
694
|
+
}
|
|
695
|
+
|
|
696
|
+
export function stream<TApi extends Api>(
|
|
697
|
+
model: Model<TApi>,
|
|
698
|
+
context: Context,
|
|
699
|
+
options?: OptionsForApi<TApi>,
|
|
700
|
+
): AssistantMessageEventStream {
|
|
701
|
+
return withGeminiThinkingLoopGuard(model, options, opts =>
|
|
702
|
+
withProviderInFlightLimit(model, opts, () => streamDispatch(model, context, opts)),
|
|
703
|
+
);
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
function streamDispatch<TApi extends Api>(
|
|
707
|
+
model: Model<TApi>,
|
|
708
|
+
context: Context,
|
|
709
|
+
options?: OptionsForApi<TApi>,
|
|
710
|
+
): AssistantMessageEventStream {
|
|
711
|
+
const baseOptions = (options || {}) as StreamOptions;
|
|
712
|
+
const debugOptions = withExtraCaFetch(withRequestDebugFetch(baseOptions));
|
|
713
|
+
const requestOptions = {
|
|
714
|
+
...debugOptions,
|
|
715
|
+
fetch: wrapFetchForProxy(debugOptions.fetch ?? (globalThis.fetch as FetchImpl), model.provider),
|
|
716
|
+
} as OptionsForApi<TApi>;
|
|
717
|
+
|
|
718
|
+
// Check custom API registry first (extension-provided APIs like "vertex-claude-api")
|
|
719
|
+
const customApiProvider = getCustomApi(model.api);
|
|
720
|
+
if (customApiProvider) {
|
|
721
|
+
return customApiProvider.stream(model, context, requestOptions as StreamOptions);
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
if (isGitLabDuoModel(model)) {
|
|
725
|
+
const apiKey = requestOptions.apiKey || getEnvApiKey(model.provider);
|
|
726
|
+
if (!apiKey) {
|
|
727
|
+
throw new AIError.MissingApiKeyError(model.provider);
|
|
728
|
+
}
|
|
729
|
+
return streamGitLabDuo(model, context, {
|
|
730
|
+
...(requestOptions as SimpleStreamOptions),
|
|
731
|
+
apiKey,
|
|
732
|
+
});
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
if (model.api === "gitlab-duo-agent") {
|
|
736
|
+
const apiKey = (requestOptions as StreamOptions | undefined)?.apiKey || getEnvApiKey(model.provider);
|
|
737
|
+
if (!apiKey) {
|
|
738
|
+
throw new AIError.MissingApiKeyError(model.provider);
|
|
739
|
+
}
|
|
740
|
+
return streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
|
|
741
|
+
...(requestOptions as StreamOptions | undefined),
|
|
742
|
+
apiKey,
|
|
743
|
+
} as GitLabDuoWorkflowOptions);
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
// Vertex AI uses Application Default Credentials, not API keys
|
|
747
|
+
if (model.api === "google-vertex") {
|
|
748
|
+
return streamGoogleVertex(model as Model<"google-vertex">, context, requestOptions as GoogleVertexOptions);
|
|
749
|
+
} else if (model.api === "bedrock-converse-stream") {
|
|
750
|
+
// Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
|
|
751
|
+
return streamBedrock(model as Model<"bedrock-converse-stream">, context, requestOptions as BedrockOptions);
|
|
752
|
+
}
|
|
753
|
+
|
|
754
|
+
const apiKey = requestOptions.apiKey || getEnvApiKey(model.provider);
|
|
755
|
+
if (!apiKey) {
|
|
756
|
+
throw new AIError.MissingApiKeyError(model.provider);
|
|
757
|
+
}
|
|
758
|
+
const providerOptions = isGoogleVertexAuthenticatedModel(model)
|
|
759
|
+
? {
|
|
760
|
+
...requestOptions,
|
|
761
|
+
apiKey: "vertex-adc",
|
|
762
|
+
fetch: createVertexAuthenticatedFetch(requestOptions),
|
|
763
|
+
}
|
|
764
|
+
: { ...requestOptions, apiKey };
|
|
765
|
+
|
|
766
|
+
const api: Api = model.api;
|
|
767
|
+
switch (api) {
|
|
768
|
+
case "anthropic-messages": {
|
|
769
|
+
const anthropicOptions = providerOptions as AnthropicOptions;
|
|
770
|
+
return streamAnthropic(model as Model<"anthropic-messages">, context, {
|
|
771
|
+
...anthropicOptions,
|
|
772
|
+
isOAuth: anthropicOptions.isOAuth ?? model.isOAuth,
|
|
773
|
+
});
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
case "openrouter": {
|
|
777
|
+
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
|
|
778
|
+
if (useResponses) {
|
|
779
|
+
return streamOpenAIResponses(
|
|
780
|
+
model as Model<"openai-responses">,
|
|
781
|
+
context,
|
|
782
|
+
providerOptions as OptionsForApi<"openai-responses">,
|
|
783
|
+
);
|
|
784
|
+
}
|
|
785
|
+
return streamOpenAICompletions(
|
|
786
|
+
model as Model<"openai-completions">,
|
|
787
|
+
context,
|
|
788
|
+
providerOptions as OptionsForApi<"openai-completions">,
|
|
789
|
+
);
|
|
790
|
+
}
|
|
791
|
+
|
|
792
|
+
case "openai-completions":
|
|
793
|
+
return streamOpenAICompletions(
|
|
794
|
+
model as Model<"openai-completions">,
|
|
795
|
+
context,
|
|
796
|
+
providerOptions as OptionsForApi<"openai-completions">,
|
|
797
|
+
);
|
|
798
|
+
|
|
799
|
+
case "openai-responses":
|
|
800
|
+
return streamOpenAIResponses(
|
|
801
|
+
model as Model<"openai-responses">,
|
|
802
|
+
context,
|
|
803
|
+
providerOptions as OptionsForApi<"openai-responses">,
|
|
804
|
+
);
|
|
805
|
+
|
|
806
|
+
case "azure-openai-responses":
|
|
807
|
+
return streamAzureOpenAIResponses(
|
|
808
|
+
model as Model<"azure-openai-responses">,
|
|
809
|
+
context,
|
|
810
|
+
providerOptions as OptionsForApi<"azure-openai-responses">,
|
|
811
|
+
);
|
|
812
|
+
|
|
813
|
+
case "openai-codex-responses":
|
|
814
|
+
return streamOpenAICodexResponses(
|
|
815
|
+
model as Model<"openai-codex-responses">,
|
|
816
|
+
context,
|
|
817
|
+
providerOptions as OptionsForApi<"openai-codex-responses">,
|
|
818
|
+
);
|
|
819
|
+
|
|
820
|
+
case "google-generative-ai":
|
|
821
|
+
return streamGoogle(model as Model<"google-generative-ai">, context, providerOptions);
|
|
822
|
+
|
|
823
|
+
case "google-gemini-cli":
|
|
824
|
+
return streamGoogleGeminiCli(
|
|
825
|
+
model as Model<"google-gemini-cli">,
|
|
826
|
+
context,
|
|
827
|
+
providerOptions as GoogleGeminiCliOptions,
|
|
828
|
+
);
|
|
829
|
+
|
|
830
|
+
case "ollama-chat":
|
|
831
|
+
return streamOllama(model as Model<"ollama-chat">, context, providerOptions as OllamaChatOptions);
|
|
832
|
+
|
|
833
|
+
case "cursor-agent":
|
|
834
|
+
return streamCursor(model as Model<"cursor-agent">, context, providerOptions as CursorOptions);
|
|
835
|
+
|
|
836
|
+
case "devin-agent":
|
|
837
|
+
return streamDevin(model as Model<"devin-agent">, context, providerOptions as DevinOptions);
|
|
838
|
+
|
|
839
|
+
default:
|
|
840
|
+
throw new AIError.ConfigurationError(`Unhandled API: ${api}`);
|
|
841
|
+
}
|
|
842
|
+
}
|
|
843
|
+
|
|
844
|
+
/** Thinking-loop re-samples spent before {@link resolveWithThinkingLoopCook} cooks. */
|
|
845
|
+
const THINKING_LOOP_MAX_ABORTS = 3;
|
|
846
|
+
const THINKING_LOOP_RETRY_BASE_DELAY_MS = 500;
|
|
847
|
+
const THINKING_LOOP_RETRY_MAX_DELAY_MS = 8_000;
|
|
848
|
+
|
|
849
|
+
/**
|
|
850
|
+
* Resolve a completion, re-sampling a thinking-loop stall up to
|
|
851
|
+
* {@link THINKING_LOOP_MAX_ABORTS} times before letting it cook. The loop guard
|
|
852
|
+
* raises an empty `stopReason: "error"` stall on each guarded attempt; this
|
|
853
|
+
* result-path consumer re-dispatches a fresh request per stall and, once the abort
|
|
854
|
+
* budget is spent, runs one final pass with the guard disabled so a stubborn loop
|
|
855
|
+
* returns the model's raw output instead of a fatal stall. Non-stall results —
|
|
856
|
+
* including genuine errors — return immediately; a caller abort during backoff
|
|
857
|
+
* propagates so cancellation surfaces as an abort, never a stale stall result.
|
|
858
|
+
*/
|
|
859
|
+
async function resolveWithThinkingLoopCook(
|
|
860
|
+
signal: AbortSignal | undefined,
|
|
861
|
+
dispatch: () => AssistantMessageEventStream,
|
|
862
|
+
cook: () => AssistantMessageEventStream,
|
|
863
|
+
): Promise<AssistantMessage> {
|
|
864
|
+
let message = await dispatch().result();
|
|
865
|
+
let thinkingLoopRetry = AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
|
|
866
|
+
for (let attempt = 0; thinkingLoopRetry && attempt < THINKING_LOOP_MAX_ABORTS - 1; attempt += 1) {
|
|
867
|
+
// A caller abort surfaces as a thrown abort (never the stall, which would
|
|
868
|
+
// misclassify as a 502): throwIfAborted before backoff, and scheduler.wait
|
|
869
|
+
// rejects if the abort lands mid-delay.
|
|
870
|
+
signal?.throwIfAborted();
|
|
871
|
+
const delay = Math.min(THINKING_LOOP_RETRY_BASE_DELAY_MS * 2 ** attempt, THINKING_LOOP_RETRY_MAX_DELAY_MS);
|
|
872
|
+
await scheduler.wait(delay, { signal });
|
|
873
|
+
message = await dispatch().result();
|
|
874
|
+
thinkingLoopRetry =
|
|
875
|
+
message.stopReason === "error" &&
|
|
876
|
+
message.content.length === 0 &&
|
|
877
|
+
AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
|
|
878
|
+
}
|
|
879
|
+
if (!thinkingLoopRetry) return message;
|
|
880
|
+
signal?.throwIfAborted();
|
|
881
|
+
// Abort budget spent and still looping: let it cook with the guard disabled.
|
|
882
|
+
return cook().result();
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
export async function complete<TApi extends Api>(
|
|
886
|
+
model: Model<TApi>,
|
|
887
|
+
context: Context,
|
|
888
|
+
options?: OptionsForApi<TApi>,
|
|
889
|
+
): Promise<AssistantMessage> {
|
|
890
|
+
return resolveWithThinkingLoopCook(
|
|
891
|
+
options?.signal,
|
|
892
|
+
() => stream(model, context, options),
|
|
893
|
+
() => stream(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
|
|
894
|
+
);
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
type AuthRetryFailure = {
|
|
898
|
+
error: unknown;
|
|
899
|
+
bufferedEvents: AssistantMessageEvent[];
|
|
900
|
+
terminalEvent?: Extract<AssistantMessageEvent, { type: "error" }>;
|
|
901
|
+
};
|
|
902
|
+
|
|
903
|
+
function extractStatusFromAssistantError(message: AssistantMessage): number | undefined {
|
|
904
|
+
if (message.errorStatus !== undefined) return message.errorStatus;
|
|
905
|
+
if (!message.errorMessage) return undefined;
|
|
906
|
+
return AIError.status({ message: message.errorMessage });
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
function isRetryableUpstreamError(error: unknown, status: number | undefined, message: string | undefined): boolean {
|
|
910
|
+
// 401 means the credential is bad. Usage-limit phrasing (Codex's
|
|
911
|
+
// "You have hit your ChatGPT usage limit", Anthropic's "usage_limit_reached",
|
|
912
|
+
// Google's "resource_exhausted", OpenAI's "insufficient_quota") and 429s
|
|
913
|
+
// without transient rate-limit wording mean this account is parked but a
|
|
914
|
+
// sibling credential can usually pick the request up. Both are rotatable
|
|
915
|
+
// via `onAuthError` — the auth-gateway maps the former to
|
|
916
|
+
// `invalidateCredentialMatching` and the latter to
|
|
917
|
+
// `markUsageLimitReached`. Transient 429s ("Too many requests",
|
|
918
|
+
// per-minute caps) classify as RATE_LIMIT_EXCEEDED in
|
|
919
|
+
// `parseRateLimitReason` and stay in the provider's own backoff layer
|
|
920
|
+
// instead of burning siblings.
|
|
921
|
+
if (status === 401) return true;
|
|
922
|
+
void error;
|
|
923
|
+
return isUsageLimitOutcome(status, message);
|
|
924
|
+
}
|
|
925
|
+
|
|
926
|
+
function createAssistantAuthError(message: AssistantMessage): Error {
|
|
927
|
+
const text = message.errorMessage ?? "Provider authentication failed";
|
|
928
|
+
const status = extractStatusFromAssistantError(message);
|
|
929
|
+
return status === undefined
|
|
930
|
+
? new AIError.ProviderResponseError(text, { kind: "runtime" })
|
|
931
|
+
: new ProviderHttpError(text, status);
|
|
932
|
+
}
|
|
933
|
+
|
|
934
|
+
function emitBufferedEvents(stream: AssistantMessageEventStream, events: AssistantMessageEvent[]): void {
|
|
935
|
+
for (const event of events) {
|
|
936
|
+
stream.push(event);
|
|
937
|
+
}
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
export function streamSimple<TApi extends Api>(
|
|
941
|
+
model: Model<TApi>,
|
|
942
|
+
context: Context,
|
|
943
|
+
options?: SimpleStreamOptions,
|
|
944
|
+
): AssistantMessageEventStream {
|
|
945
|
+
const baseOptions = (options || {}) as SimpleStreamOptions;
|
|
946
|
+
const debugOptions = withExtraCaFetch(withRequestDebugFetch(baseOptions));
|
|
947
|
+
const requestOptions = {
|
|
948
|
+
...debugOptions,
|
|
949
|
+
fetch: wrapFetchForProxy(debugOptions.fetch ?? (globalThis.fetch as FetchImpl), model.provider),
|
|
950
|
+
} as SimpleStreamOptions;
|
|
951
|
+
const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined;
|
|
952
|
+
if (apiKeyResolver) {
|
|
953
|
+
const outer = new AssistantMessageEventStream();
|
|
954
|
+
const signal = requestOptions?.signal;
|
|
955
|
+
// One inner attempt against a resolved string key. When
|
|
956
|
+
// `captureAuthFailure` is set, a retryable auth error that arrives before
|
|
957
|
+
// any replay-unsafe event is buffered and returned (so the caller can
|
|
958
|
+
// retry with a fresh key) instead of surfaced. The terminal attempt
|
|
959
|
+
// clears the flag and emits whatever it gets.
|
|
960
|
+
const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
|
|
961
|
+
const bufferedEvents: AssistantMessageEvent[] = [];
|
|
962
|
+
let emittedReplayUnsafeEvent = false;
|
|
963
|
+
const flushBuffered = (): void => {
|
|
964
|
+
emitBufferedEvents(outer, bufferedEvents);
|
|
965
|
+
bufferedEvents.length = 0;
|
|
966
|
+
};
|
|
967
|
+
|
|
968
|
+
try {
|
|
969
|
+
const inner = streamSimple(model, context, { ...requestOptions, apiKey });
|
|
970
|
+
for await (const event of inner) {
|
|
971
|
+
if (!emittedReplayUnsafeEvent && event.type === "start") {
|
|
972
|
+
bufferedEvents.push(event);
|
|
973
|
+
continue;
|
|
974
|
+
}
|
|
975
|
+
if (
|
|
976
|
+
!emittedReplayUnsafeEvent &&
|
|
977
|
+
captureAuthFailure &&
|
|
978
|
+
event.type === "error" &&
|
|
979
|
+
isRetryableUpstreamError(
|
|
980
|
+
event.error,
|
|
981
|
+
extractStatusFromAssistantError(event.error),
|
|
982
|
+
event.error.errorMessage,
|
|
983
|
+
)
|
|
984
|
+
) {
|
|
985
|
+
return { error: createAssistantAuthError(event.error), bufferedEvents, terminalEvent: event };
|
|
986
|
+
}
|
|
987
|
+
flushBuffered();
|
|
988
|
+
emittedReplayUnsafeEvent = true;
|
|
989
|
+
outer.push(event);
|
|
990
|
+
if (outer.done) return undefined;
|
|
991
|
+
}
|
|
992
|
+
flushBuffered();
|
|
993
|
+
if (!outer.done) outer.end(await inner.result());
|
|
994
|
+
} catch (error) {
|
|
995
|
+
if (
|
|
996
|
+
!emittedReplayUnsafeEvent &&
|
|
997
|
+
captureAuthFailure &&
|
|
998
|
+
isRetryableUpstreamError(
|
|
999
|
+
error,
|
|
1000
|
+
AIError.status(error),
|
|
1001
|
+
error instanceof Error ? error.message : undefined,
|
|
1002
|
+
)
|
|
1003
|
+
) {
|
|
1004
|
+
return { error, bufferedEvents };
|
|
1005
|
+
}
|
|
1006
|
+
flushBuffered();
|
|
1007
|
+
outer.fail(error);
|
|
1008
|
+
}
|
|
1009
|
+
return undefined;
|
|
1010
|
+
};
|
|
1011
|
+
const emitFailure = (failure: AuthRetryFailure): void => {
|
|
1012
|
+
emitBufferedEvents(outer, failure.bufferedEvents);
|
|
1013
|
+
if (failure.terminalEvent) {
|
|
1014
|
+
outer.push(failure.terminalEvent);
|
|
1015
|
+
} else {
|
|
1016
|
+
outer.fail(failure.error);
|
|
1017
|
+
}
|
|
1018
|
+
};
|
|
1019
|
+
|
|
1020
|
+
void (async () => {
|
|
1021
|
+
let lastKey: string | undefined;
|
|
1022
|
+
try {
|
|
1023
|
+
lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined;
|
|
1024
|
+
} catch (error) {
|
|
1025
|
+
// A thrown resolver is a broker/OAuth/network failure, not a missing
|
|
1026
|
+
// key — surface the cause instead of masking it as "No API key".
|
|
1027
|
+
outer.fail(
|
|
1028
|
+
new AIError.ConfigurationError(
|
|
1029
|
+
`Failed to resolve API key for provider ${model.provider}: ${error instanceof Error ? error.message : String(error)}`,
|
|
1030
|
+
{ cause: error },
|
|
1031
|
+
),
|
|
1032
|
+
);
|
|
1033
|
+
return;
|
|
1034
|
+
}
|
|
1035
|
+
if (lastKey === undefined) {
|
|
1036
|
+
outer.fail(new AIError.MissingApiKeyError(model.provider));
|
|
1037
|
+
return;
|
|
1038
|
+
}
|
|
1039
|
+
let failure = await runAttempt(lastKey, true);
|
|
1040
|
+
if (!failure) return;
|
|
1041
|
+
// a/b/c policy: refresh the same account (lastChance=false), then
|
|
1042
|
+
// switch to a sibling (lastChance=true). A step is skipped when the
|
|
1043
|
+
// resolver yields the same key it just tried or `undefined`; the
|
|
1044
|
+
// final step's attempt clears the capture flag so it emits directly.
|
|
1045
|
+
for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) {
|
|
1046
|
+
// Caller aborted between attempts: don't mint a fresh token or fire
|
|
1047
|
+
// another doomed request — emit the captured failure instead.
|
|
1048
|
+
if (signal?.aborted) break;
|
|
1049
|
+
const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal);
|
|
1050
|
+
if (nextKey === undefined || nextKey === lastKey) continue;
|
|
1051
|
+
lastKey = nextKey;
|
|
1052
|
+
const isLastStep = step === AUTH_RETRY_STEPS.length - 1;
|
|
1053
|
+
const next = await runAttempt(nextKey, !isLastStep);
|
|
1054
|
+
if (!next) return;
|
|
1055
|
+
failure = next;
|
|
1056
|
+
}
|
|
1057
|
+
emitFailure(failure);
|
|
1058
|
+
})();
|
|
1059
|
+
return outer;
|
|
1060
|
+
}
|
|
1061
|
+
|
|
1062
|
+
// Pi-native transport short-circuits the per-provider dispatch entirely:
|
|
1063
|
+
// the gateway resolves provider + credential server-side, so we don't
|
|
1064
|
+
// need an `apiKey` from `getEnvApiKey` here — `options.apiKey` carries
|
|
1065
|
+
// the gateway bearer instead. Comes BEFORE the custom-API check so
|
|
1066
|
+
// extension-registered APIs can't accidentally override a configured
|
|
1067
|
+
// pi-native transport.
|
|
1068
|
+
if (model.transport === "pi-native") {
|
|
1069
|
+
return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
|
|
1070
|
+
withProviderInFlightLimit(model, opts, () => streamPiNative(model, context, opts)),
|
|
1071
|
+
);
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1074
|
+
// Check custom API registry (extension-provided APIs)
|
|
1075
|
+
const customApiProvider = getCustomApi(model.api);
|
|
1076
|
+
if (customApiProvider) {
|
|
1077
|
+
return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
|
|
1078
|
+
withProviderInFlightLimit(model, opts, () => customApiProvider.streamSimple(model, context, opts)),
|
|
1079
|
+
);
|
|
1080
|
+
}
|
|
1081
|
+
|
|
1082
|
+
// Vertex AI uses Application Default Credentials, not API keys
|
|
1083
|
+
if (model.api === "google-vertex") {
|
|
1084
|
+
const providerOptions = mapOptionsForApi(model, requestOptions, undefined);
|
|
1085
|
+
return stream(model, context, providerOptions);
|
|
1086
|
+
} else if (model.api === "bedrock-converse-stream") {
|
|
1087
|
+
// Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
|
|
1088
|
+
const providerOptions = mapOptionsForApi(model, requestOptions, undefined);
|
|
1089
|
+
return stream(model, context, providerOptions);
|
|
1090
|
+
}
|
|
1091
|
+
|
|
1092
|
+
// The resolver form is handled by the wrapper above; only a static string
|
|
1093
|
+
// key reaches this point.
|
|
1094
|
+
const apiKey =
|
|
1095
|
+
(typeof requestOptions?.apiKey === "string" ? requestOptions.apiKey : undefined) || getEnvApiKey(model.provider);
|
|
1096
|
+
if (!apiKey) {
|
|
1097
|
+
throw new AIError.MissingApiKeyError(model.provider);
|
|
1098
|
+
}
|
|
1099
|
+
|
|
1100
|
+
// GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
|
|
1101
|
+
if (isGitLabDuoModel(model)) {
|
|
1102
|
+
return withProviderInFlightLimit(model, requestOptions, () =>
|
|
1103
|
+
streamGitLabDuo(model, context, {
|
|
1104
|
+
...requestOptions,
|
|
1105
|
+
apiKey,
|
|
1106
|
+
}),
|
|
1107
|
+
);
|
|
1108
|
+
}
|
|
1109
|
+
|
|
1110
|
+
// GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge
|
|
1111
|
+
if (model.api === "gitlab-duo-agent") {
|
|
1112
|
+
// Does not route through withProviderInFlightLimit, so heal explicitly.
|
|
1113
|
+
return wrapLeakedThinkingStream(
|
|
1114
|
+
streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
|
|
1115
|
+
...requestOptions,
|
|
1116
|
+
apiKey,
|
|
1117
|
+
}),
|
|
1118
|
+
);
|
|
1119
|
+
}
|
|
1120
|
+
|
|
1121
|
+
// Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API
|
|
1122
|
+
if (isKimiModel(model)) {
|
|
1123
|
+
// Pass raw SimpleStreamOptions - streamKimi handles mapping internally
|
|
1124
|
+
return withProviderInFlightLimit(model, requestOptions, () =>
|
|
1125
|
+
streamKimi(model as Model<"openai-completions">, context, {
|
|
1126
|
+
...requestOptions,
|
|
1127
|
+
apiKey,
|
|
1128
|
+
format: requestOptions?.kimiApiFormat ?? "anthropic",
|
|
1129
|
+
}),
|
|
1130
|
+
);
|
|
1131
|
+
}
|
|
1132
|
+
|
|
1133
|
+
// Synthetic - route to dedicated handler that wraps OpenAI or Anthropic API
|
|
1134
|
+
if (isSyntheticModel(model)) {
|
|
1135
|
+
// Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally
|
|
1136
|
+
return withProviderInFlightLimit(model, requestOptions, () =>
|
|
1137
|
+
streamSynthetic(model as Model<"openai-completions">, context, {
|
|
1138
|
+
...requestOptions,
|
|
1139
|
+
apiKey,
|
|
1140
|
+
format: requestOptions?.syntheticApiFormat ?? "openai", // Default to OpenAI format
|
|
1141
|
+
}),
|
|
1142
|
+
);
|
|
1143
|
+
}
|
|
1144
|
+
const providerOptions = mapOptionsForApi(model, requestOptions, apiKey);
|
|
1145
|
+
return stream(model, context, providerOptions);
|
|
1146
|
+
}
|
|
1147
|
+
|
|
1148
|
+
export async function completeSimple<TApi extends Api>(
|
|
1149
|
+
model: Model<TApi>,
|
|
1150
|
+
context: Context,
|
|
1151
|
+
options?: SimpleStreamOptions,
|
|
1152
|
+
): Promise<AssistantMessage> {
|
|
1153
|
+
return resolveWithThinkingLoopCook(
|
|
1154
|
+
options?.signal,
|
|
1155
|
+
() => streamSimple(model, context, options),
|
|
1156
|
+
() => streamSimple(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
|
|
1157
|
+
);
|
|
1158
|
+
}
|
|
1159
|
+
|
|
1160
|
+
const MIN_OUTPUT_TOKENS = 1024;
|
|
1161
|
+
// Fallback total output cap for models whose catalog entry has no maxTokens.
|
|
1162
|
+
const OUTPUT_CAP_WHEN_UNKNOWN = 64_000;
|
|
1163
|
+
function maxTokensWithThinkingBudget(
|
|
1164
|
+
baseMaxTokens: number | undefined,
|
|
1165
|
+
modelMaxTokens: number | null,
|
|
1166
|
+
thinkingBudget: number,
|
|
1167
|
+
): number {
|
|
1168
|
+
const uncappedMaxTokens = baseMaxTokens === undefined ? OUTPUT_CAP_WHEN_UNKNOWN : baseMaxTokens + thinkingBudget;
|
|
1169
|
+
return Math.min(uncappedMaxTokens, modelMaxTokens ?? Number.POSITIVE_INFINITY);
|
|
1170
|
+
}
|
|
1171
|
+
export const OUTPUT_FALLBACK_BUFFER = 4000;
|
|
1172
|
+
const ANTHROPIC_USE_INTERLEAVED_THINKING = Bun.env.PI_NO_INTERLEAVED_THINKING !== "1";
|
|
1173
|
+
|
|
1174
|
+
export const ANTHROPIC_THINKING: Record<Effort, number> = {
|
|
1175
|
+
minimal: 1024,
|
|
1176
|
+
low: 4096,
|
|
1177
|
+
medium: 8192,
|
|
1178
|
+
high: 16384,
|
|
1179
|
+
xhigh: 32768,
|
|
1180
|
+
};
|
|
1181
|
+
|
|
1182
|
+
const GOOGLE_THINKING: Record<Effort, number> = {
|
|
1183
|
+
minimal: 1024,
|
|
1184
|
+
low: 4096,
|
|
1185
|
+
medium: 8192,
|
|
1186
|
+
high: 16384,
|
|
1187
|
+
xhigh: 24575,
|
|
1188
|
+
};
|
|
1189
|
+
|
|
1190
|
+
const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
|
|
1191
|
+
minimal: 1024,
|
|
1192
|
+
low: 2048,
|
|
1193
|
+
medium: 8192,
|
|
1194
|
+
high: 16384,
|
|
1195
|
+
xhigh: 16384,
|
|
1196
|
+
};
|
|
1197
|
+
|
|
1198
|
+
function resolveBedrockThinkingBudget(
|
|
1199
|
+
model: Model<"bedrock-converse-stream">,
|
|
1200
|
+
options?: SimpleStreamOptions,
|
|
1201
|
+
): { budget: number; level: Effort } | null {
|
|
1202
|
+
if (!options?.reasoning || !model.reasoning) return null;
|
|
1203
|
+
const level = requireSupportedEffort(model, options.reasoning);
|
|
1204
|
+
const budget = options.thinkingBudgets?.[level] ?? BEDROCK_CLAUDE_THINKING[level];
|
|
1205
|
+
return { budget, level };
|
|
1206
|
+
}
|
|
1207
|
+
|
|
1208
|
+
export function mapAnthropicToolChoice(choice?: ToolChoice): AnthropicOptions["toolChoice"] {
|
|
1209
|
+
if (!choice) return undefined;
|
|
1210
|
+
if (typeof choice === "string") {
|
|
1211
|
+
if (choice === "required") return "any";
|
|
1212
|
+
if (choice === "auto" || choice === "none" || choice === "any") return choice;
|
|
1213
|
+
return undefined;
|
|
1214
|
+
}
|
|
1215
|
+
if (choice.type === "tool") {
|
|
1216
|
+
return choice.name ? { type: "tool", name: choice.name } : undefined;
|
|
1217
|
+
}
|
|
1218
|
+
if (choice.type === "function") {
|
|
1219
|
+
const name = "function" in choice ? choice.function?.name : choice.name;
|
|
1220
|
+
return name ? { type: "tool", name } : undefined;
|
|
1221
|
+
}
|
|
1222
|
+
return undefined;
|
|
1223
|
+
}
|
|
1224
|
+
|
|
1225
|
+
export function mapGoogleToolChoice(
|
|
1226
|
+
choice?: ToolChoice,
|
|
1227
|
+
): GoogleOptions["toolChoice"] | GoogleGeminiCliOptions["toolChoice"] | GoogleVertexOptions["toolChoice"] {
|
|
1228
|
+
if (!choice) return undefined;
|
|
1229
|
+
if (typeof choice === "string") {
|
|
1230
|
+
if (choice === "required") return "any";
|
|
1231
|
+
if (choice === "auto" || choice === "none" || choice === "any") return choice;
|
|
1232
|
+
return undefined;
|
|
1233
|
+
}
|
|
1234
|
+
// Named-tool routing on Google: emit an `ANY`-mode allow-list of one entry,
|
|
1235
|
+
// mirroring the Anthropic mapper that returns `{type: "tool", name}`.
|
|
1236
|
+
if (choice.type === "tool") {
|
|
1237
|
+
return choice.name ? { mode: "ANY", allowedFunctionNames: [choice.name] } : undefined;
|
|
1238
|
+
}
|
|
1239
|
+
if (choice.type === "function") {
|
|
1240
|
+
const name = "function" in choice ? choice.function?.name : choice.name;
|
|
1241
|
+
return name ? { mode: "ANY", allowedFunctionNames: [name] } : undefined;
|
|
1242
|
+
}
|
|
1243
|
+
return undefined;
|
|
1244
|
+
}
|
|
1245
|
+
|
|
1246
|
+
function mapOpenAiToolChoice(choice?: ToolChoice): OpenAICompletionsOptions["toolChoice"] {
|
|
1247
|
+
if (!choice) return undefined;
|
|
1248
|
+
if (typeof choice === "string") {
|
|
1249
|
+
if (choice === "any") return "required";
|
|
1250
|
+
if (choice === "auto" || choice === "none" || choice === "required") return choice;
|
|
1251
|
+
return undefined;
|
|
1252
|
+
}
|
|
1253
|
+
if (choice.type === "tool") {
|
|
1254
|
+
return choice.name ? { type: "function", function: { name: choice.name } } : undefined;
|
|
1255
|
+
}
|
|
1256
|
+
if (choice.type === "function") {
|
|
1257
|
+
const name = "function" in choice ? choice.function?.name : choice.name;
|
|
1258
|
+
return name ? { type: "function", function: { name } } : undefined;
|
|
1259
|
+
}
|
|
1260
|
+
return undefined;
|
|
1261
|
+
}
|
|
1262
|
+
|
|
1263
|
+
type ReasoningEffortMapCompat = {
|
|
1264
|
+
reasoningEffortMap?: Partial<Record<Effort, string>>;
|
|
1265
|
+
};
|
|
1266
|
+
|
|
1267
|
+
function getCompatReasoningEffortMap<TApi extends Api>(
|
|
1268
|
+
model: Model<TApi>,
|
|
1269
|
+
): Partial<Record<Effort, string>> | undefined {
|
|
1270
|
+
const compat = model.compat;
|
|
1271
|
+
if (compat === undefined || typeof compat !== "object" || !("reasoningEffortMap" in compat)) {
|
|
1272
|
+
return undefined;
|
|
1273
|
+
}
|
|
1274
|
+
return (compat as ReasoningEffortMapCompat).reasoningEffortMap;
|
|
1275
|
+
}
|
|
1276
|
+
|
|
1277
|
+
function resolveSupportedMappedReasoningEffort<TApi extends Api>(
|
|
1278
|
+
model: Model<TApi>,
|
|
1279
|
+
reasoning: Effort,
|
|
1280
|
+
): Effort | undefined {
|
|
1281
|
+
const mapped = getCompatReasoningEffortMap(model)?.[reasoning];
|
|
1282
|
+
if (!mapped) return undefined;
|
|
1283
|
+
const mappedEffort = mapped as Effort;
|
|
1284
|
+
return model.thinking?.efforts.includes(mappedEffort) ? mappedEffort : undefined;
|
|
1285
|
+
}
|
|
1286
|
+
|
|
1287
|
+
function resolveOpenAiReasoningEffort<TApi extends Api>(
|
|
1288
|
+
model: Model<TApi>,
|
|
1289
|
+
options?: SimpleStreamOptions,
|
|
1290
|
+
): Effort | undefined {
|
|
1291
|
+
const reasoning = options?.reasoning;
|
|
1292
|
+
if (!reasoning || !model.reasoning) return undefined;
|
|
1293
|
+
// Models that reason natively but expose no effort dial carry
|
|
1294
|
+
// `thinking: undefined` (baked at build time from
|
|
1295
|
+
// `compat.supportsReasoningEffort: false` on openai-responses*). The
|
|
1296
|
+
// wire-side omitReasoningEffort gate (stream.ts) is the actual strip; returning
|
|
1297
|
+
// undefined here avoids a redundant requireSupportedEffort throw that would
|
|
1298
|
+
// defeat the gate and surface a confusing "Compaction failed: Thinking effort
|
|
1299
|
+
// high is not supported by..." to the user.
|
|
1300
|
+
if (!model.thinking) return undefined;
|
|
1301
|
+
if (model.thinking.efforts.includes(reasoning)) return reasoning;
|
|
1302
|
+
const mappedReasoning = resolveSupportedMappedReasoningEffort(model, reasoning);
|
|
1303
|
+
if (mappedReasoning) return mappedReasoning;
|
|
1304
|
+
if (getCompatReasoningEffortMap(model)?.[reasoning] !== undefined) return reasoning;
|
|
1305
|
+
if (model.thinking.effortMap?.[reasoning] !== undefined) return reasoning;
|
|
1306
|
+
return requireSupportedEffort(model, reasoning);
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
const castApi = <TApi extends Api>(api: OptionsForApi<TApi>): OptionsForApi<Api> => api as OptionsForApi<Api>;
|
|
1310
|
+
|
|
1311
|
+
/**
|
|
1312
|
+
* Mandatory-reasoning endpoints (`thinking.requiresEffort`) reject disabled
|
|
1313
|
+
* or omitted thinking ("Reasoning is mandatory for this endpoint and cannot
|
|
1314
|
+
* be disabled") — clamp to the lowest supported effort instead.
|
|
1315
|
+
* `suppressWhenOff` models handle off provider-side via explicit wire
|
|
1316
|
+
* suppression. Collapsed pairs interplay: pair derivation strips member
|
|
1317
|
+
* flags (off routes to a bare SKU that CAN disable), while identity backfill
|
|
1318
|
+
* re-flags pairs whose logical id is itself mandatory (Gemini 3.x) — there
|
|
1319
|
+
* the clamp wins and the floored effort routes to the thinking SKU.
|
|
1320
|
+
*/
|
|
1321
|
+
function normalizeMandatoryReasoningOptions<TApi extends Api>(
|
|
1322
|
+
model: Model<TApi>,
|
|
1323
|
+
options?: SimpleStreamOptions,
|
|
1324
|
+
): SimpleStreamOptions | undefined {
|
|
1325
|
+
if (
|
|
1326
|
+
!model.reasoning ||
|
|
1327
|
+
!model.thinking?.requiresEffort ||
|
|
1328
|
+
model.thinking.suppressWhenOff ||
|
|
1329
|
+
(options?.reasoning !== undefined && !options.disableReasoning)
|
|
1330
|
+
) {
|
|
1331
|
+
return options;
|
|
1332
|
+
}
|
|
1333
|
+
const floor = minimumSupportedEffort(model);
|
|
1334
|
+
if (floor === undefined) return options;
|
|
1335
|
+
return { ...options, reasoning: floor, disableReasoning: undefined };
|
|
1336
|
+
}
|
|
1337
|
+
|
|
1338
|
+
function mapOptionsForApi<TApi extends Api>(
|
|
1339
|
+
model: Model<TApi>,
|
|
1340
|
+
rawOptions?: SimpleStreamOptions,
|
|
1341
|
+
apiKey?: string,
|
|
1342
|
+
): OptionsForApi<TApi> {
|
|
1343
|
+
const options = normalizeMandatoryReasoningOptions(model, rawOptions);
|
|
1344
|
+
const base = {
|
|
1345
|
+
temperature: options?.temperature,
|
|
1346
|
+
topP: options?.topP,
|
|
1347
|
+
topK: options?.topK,
|
|
1348
|
+
minP: options?.minP,
|
|
1349
|
+
presencePenalty: options?.presencePenalty,
|
|
1350
|
+
repetitionPenalty: options?.repetitionPenalty,
|
|
1351
|
+
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
|
1352
|
+
signal: options?.signal,
|
|
1353
|
+
apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined),
|
|
1354
|
+
cacheRetention: options?.cacheRetention,
|
|
1355
|
+
headers: options?.headers,
|
|
1356
|
+
initiatorOverride: options?.initiatorOverride,
|
|
1357
|
+
maxRetryDelayMs: options?.maxRetryDelayMs,
|
|
1358
|
+
metadata: options?.metadata,
|
|
1359
|
+
taskBudget: options?.taskBudget,
|
|
1360
|
+
sessionId: options?.sessionId,
|
|
1361
|
+
promptCacheKey: options?.promptCacheKey,
|
|
1362
|
+
streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
|
|
1363
|
+
streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
|
|
1364
|
+
providerSessionState: options?.providerSessionState,
|
|
1365
|
+
useInteractionsApi: options?.useInteractionsApi,
|
|
1366
|
+
storeInteraction: options?.storeInteraction,
|
|
1367
|
+
previousInteractionId: options?.previousInteractionId,
|
|
1368
|
+
maxInFlightRequests: options?.maxInFlightRequests,
|
|
1369
|
+
onPayload: options?.onPayload,
|
|
1370
|
+
onResponse: options?.onResponse,
|
|
1371
|
+
onSseEvent: options?.onSseEvent,
|
|
1372
|
+
execHandlers: options?.execHandlers,
|
|
1373
|
+
fetch: options?.fetch,
|
|
1374
|
+
fallbacks: options?.fallbacks,
|
|
1375
|
+
};
|
|
1376
|
+
|
|
1377
|
+
switch (model.api) {
|
|
1378
|
+
case "anthropic-messages": {
|
|
1379
|
+
// Explicitly disable thinking when reasoning is not specified or model doesn't support it
|
|
1380
|
+
const reasoning = options?.reasoning;
|
|
1381
|
+
if (!reasoning || !model.reasoning) {
|
|
1382
|
+
return castApi<"anthropic-messages">({
|
|
1383
|
+
...base,
|
|
1384
|
+
requestModelId: resolveWireModelId(model, undefined),
|
|
1385
|
+
thinkingEnabled: false,
|
|
1386
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1387
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1388
|
+
serviceTier: options?.serviceTier,
|
|
1389
|
+
});
|
|
1390
|
+
}
|
|
1391
|
+
|
|
1392
|
+
let thinkingBudget = options.thinkingBudgets?.[reasoning] ?? ANTHROPIC_THINKING[reasoning];
|
|
1393
|
+
if (thinkingBudget <= 0) {
|
|
1394
|
+
return castApi<"anthropic-messages">({
|
|
1395
|
+
...base,
|
|
1396
|
+
requestModelId: resolveWireModelId(model, undefined),
|
|
1397
|
+
thinkingEnabled: false,
|
|
1398
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1399
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1400
|
+
serviceTier: options?.serviceTier,
|
|
1401
|
+
});
|
|
1402
|
+
}
|
|
1403
|
+
|
|
1404
|
+
const thinkingMode = model.thinking?.mode;
|
|
1405
|
+
const effort =
|
|
1406
|
+
thinkingMode === "anthropic-adaptive" || thinkingMode === "anthropic-budget-effort"
|
|
1407
|
+
? mapEffortToAnthropicAdaptiveEffort(model, reasoning)
|
|
1408
|
+
: undefined;
|
|
1409
|
+
|
|
1410
|
+
// For Opus 4.6+ and Sonnet 4.6+: use adaptive thinking with effort level
|
|
1411
|
+
// For older models: use budget-based thinking
|
|
1412
|
+
if (thinkingMode === "anthropic-adaptive") {
|
|
1413
|
+
return castApi<"anthropic-messages">({
|
|
1414
|
+
...base,
|
|
1415
|
+
requestModelId: resolveWireModelId(model, reasoning),
|
|
1416
|
+
thinkingEnabled: true,
|
|
1417
|
+
effort,
|
|
1418
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1419
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1420
|
+
serviceTier: options?.serviceTier,
|
|
1421
|
+
});
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1424
|
+
if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
|
|
1425
|
+
return castApi<"anthropic-messages">({
|
|
1426
|
+
...base,
|
|
1427
|
+
requestModelId: resolveWireModelId(model, reasoning),
|
|
1428
|
+
thinkingEnabled: true,
|
|
1429
|
+
thinkingBudgetTokens: thinkingBudget,
|
|
1430
|
+
effort,
|
|
1431
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1432
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1433
|
+
serviceTier: options?.serviceTier,
|
|
1434
|
+
});
|
|
1435
|
+
}
|
|
1436
|
+
|
|
1437
|
+
// Caller's maxTokens is desired output, so add thinking budget on top. With no caller/model cap, use a finite total fallback.
|
|
1438
|
+
const maxTokens = maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
|
|
1439
|
+
|
|
1440
|
+
// If not enough room for thinking + output, reduce thinking budget
|
|
1441
|
+
if (maxTokens <= thinkingBudget) {
|
|
1442
|
+
thinkingBudget = maxTokens - MIN_OUTPUT_TOKENS;
|
|
1443
|
+
}
|
|
1444
|
+
|
|
1445
|
+
// If thinking budget is too low, disable thinking
|
|
1446
|
+
if (thinkingBudget <= 0) {
|
|
1447
|
+
return castApi<"anthropic-messages">({
|
|
1448
|
+
...base,
|
|
1449
|
+
requestModelId: resolveWireModelId(model, undefined),
|
|
1450
|
+
thinkingEnabled: false,
|
|
1451
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1452
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1453
|
+
serviceTier: options?.serviceTier,
|
|
1454
|
+
});
|
|
1455
|
+
} else {
|
|
1456
|
+
return castApi<"anthropic-messages">({
|
|
1457
|
+
...base,
|
|
1458
|
+
maxTokens,
|
|
1459
|
+
requestModelId: resolveWireModelId(model, reasoning),
|
|
1460
|
+
thinkingEnabled: true,
|
|
1461
|
+
thinkingBudgetTokens: thinkingBudget,
|
|
1462
|
+
effort,
|
|
1463
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1464
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1465
|
+
serviceTier: options?.serviceTier,
|
|
1466
|
+
});
|
|
1467
|
+
}
|
|
1468
|
+
}
|
|
1469
|
+
|
|
1470
|
+
case "bedrock-converse-stream": {
|
|
1471
|
+
const bedrockBase: BedrockOptions = {
|
|
1472
|
+
...base,
|
|
1473
|
+
reasoning: options?.reasoning,
|
|
1474
|
+
thinkingBudgets: options?.thinkingBudgets,
|
|
1475
|
+
toolChoice: mapAnthropicToolChoice(options?.toolChoice),
|
|
1476
|
+
thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
|
|
1477
|
+
};
|
|
1478
|
+
// Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
|
|
1479
|
+
if (model.thinking?.mode === "anthropic-adaptive") {
|
|
1480
|
+
return castApi<"bedrock-converse-stream">(bedrockBase);
|
|
1481
|
+
}
|
|
1482
|
+
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
|
|
1483
|
+
if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
|
|
1484
|
+
let maxTokens = bedrockBase.maxTokens ?? model.maxTokens ?? OUTPUT_CAP_WHEN_UNKNOWN;
|
|
1485
|
+
let thinkingBudgets = bedrockBase.thinkingBudgets;
|
|
1486
|
+
if (maxTokens <= budgetInfo.budget) {
|
|
1487
|
+
const desiredMaxTokens = Math.min(
|
|
1488
|
+
model.maxTokens ?? Number.POSITIVE_INFINITY,
|
|
1489
|
+
budgetInfo.budget + MIN_OUTPUT_TOKENS,
|
|
1490
|
+
);
|
|
1491
|
+
if (desiredMaxTokens > maxTokens) {
|
|
1492
|
+
maxTokens = desiredMaxTokens;
|
|
1493
|
+
}
|
|
1494
|
+
}
|
|
1495
|
+
if (maxTokens <= budgetInfo.budget) {
|
|
1496
|
+
const adjustedBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS);
|
|
1497
|
+
thinkingBudgets = { ...(thinkingBudgets ?? {}), [budgetInfo.level]: adjustedBudget };
|
|
1498
|
+
}
|
|
1499
|
+
return castApi<"bedrock-converse-stream">({ ...bedrockBase, maxTokens, thinkingBudgets });
|
|
1500
|
+
}
|
|
1501
|
+
|
|
1502
|
+
case "openrouter": {
|
|
1503
|
+
const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
|
|
1504
|
+
if (useResponses) {
|
|
1505
|
+
return castApi<"openai-responses">({
|
|
1506
|
+
...base,
|
|
1507
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1508
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1509
|
+
serviceTier: options?.serviceTier,
|
|
1510
|
+
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
|
1511
|
+
openrouterVariant: options?.openrouterVariant,
|
|
1512
|
+
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
|
1513
|
+
disableReasoning: options?.disableReasoning,
|
|
1514
|
+
textVerbosity: options?.textVerbosity,
|
|
1515
|
+
});
|
|
1516
|
+
}
|
|
1517
|
+
return castApi<"openai-completions">({
|
|
1518
|
+
...base,
|
|
1519
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1520
|
+
disableReasoning: options?.disableReasoning,
|
|
1521
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1522
|
+
serviceTier: options?.serviceTier,
|
|
1523
|
+
openrouterVariant: options?.openrouterVariant,
|
|
1524
|
+
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
|
1525
|
+
});
|
|
1526
|
+
}
|
|
1527
|
+
|
|
1528
|
+
case "openai-completions":
|
|
1529
|
+
return castApi<"openai-completions">({
|
|
1530
|
+
...base,
|
|
1531
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1532
|
+
disableReasoning: options?.disableReasoning,
|
|
1533
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1534
|
+
serviceTier: options?.serviceTier,
|
|
1535
|
+
openrouterVariant: options?.openrouterVariant,
|
|
1536
|
+
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
|
1537
|
+
});
|
|
1538
|
+
|
|
1539
|
+
case "openai-responses":
|
|
1540
|
+
return castApi<"openai-responses">({
|
|
1541
|
+
...base,
|
|
1542
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1543
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1544
|
+
serviceTier: options?.serviceTier,
|
|
1545
|
+
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
|
1546
|
+
openrouterVariant: options?.openrouterVariant,
|
|
1547
|
+
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
|
1548
|
+
disableReasoning: options?.disableReasoning,
|
|
1549
|
+
textVerbosity: options?.textVerbosity,
|
|
1550
|
+
});
|
|
1551
|
+
|
|
1552
|
+
case "azure-openai-responses":
|
|
1553
|
+
return castApi<"azure-openai-responses">({
|
|
1554
|
+
...base,
|
|
1555
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1556
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1557
|
+
serviceTier: options?.serviceTier,
|
|
1558
|
+
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
|
1559
|
+
});
|
|
1560
|
+
|
|
1561
|
+
case "openai-codex-responses":
|
|
1562
|
+
return castApi<"openai-codex-responses">({
|
|
1563
|
+
...base,
|
|
1564
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1565
|
+
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
|
1566
|
+
serviceTier: options?.serviceTier,
|
|
1567
|
+
preferWebsockets: options?.preferWebsockets,
|
|
1568
|
+
reasoningSummary: options?.hideThinkingSummary ? null : "detailed",
|
|
1569
|
+
textVerbosity: options?.textVerbosity,
|
|
1570
|
+
});
|
|
1571
|
+
|
|
1572
|
+
case "google-generative-ai": {
|
|
1573
|
+
// Explicitly disable thinking when reasoning is not specified or model doesn't support it
|
|
1574
|
+
// This is needed because Gemini has "dynamic thinking" enabled by default
|
|
1575
|
+
const reasoning = options?.reasoning;
|
|
1576
|
+
if (!reasoning || !model.reasoning) {
|
|
1577
|
+
return castApi<"google-generative-ai">({
|
|
1578
|
+
...base,
|
|
1579
|
+
serviceTier: options?.serviceTier,
|
|
1580
|
+
thinking: { enabled: false },
|
|
1581
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1582
|
+
});
|
|
1583
|
+
}
|
|
1584
|
+
|
|
1585
|
+
const googleModel = model as Model<"google-generative-ai">;
|
|
1586
|
+
const effort = requireSupportedEffort(googleModel, reasoning);
|
|
1587
|
+
|
|
1588
|
+
// Gemini 3+ models use thinkingLevel exclusively instead of thinkingBudget.
|
|
1589
|
+
// https://ai.google.dev/gemini-api/docs/thinking#set-budget
|
|
1590
|
+
if (googleModel.thinking?.mode === "google-level") {
|
|
1591
|
+
return castApi<"google-generative-ai">({
|
|
1592
|
+
...base,
|
|
1593
|
+
serviceTier: options?.serviceTier,
|
|
1594
|
+
thinking: {
|
|
1595
|
+
enabled: true,
|
|
1596
|
+
level: mapEffortToGoogleThinkingLevel(effort),
|
|
1597
|
+
},
|
|
1598
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1599
|
+
});
|
|
1600
|
+
}
|
|
1601
|
+
|
|
1602
|
+
return castApi<"google-gemini-cli">({
|
|
1603
|
+
...base,
|
|
1604
|
+
thinking: {
|
|
1605
|
+
enabled: true,
|
|
1606
|
+
budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets),
|
|
1607
|
+
},
|
|
1608
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1609
|
+
});
|
|
1610
|
+
}
|
|
1611
|
+
|
|
1612
|
+
case "google-gemini-cli": {
|
|
1613
|
+
const reasoning = options?.reasoning;
|
|
1614
|
+
const toolChoice = mapGoogleToolChoice(options?.toolChoice);
|
|
1615
|
+
if (reasoning && model.reasoning) {
|
|
1616
|
+
const effort = requireSupportedEffort(model, reasoning);
|
|
1617
|
+
|
|
1618
|
+
// Gemini 3+ models use thinkingLevel instead of thinkingBudget
|
|
1619
|
+
if (model.thinking?.mode === "google-level") {
|
|
1620
|
+
return castApi<"google-gemini-cli">({
|
|
1621
|
+
...base,
|
|
1622
|
+
requestModelId: resolveWireModelId(model, effort),
|
|
1623
|
+
thinking: {
|
|
1624
|
+
enabled: true,
|
|
1625
|
+
level: mapEffortToGoogleThinkingLevel(effort),
|
|
1626
|
+
},
|
|
1627
|
+
toolChoice,
|
|
1628
|
+
antigravityEndpointMode: options?.antigravityEndpointMode,
|
|
1629
|
+
});
|
|
1630
|
+
}
|
|
1631
|
+
|
|
1632
|
+
let thinkingBudget =
|
|
1633
|
+
options.thinkingBudgets?.[effort] ?? model.thinking?.effortBudgets?.[effort] ?? GOOGLE_THINKING[effort];
|
|
1634
|
+
|
|
1635
|
+
// Caller's maxTokens is desired output, so add thinking budget on top. With no caller/model cap, use a finite total fallback.
|
|
1636
|
+
const maxTokens = maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
|
|
1637
|
+
|
|
1638
|
+
// If not enough room for thinking + output, reduce thinking budget
|
|
1639
|
+
if (maxTokens <= thinkingBudget) {
|
|
1640
|
+
thinkingBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS);
|
|
1641
|
+
}
|
|
1642
|
+
|
|
1643
|
+
if (thinkingBudget > 0) {
|
|
1644
|
+
return castApi<"google-gemini-cli">({
|
|
1645
|
+
...base,
|
|
1646
|
+
maxTokens,
|
|
1647
|
+
requestModelId: resolveWireModelId(model, effort),
|
|
1648
|
+
thinking: { enabled: true, budgetTokens: thinkingBudget },
|
|
1649
|
+
toolChoice,
|
|
1650
|
+
antigravityEndpointMode: options?.antigravityEndpointMode,
|
|
1651
|
+
});
|
|
1652
|
+
}
|
|
1653
|
+
// Budget clamped to zero — fall through to the thinking-off path.
|
|
1654
|
+
}
|
|
1655
|
+
|
|
1656
|
+
const thinking: GoogleGeminiCliOptions["thinking"] = { enabled: false };
|
|
1657
|
+
if (model.reasoning && model.thinking?.suppressWhenOff) {
|
|
1658
|
+
// CCA re-applies the per-id baked server default when the config
|
|
1659
|
+
// is omitted; suppression must be explicit on the wire.
|
|
1660
|
+
thinking.suppress = model.thinking.mode === "google-level" ? { level: "MINIMAL" } : { budget: 0 };
|
|
1661
|
+
}
|
|
1662
|
+
return castApi<"google-gemini-cli">({
|
|
1663
|
+
...base,
|
|
1664
|
+
requestModelId: resolveWireModelId(model, undefined),
|
|
1665
|
+
thinking,
|
|
1666
|
+
toolChoice,
|
|
1667
|
+
antigravityEndpointMode: options?.antigravityEndpointMode,
|
|
1668
|
+
});
|
|
1669
|
+
}
|
|
1670
|
+
|
|
1671
|
+
case "google-vertex": {
|
|
1672
|
+
// Explicitly disable thinking when reasoning is not specified or model doesn't support it
|
|
1673
|
+
const reasoning = options?.reasoning;
|
|
1674
|
+
if (!reasoning || !model.reasoning) {
|
|
1675
|
+
return castApi<"google-vertex">({
|
|
1676
|
+
...base,
|
|
1677
|
+
serviceTier: options?.serviceTier,
|
|
1678
|
+
thinking: { enabled: false },
|
|
1679
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1680
|
+
});
|
|
1681
|
+
}
|
|
1682
|
+
|
|
1683
|
+
const vertexModel = model as Model<"google-vertex">;
|
|
1684
|
+
const effort = requireSupportedEffort(vertexModel, reasoning);
|
|
1685
|
+
const geminiModel = vertexModel as unknown as Model<"google-generative-ai">;
|
|
1686
|
+
|
|
1687
|
+
if (geminiModel.thinking?.mode === "google-level") {
|
|
1688
|
+
return castApi<"google-vertex">({
|
|
1689
|
+
...base,
|
|
1690
|
+
serviceTier: options?.serviceTier,
|
|
1691
|
+
thinking: {
|
|
1692
|
+
enabled: true,
|
|
1693
|
+
level: mapEffortToGoogleThinkingLevel(effort),
|
|
1694
|
+
},
|
|
1695
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1696
|
+
});
|
|
1697
|
+
}
|
|
1698
|
+
|
|
1699
|
+
return castApi<"google-vertex">({
|
|
1700
|
+
...base,
|
|
1701
|
+
serviceTier: options?.serviceTier,
|
|
1702
|
+
thinking: {
|
|
1703
|
+
enabled: true,
|
|
1704
|
+
budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
|
|
1705
|
+
},
|
|
1706
|
+
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
|
1707
|
+
});
|
|
1708
|
+
}
|
|
1709
|
+
|
|
1710
|
+
case "ollama-chat":
|
|
1711
|
+
return castApi<"ollama-chat">({
|
|
1712
|
+
...base,
|
|
1713
|
+
reasoning: resolveOpenAiReasoningEffort(model, options),
|
|
1714
|
+
disableReasoning: options?.disableReasoning,
|
|
1715
|
+
toolChoice: options?.toolChoice,
|
|
1716
|
+
});
|
|
1717
|
+
|
|
1718
|
+
case "cursor-agent": {
|
|
1719
|
+
const execHandlers = options?.cursorExecHandlers ?? options?.execHandlers;
|
|
1720
|
+
const onToolResult = options?.cursorOnToolResult ?? execHandlers?.onToolResult;
|
|
1721
|
+
return castApi<"cursor-agent">({
|
|
1722
|
+
...base,
|
|
1723
|
+
execHandlers,
|
|
1724
|
+
onToolResult,
|
|
1725
|
+
});
|
|
1726
|
+
}
|
|
1727
|
+
|
|
1728
|
+
case "gitlab-duo-agent":
|
|
1729
|
+
return castApi<"gitlab-duo-agent">({
|
|
1730
|
+
...base,
|
|
1731
|
+
cwd: options?.cwd,
|
|
1732
|
+
toolChoice: options?.toolChoice,
|
|
1733
|
+
});
|
|
1734
|
+
case "devin-agent": {
|
|
1735
|
+
const devinModel = model as Model<"devin-agent">;
|
|
1736
|
+
const effort =
|
|
1737
|
+
options?.reasoning && !options.disableReasoning
|
|
1738
|
+
? requireSupportedEffort(devinModel, options.reasoning)
|
|
1739
|
+
: undefined;
|
|
1740
|
+
return castApi<"devin-agent">({
|
|
1741
|
+
...base,
|
|
1742
|
+
chatModelUid: resolveWireModelId(devinModel, effort),
|
|
1743
|
+
});
|
|
1744
|
+
}
|
|
1745
|
+
default:
|
|
1746
|
+
throw new AIError.ConfigurationError(`Unhandled API in mapOptionsForApi: ${model.api}`);
|
|
1747
|
+
}
|
|
1748
|
+
}
|
|
1749
|
+
|
|
1750
|
+
function getGoogleBudget(
|
|
1751
|
+
model: Model<"google-generative-ai">,
|
|
1752
|
+
effort: Effort,
|
|
1753
|
+
customBudgets?: ThinkingBudgets,
|
|
1754
|
+
): number {
|
|
1755
|
+
requireSupportedEffort(model, effort);
|
|
1756
|
+
|
|
1757
|
+
// Custom budgets take precedence if provided for this level
|
|
1758
|
+
if (customBudgets?.[effort] !== undefined) {
|
|
1759
|
+
return customBudgets[effort]!;
|
|
1760
|
+
}
|
|
1761
|
+
|
|
1762
|
+
// See https://ai.google.dev/gemini-api/docs/thinking#set-budget
|
|
1763
|
+
if (model.id.includes("2.5-")) {
|
|
1764
|
+
switch (effort) {
|
|
1765
|
+
case "minimal":
|
|
1766
|
+
return 128;
|
|
1767
|
+
case "low":
|
|
1768
|
+
return 2048;
|
|
1769
|
+
case "medium":
|
|
1770
|
+
return 8192;
|
|
1771
|
+
default:
|
|
1772
|
+
return model.id.includes("2.5-flash") ? 24576 : 32768;
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1775
|
+
|
|
1776
|
+
// Unknown model - use dynamic
|
|
1777
|
+
return -1;
|
|
1778
|
+
}
|