headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,868 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenRouter client.
|
|
3
|
+
*
|
|
4
|
+
* Uses the native `fetch` API (Node >= 18) — no axios / node-fetch / openai SDK.
|
|
5
|
+
*
|
|
6
|
+
* Chat completions: POST {OPENROUTER_BASE_URL}/api/v1/chat/completions
|
|
7
|
+
* Embeddings: POST {OPENROUTER_BASE_URL}/api/v1/embeddings
|
|
8
|
+
* - Base URL override: OPENROUTER_BASE_URL env var (default
|
|
9
|
+
* https://openrouter.ai) — lets tests/proxies point the client at a mock
|
|
10
|
+
* server (e.g. http://127.0.0.1:<port>).
|
|
11
|
+
* - Authorization: Bearer <HEADLESSCODE_OPENROUTER_API_KEY>
|
|
12
|
+
* - Optional headers from env: HTTP-Referer (OPENROUTER_HTTP_REFERER),
|
|
13
|
+
* X-Title (OPENROUTER_APP_TITLE) — recommended by OpenRouter for
|
|
14
|
+
* identifying the app and enabling higher rate limits.
|
|
15
|
+
*
|
|
16
|
+
* Chat model resolution order (per request): `request.model` → constructor
|
|
17
|
+
* `defaultModel` → `OPENROUTER_MODEL` env var → `deepseek/deepseek-v4-flash-0731`.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import type {
|
|
21
|
+
LlmClient,
|
|
22
|
+
LlmRequest,
|
|
23
|
+
LlmResponse,
|
|
24
|
+
ChatMessage,
|
|
25
|
+
ChatTool,
|
|
26
|
+
ChatToolCall,
|
|
27
|
+
} from "../engine/types.js"
|
|
28
|
+
import { parseEndpointPricing, type EndpointPricingEntry, type ModelPrice } from "../budget/cost.js"
|
|
29
|
+
import { captureTranscript, isTranscriptCaptureEnabled } from "./transcript-capture.js"
|
|
30
|
+
|
|
31
|
+
export const OPENROUTER_BASE_URL = "https://openrouter.ai"
|
|
32
|
+
export const DEFAULT_MODEL = "deepseek/deepseek-v4-flash-0731"
|
|
33
|
+
|
|
34
|
+
export interface OpenRouterClientOptions {
|
|
35
|
+
apiKey?: string
|
|
36
|
+
baseUrl?: string
|
|
37
|
+
defaultModel?: string
|
|
38
|
+
httpReferer?: string
|
|
39
|
+
appTitle?: string
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** Typed error for non-2xx responses or malformed payloads. */
|
|
43
|
+
export class OpenRouterError extends Error {
|
|
44
|
+
readonly status?: number
|
|
45
|
+
readonly body?: string
|
|
46
|
+
|
|
47
|
+
constructor(message: string, status?: number, body?: string) {
|
|
48
|
+
super(message)
|
|
49
|
+
this.name = "OpenRouterError"
|
|
50
|
+
this.status = status
|
|
51
|
+
this.body = body
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Whether a `createChatCompletion` failure is worth ONE retry rather than
|
|
57
|
+
* failing the caller outright. Observed live (issue: harness provider-error
|
|
58
|
+
* resilience, 2026-08-08): a hard-pinned model (`allow_fallbacks: false`,
|
|
59
|
+
* see buildRequestBody's deepseek/* pin) has NO fallback to smooth over a
|
|
60
|
+
* blip on that one provider, so a transient hiccup that a normal
|
|
61
|
+
* multi-provider request would silently route around instead kills the
|
|
62
|
+
* request outright. Two shapes seen in one session:
|
|
63
|
+
* - HTTP 404 "No allowed providers are available for the selected model"
|
|
64
|
+
* — despite the 4xx status this is an AVAILABILITY signal (the pinned
|
|
65
|
+
* provider is temporarily down), not a real "this model/slug doesn't
|
|
66
|
+
* exist" error, which is the normal meaning of 404 elsewhere in this
|
|
67
|
+
* client (see fetchModelContextWindow's comment) — so it's retryable
|
|
68
|
+
* here specifically, by message content, not by status code alone.
|
|
69
|
+
* - HTTP 520 "Provider returned error" — an opaque upstream failure,
|
|
70
|
+
* retryable like any 5xx.
|
|
71
|
+
* Also retryable: 429 (rate limit) and network-transport failures (thrown
|
|
72
|
+
* with no `.status` — see the network-error catch below). NOT retryable:
|
|
73
|
+
* any other 4xx (auth failure, malformed request, unknown model) — those
|
|
74
|
+
* are deterministic and retrying would just fail the same way again.
|
|
75
|
+
*/
|
|
76
|
+
export function isRetryableOpenRouterError(error: unknown): boolean {
|
|
77
|
+
if (!(error instanceof OpenRouterError)) {
|
|
78
|
+
return false
|
|
79
|
+
}
|
|
80
|
+
if (error.status === undefined) {
|
|
81
|
+
return true // network/transport failure — see the catch in createChatCompletion
|
|
82
|
+
}
|
|
83
|
+
if (error.status === 429 || error.status >= 500) {
|
|
84
|
+
return true
|
|
85
|
+
}
|
|
86
|
+
return error.status === 404 && /no allowed providers/i.test(error.message)
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* One embedding vector plus the model used to produce it (OpenRouter echoes
|
|
91
|
+
* the resolved model id on the response envelope).
|
|
92
|
+
*/
|
|
93
|
+
export interface OpenRouterEmbedding {
|
|
94
|
+
model: string
|
|
95
|
+
/** One embedding per input string, in request order. */
|
|
96
|
+
embeddings: number[][]
|
|
97
|
+
/** Total prompt tokens consumed across all inputs in the request. */
|
|
98
|
+
promptTokens: number
|
|
99
|
+
/** Total tokens (prompt + completion; embeddings have no completion tokens). */
|
|
100
|
+
totalTokens: number
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* A shared HTTP helper: both the chat-completions and embeddings methods hit
|
|
105
|
+
* the same base URL with the same auth/app-identification headers, and share
|
|
106
|
+
* the same "surface the raw body, don't collapse 200-with-error-envelope into
|
|
107
|
+
* an uninformative message" error philosophy (see createChatCompletion).
|
|
108
|
+
*/
|
|
109
|
+
export class OpenRouterClient implements LlmClient {
|
|
110
|
+
private readonly apiKey: string | undefined
|
|
111
|
+
private readonly baseUrl: string
|
|
112
|
+
private readonly defaultModel: string
|
|
113
|
+
private readonly httpReferer?: string
|
|
114
|
+
private readonly appTitle?: string
|
|
115
|
+
|
|
116
|
+
constructor(options: OpenRouterClientOptions = {}) {
|
|
117
|
+
this.apiKey = options.apiKey ?? process.env.HEADLESSCODE_OPENROUTER_API_KEY
|
|
118
|
+
this.baseUrl = (options.baseUrl ?? process.env.OPENROUTER_BASE_URL ?? OPENROUTER_BASE_URL).replace(/\/+$/, "")
|
|
119
|
+
this.defaultModel = options.defaultModel ?? process.env.OPENROUTER_MODEL ?? DEFAULT_MODEL
|
|
120
|
+
this.httpReferer = options.httpReferer ?? process.env.OPENROUTER_HTTP_REFERER
|
|
121
|
+
this.appTitle = options.appTitle ?? process.env.OPENROUTER_APP_TITLE
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Resolve the model id to use for a request. The per-request model wins,
|
|
126
|
+
* then the client default (which itself falls back to env + built-in).
|
|
127
|
+
*/
|
|
128
|
+
resolveModel(requestModel?: string): string {
|
|
129
|
+
return requestModel?.trim() || this.defaultModel
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** Shared request headers: auth + the app-identification pair OpenRouter recommends. */
|
|
133
|
+
private authHeaders(): Record<string, string> {
|
|
134
|
+
const headers: Record<string, string> = {
|
|
135
|
+
Authorization: `Bearer ${this.apiKey ?? ""}`,
|
|
136
|
+
"Content-Type": "application/json",
|
|
137
|
+
}
|
|
138
|
+
if (this.httpReferer) {
|
|
139
|
+
headers["HTTP-Referer"] = this.httpReferer
|
|
140
|
+
}
|
|
141
|
+
if (this.appTitle) {
|
|
142
|
+
headers["X-Title"] = this.appTitle
|
|
143
|
+
}
|
|
144
|
+
return headers
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Embed a batch of text chunks via OpenRouter's embeddings endpoint.
|
|
149
|
+
*
|
|
150
|
+
* The API accepts an ARRAY of inputs in one call (verified live 2026-08-01:
|
|
151
|
+
* N inputs → N embeddings in a single request, with `usage.prompt_tokens`
|
|
152
|
+
* summed across all inputs), so a whole repo's chunk batch is sent as one
|
|
153
|
+
* HTTP request instead of one call per chunk. Embedding models are NOT in
|
|
154
|
+
* OpenRouter's public `/api/v1/models` listing, but the endpoint is live —
|
|
155
|
+
* see src/codesearch/embedder.ts's header comment for the full findings.
|
|
156
|
+
*
|
|
157
|
+
* `options.provider` (when given) pins routing via `extra_body.provider`
|
|
158
|
+
* (OpenRouter's `{"order": [...], "allow_fallbacks": bool}` convention,
|
|
159
|
+
* same shape buildRequestBody uses for the deepseek/* chat pin).
|
|
160
|
+
*/
|
|
161
|
+
async embed(
|
|
162
|
+
inputs: string[],
|
|
163
|
+
model: string,
|
|
164
|
+
options: { signal?: AbortSignal; provider?: { order?: string[]; allowFallbacks?: boolean } } = {},
|
|
165
|
+
): Promise<OpenRouterEmbedding> {
|
|
166
|
+
const { signal, provider } = options
|
|
167
|
+
if (!this.apiKey) {
|
|
168
|
+
throw new OpenRouterError(
|
|
169
|
+
"HEADLESSCODE_OPENROUTER_API_KEY is not set. Set the environment variable HEADLESSCODE_OPENROUTER_API_KEY (or pass apiKey to OpenRouterClient).",
|
|
170
|
+
)
|
|
171
|
+
}
|
|
172
|
+
if (inputs.length === 0) {
|
|
173
|
+
return { model, embeddings: [], promptTokens: 0, totalTokens: 0 }
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const url = `${this.baseUrl}/api/v1/embeddings`
|
|
177
|
+
const body: Record<string, unknown> = { model, input: inputs }
|
|
178
|
+
if (provider) {
|
|
179
|
+
body.provider = provider
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
let response: Response
|
|
183
|
+
try {
|
|
184
|
+
response = await fetch(url, {
|
|
185
|
+
method: "POST",
|
|
186
|
+
headers: this.authHeaders(),
|
|
187
|
+
body: JSON.stringify(body),
|
|
188
|
+
signal,
|
|
189
|
+
})
|
|
190
|
+
} catch (err) {
|
|
191
|
+
if (err instanceof Error && err.name === "AbortError") {
|
|
192
|
+
throw err // caller-managed abort
|
|
193
|
+
}
|
|
194
|
+
throw new OpenRouterError(
|
|
195
|
+
`Network error calling OpenRouter embeddings: ${err instanceof Error ? err.message : String(err)}`,
|
|
196
|
+
)
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
if (!response.ok) {
|
|
200
|
+
const rawBody = await response.text().catch(() => "")
|
|
201
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
202
|
+
throw new OpenRouterError(
|
|
203
|
+
`OpenRouter embeddings returned HTTP ${response.status}: ${excerpt}`,
|
|
204
|
+
response.status,
|
|
205
|
+
excerpt,
|
|
206
|
+
)
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// Same raw-body-first discipline as chat completions: 200 does not
|
|
210
|
+
// guarantee a well-formed data[] payload.
|
|
211
|
+
const rawBody = await response.text()
|
|
212
|
+
let data:
|
|
213
|
+
| {
|
|
214
|
+
data?: Array<{ embedding?: number[] | string }>
|
|
215
|
+
usage?: RawUsage
|
|
216
|
+
error?: { message?: string; code?: unknown }
|
|
217
|
+
}
|
|
218
|
+
| undefined
|
|
219
|
+
try {
|
|
220
|
+
data = JSON.parse(rawBody)
|
|
221
|
+
} catch {
|
|
222
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
223
|
+
throw new OpenRouterError(
|
|
224
|
+
`OpenRouter embeddings returned HTTP 200 with a non-JSON/unparseable body: ${excerpt || "(empty)"}`,
|
|
225
|
+
)
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
if (data?.error) {
|
|
229
|
+
throw new OpenRouterError(
|
|
230
|
+
`OpenRouter embeddings returned HTTP 200 with an error envelope: ${data.error.message ?? JSON.stringify(data.error)}`,
|
|
231
|
+
)
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
if (!data?.data || data.data.length !== inputs.length) {
|
|
235
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
236
|
+
throw new OpenRouterError(
|
|
237
|
+
`OpenRouter embeddings response contained ${data?.data?.length ?? 0} embeddings for ${inputs.length} inputs. Raw body: ${excerpt || "(empty)"}`,
|
|
238
|
+
)
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
const embeddings = data.data.map((item) => {
|
|
242
|
+
if (typeof item.embedding === "string") {
|
|
243
|
+
// base64-encoded float32 vector (some models/providers return this
|
|
244
|
+
// when requested or by default) — decode rather than storing a string.
|
|
245
|
+
const buf = Buffer.from(item.embedding, "base64")
|
|
246
|
+
const floats = new Float32Array(buf.buffer, buf.byteOffset, buf.byteLength / 4)
|
|
247
|
+
return Array.from(floats)
|
|
248
|
+
}
|
|
249
|
+
if (!item.embedding || !Array.isArray(item.embedding)) {
|
|
250
|
+
throw new OpenRouterError(
|
|
251
|
+
`OpenRouter embeddings response contained a data entry without a numeric embedding array. Raw body: ${excerpt(rawBody)}`,
|
|
252
|
+
)
|
|
253
|
+
}
|
|
254
|
+
return item.embedding
|
|
255
|
+
})
|
|
256
|
+
|
|
257
|
+
return {
|
|
258
|
+
model: typeof (data as { model?: unknown }).model === "string" ? (data as { model: string }).model : model,
|
|
259
|
+
embeddings,
|
|
260
|
+
promptTokens: data.usage?.prompt_tokens ?? 0,
|
|
261
|
+
totalTokens: data.usage?.total_tokens ?? data.usage?.prompt_tokens ?? 0,
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Fetch BOTH the advertised context window and per-token pricing for a
|
|
267
|
+
* model from OpenRouter's `/api/v1/models/<id>/endpoints` endpoint — ONE
|
|
268
|
+
* fetch resolves both, so a session never pays for two round-trips to the
|
|
269
|
+
* same URL (the pricing capture is Part E of the central-store round; the
|
|
270
|
+
* context-window side is the pre-existing mechanism from
|
|
271
|
+
* plans/speed-and-context-efficiency.md).
|
|
272
|
+
*
|
|
273
|
+
* Two live-verified quirks are handled here:
|
|
274
|
+
* - The `<id>` path segment is the model's `provider/model` slug with a
|
|
275
|
+
* LITERAL slash — percent-encoding the whole string
|
|
276
|
+
* (`encodeURIComponent("deepseek/deepseek-v4-flash")`) yields
|
|
277
|
+
* `deepseek%2Fdeepseek-v4-flash` which OpenRouter 404s on. Encode each
|
|
278
|
+
* segment separately and rejoin with the literal `/`.
|
|
279
|
+
* - The response body is `{"data":{"endpoints":[{...}]}}`
|
|
280
|
+
* (per-endpoint data nested under `data.endpoints`), NOT the
|
|
281
|
+
* `{"data":[...]}` array some proxies return. Accept both shapes.
|
|
282
|
+
*
|
|
283
|
+
* The largest non-zero context_length across endpoints is the window. The
|
|
284
|
+
* pricing selection is delegated to parseEndpointPricing (src/budget/
|
|
285
|
+
* cost.ts): for the deepseek/* family the OFFICIAL DeepSeek endpoint's
|
|
286
|
+
* numbers are used (that is the price actually charged — this harness
|
|
287
|
+
* pins routing to it); for everything else the per-field max across
|
|
288
|
+
* endpoints is used (fail-closed for a cost guardrail). Returns
|
|
289
|
+
* `undefined` when the endpoint is unreachable/unauthorized or carries no
|
|
290
|
+
* usable data — callers fall back to their conservative defaults.
|
|
291
|
+
*/
|
|
292
|
+
async fetchModelInfo(
|
|
293
|
+
model: string,
|
|
294
|
+
signal?: AbortSignal,
|
|
295
|
+
): Promise<{ contextWindow?: number; price?: ModelPrice } | undefined> {
|
|
296
|
+
if (!this.apiKey) {
|
|
297
|
+
return undefined
|
|
298
|
+
}
|
|
299
|
+
// Encode each slug segment separately so the `provider/model` literal
|
|
300
|
+
// slash survives (see comment above — encodeURIComponent on the whole
|
|
301
|
+
// id makes OpenRouter return 404).
|
|
302
|
+
const id = model
|
|
303
|
+
.split("/")
|
|
304
|
+
.map((segment) => encodeURIComponent(segment))
|
|
305
|
+
.join("/")
|
|
306
|
+
const url = `${this.baseUrl}/api/v1/models/${id}/endpoints`
|
|
307
|
+
|
|
308
|
+
// This call happens at most ONCE per session (loop.ts caches the
|
|
309
|
+
// result), so a single transient hiccup would silently poison the
|
|
310
|
+
// whole session with the conservative defaults. Retry transport
|
|
311
|
+
// failures and 5xx/429 once with a short backoff; a deterministic 4xx
|
|
312
|
+
// (e.g. 404 from a wrong slug) is not retried.
|
|
313
|
+
let response: Response | undefined
|
|
314
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
315
|
+
try {
|
|
316
|
+
const res = await fetch(url, {
|
|
317
|
+
method: "GET",
|
|
318
|
+
headers: this.authHeaders(),
|
|
319
|
+
signal,
|
|
320
|
+
})
|
|
321
|
+
if (res.ok) {
|
|
322
|
+
response = res
|
|
323
|
+
break
|
|
324
|
+
}
|
|
325
|
+
if (!(res.status === 429 || res.status >= 500)) {
|
|
326
|
+
return undefined
|
|
327
|
+
}
|
|
328
|
+
} catch {
|
|
329
|
+
// transport failure — retried below
|
|
330
|
+
}
|
|
331
|
+
if (attempt === 0) {
|
|
332
|
+
await new Promise((r) => setTimeout(r, 250))
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
if (!response) {
|
|
336
|
+
return undefined
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
// Live shape: { data: { id, endpoints: [{ context_length, pricing, ... }] } }.
|
|
340
|
+
// Legacy/proxy shape: { data: [{ context_length }] }.
|
|
341
|
+
let data:
|
|
342
|
+
| {
|
|
343
|
+
data?:
|
|
344
|
+
| Array<{ context_length?: unknown; provider_name?: unknown; pricing?: unknown }>
|
|
345
|
+
| { endpoints?: Array<{ context_length?: unknown; provider_name?: unknown; pricing?: unknown }> }
|
|
346
|
+
}
|
|
347
|
+
| undefined
|
|
348
|
+
try {
|
|
349
|
+
data = JSON.parse(await response.text()) as typeof data
|
|
350
|
+
} catch {
|
|
351
|
+
return undefined
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
const endpoints = Array.isArray(data?.data)
|
|
355
|
+
? data.data
|
|
356
|
+
: Array.isArray(data?.data?.endpoints)
|
|
357
|
+
? data.data.endpoints
|
|
358
|
+
: []
|
|
359
|
+
let contextWindow: number | undefined
|
|
360
|
+
for (const ep of endpoints) {
|
|
361
|
+
const n = typeof ep?.context_length === "number" ? ep.context_length : undefined
|
|
362
|
+
if (typeof n === "number" && Number.isFinite(n) && n > 0) {
|
|
363
|
+
contextWindow = contextWindow === undefined ? n : Math.max(contextWindow, n)
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
const providerPreference = model.startsWith("deepseek/") ? "DeepSeek" : undefined
|
|
367
|
+
const price = parseEndpointPricing(endpoints as EndpointPricingEntry[], providerPreference)
|
|
368
|
+
return {
|
|
369
|
+
...(contextWindow !== undefined ? { contextWindow } : {}),
|
|
370
|
+
...(price !== undefined ? { price } : {}),
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* Fetch a model's advertised context window (tokens) from OpenRouter's
|
|
376
|
+
* `/api/v1/models/<id>/endpoints` endpoint (thin wrapper over
|
|
377
|
+
* fetchModelInfo — see that method's doc for the full contract). Returns
|
|
378
|
+
* `undefined` when the lookup fails or carries no usable number; the
|
|
379
|
+
* caller falls back to `DEFAULT_CONTEXT_WINDOW_TOKENS`.
|
|
380
|
+
*/
|
|
381
|
+
async fetchModelContextWindow(model: string, signal?: AbortSignal): Promise<number | undefined> {
|
|
382
|
+
return (await this.fetchModelInfo(model, signal))?.contextWindow
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
async createChatCompletion(request: LlmRequest): Promise<LlmResponse> {
|
|
386
|
+
if (!this.apiKey) {
|
|
387
|
+
throw new OpenRouterError(
|
|
388
|
+
"HEADLESSCODE_OPENROUTER_API_KEY is not set. Set the environment variable HEADLESSCODE_OPENROUTER_API_KEY (or pass apiKey to OpenRouterClient).",
|
|
389
|
+
)
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
const model = this.resolveModel(request.model)
|
|
393
|
+
const url = `${this.baseUrl}/api/v1/chat/completions`
|
|
394
|
+
const startedAt = Date.now()
|
|
395
|
+
|
|
396
|
+
const headers = this.authHeaders()
|
|
397
|
+
|
|
398
|
+
const body = buildRequestBody(request, model)
|
|
399
|
+
|
|
400
|
+
// Opt-in SSE streaming (streaming-and-reasoning). When `request.stream`
|
|
401
|
+
// is true the response is parsed incrementally from the `data:` lines and
|
|
402
|
+
// the final message is assembled from the deltas; otherwise the legacy
|
|
403
|
+
// single blocking fetch path below runs unchanged (the default).
|
|
404
|
+
if (request.stream) {
|
|
405
|
+
return this.streamChatCompletion(request, url, headers, body)
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
try {
|
|
409
|
+
return await this.createChatCompletionNonStreaming(request, url, headers, body, model, startedAt)
|
|
410
|
+
} catch (err) {
|
|
411
|
+
if (isTranscriptCaptureEnabled()) {
|
|
412
|
+
captureTranscript(
|
|
413
|
+
{ provider: "openrouter", model },
|
|
414
|
+
{
|
|
415
|
+
messages: request.messages,
|
|
416
|
+
tools: request.tools,
|
|
417
|
+
temperature: request.temperature,
|
|
418
|
+
error: err instanceof Error ? err.message : String(err),
|
|
419
|
+
durationMs: Date.now() - startedAt,
|
|
420
|
+
},
|
|
421
|
+
)
|
|
422
|
+
}
|
|
423
|
+
throw err
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
private async createChatCompletionNonStreaming(
|
|
428
|
+
request: LlmRequest,
|
|
429
|
+
url: string,
|
|
430
|
+
headers: Record<string, string>,
|
|
431
|
+
body: Record<string, unknown>,
|
|
432
|
+
model: string,
|
|
433
|
+
startedAt: number,
|
|
434
|
+
): Promise<LlmResponse> {
|
|
435
|
+
let response: Response
|
|
436
|
+
try {
|
|
437
|
+
response = await fetch(url, {
|
|
438
|
+
method: "POST",
|
|
439
|
+
headers,
|
|
440
|
+
body: JSON.stringify(body),
|
|
441
|
+
signal: request.signal,
|
|
442
|
+
})
|
|
443
|
+
} catch (err) {
|
|
444
|
+
if (err instanceof Error && err.name === "AbortError") {
|
|
445
|
+
throw err // caller-managed abort (e.g. timeout in the loop)
|
|
446
|
+
}
|
|
447
|
+
throw new OpenRouterError(
|
|
448
|
+
`Network error calling OpenRouter: ${err instanceof Error ? err.message : String(err)}`,
|
|
449
|
+
)
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
if (!response.ok) {
|
|
453
|
+
const rawBody = await response.text().catch(() => "")
|
|
454
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
455
|
+
throw new OpenRouterError(
|
|
456
|
+
`OpenRouter returned HTTP ${response.status}: ${excerpt}`,
|
|
457
|
+
response.status,
|
|
458
|
+
excerpt,
|
|
459
|
+
)
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
// Read the raw body first (not response.json() directly): a 200 status
|
|
463
|
+
// doesn't guarantee a well-formed choices[] payload — OpenRouter/upstream
|
|
464
|
+
// providers can return HTTP 200 with an `{error: {...}}` envelope, an
|
|
465
|
+
// empty/truncated body (e.g. a gateway cutting the connection on a slow
|
|
466
|
+
// generation), or a `choices[0]` with `finish_reason` set but no
|
|
467
|
+
// `message`. Surfacing the raw body distinguishes these cases instead of
|
|
468
|
+
// collapsing them all into one uninformative "no choices[0].message".
|
|
469
|
+
const rawBody = await response.text()
|
|
470
|
+
let data:
|
|
471
|
+
| {
|
|
472
|
+
choices?: Array<{ message?: ChatMessage; finish_reason?: string }>
|
|
473
|
+
usage?: RawUsage
|
|
474
|
+
error?: { message?: string; code?: unknown }
|
|
475
|
+
}
|
|
476
|
+
| undefined
|
|
477
|
+
try {
|
|
478
|
+
data = JSON.parse(rawBody)
|
|
479
|
+
} catch {
|
|
480
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
481
|
+
throw new OpenRouterError(`OpenRouter returned HTTP 200 with a non-JSON/unparseable body: ${excerpt || "(empty)"}`)
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
if (data?.error) {
|
|
485
|
+
throw new OpenRouterError(
|
|
486
|
+
`OpenRouter returned HTTP 200 with an error envelope: ${data.error.message ?? JSON.stringify(data.error)}`,
|
|
487
|
+
)
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
const message = data?.choices?.[0]?.message
|
|
491
|
+
if (!data || !message) {
|
|
492
|
+
const finishReason = data?.choices?.[0]?.finish_reason
|
|
493
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
494
|
+
throw new OpenRouterError(
|
|
495
|
+
`OpenRouter response contained no choices[0].message` +
|
|
496
|
+
(finishReason ? ` (finish_reason: ${finishReason})` : "") +
|
|
497
|
+
`. Raw body: ${excerpt || "(empty)"}`,
|
|
498
|
+
)
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
const result: LlmResponse = {
|
|
502
|
+
message,
|
|
503
|
+
usage: mapUsage(data.usage),
|
|
504
|
+
}
|
|
505
|
+
if (isTranscriptCaptureEnabled()) {
|
|
506
|
+
captureTranscript(
|
|
507
|
+
{ provider: "openrouter", model },
|
|
508
|
+
{ messages: request.messages, tools: request.tools, temperature: request.temperature, response: result, durationMs: Date.now() - startedAt },
|
|
509
|
+
)
|
|
510
|
+
}
|
|
511
|
+
return result
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
/**
|
|
515
|
+
* Opt-in SSE streaming path (streaming-and-reasoning). `body` is the same
|
|
516
|
+
* request body the blocking path would send, plus `stream: true` and
|
|
517
|
+
* `stream_options: { include_usage: true }` (the usage chunk at the end of
|
|
518
|
+
* the stream is what makes cost accounting work — verified live 2026-08-01:
|
|
519
|
+
* the final `data:` chunk carries `usage` with the same fields as a
|
|
520
|
+
* non-streamed response).
|
|
521
|
+
*
|
|
522
|
+
* Assembles the final `LlmResponse` message from the streamed deltas so the
|
|
523
|
+
* caller sees exactly what a non-streamed equivalent response would have
|
|
524
|
+
* produced. Text, reasoning and tool calls are all reassembled (tool-call
|
|
525
|
+
* arguments arrive as partial JSON across many chunks — see the probe
|
|
526
|
+
* findings in the task notes; each chunk's `delta.tool_calls[i]` carries
|
|
527
|
+
* `index`, and the id/name arrive once on the first chunk for that index,
|
|
528
|
+
* with subsequent chunks carrying only partial `function.arguments`).
|
|
529
|
+
*
|
|
530
|
+
* The caller's AbortSignal (the loop's llmTimeoutMs timer) is passed to the
|
|
531
|
+
* fetch AND honored while reading the stream body: a mid-stream stall still
|
|
532
|
+
* aborts (the reader throws AbortError), so a stalled stream cannot hang the
|
|
533
|
+
* session past the timeout.
|
|
534
|
+
*/
|
|
535
|
+
private async streamChatCompletion(
|
|
536
|
+
request: LlmRequest,
|
|
537
|
+
url: string,
|
|
538
|
+
headers: Record<string, string>,
|
|
539
|
+
body: Record<string, unknown>,
|
|
540
|
+
): Promise<LlmResponse> {
|
|
541
|
+
const streamBody: Record<string, unknown> = {
|
|
542
|
+
...body,
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
let response: Response
|
|
546
|
+
try {
|
|
547
|
+
response = await fetch(url, {
|
|
548
|
+
method: "POST",
|
|
549
|
+
headers,
|
|
550
|
+
body: JSON.stringify(streamBody),
|
|
551
|
+
signal: request.signal,
|
|
552
|
+
})
|
|
553
|
+
} catch (err) {
|
|
554
|
+
if (err instanceof Error && err.name === "AbortError") {
|
|
555
|
+
throw err
|
|
556
|
+
}
|
|
557
|
+
throw new OpenRouterError(
|
|
558
|
+
`Network error calling OpenRouter (stream): ${err instanceof Error ? err.message : String(err)}`,
|
|
559
|
+
)
|
|
560
|
+
}
|
|
561
|
+
|
|
562
|
+
if (!response.ok) {
|
|
563
|
+
const rawBody = await response.text().catch(() => "")
|
|
564
|
+
const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
|
|
565
|
+
throw new OpenRouterError(
|
|
566
|
+
`OpenRouter returned HTTP ${response.status}: ${excerpt}`,
|
|
567
|
+
response.status,
|
|
568
|
+
excerpt,
|
|
569
|
+
)
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
if (!response.body) {
|
|
573
|
+
throw new OpenRouterError("OpenRouter returned a stream response with no body")
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
const reader = response.body.getReader()
|
|
577
|
+
const decoder = new TextDecoder()
|
|
578
|
+
let buffer = ""
|
|
579
|
+
let done = false
|
|
580
|
+
|
|
581
|
+
// Final assembled message. `reasoning` concatenates `delta.reasoning`
|
|
582
|
+
// strings; `content` concatenates `delta.content`; tool calls are keyed
|
|
583
|
+
// by their stream index and merged after the stream ends.
|
|
584
|
+
const contentParts: string[] = []
|
|
585
|
+
const reasoningParts: string[] = []
|
|
586
|
+
const toolCallsByIdx = new Map<number, { id: string; name: string; args: string[] }>()
|
|
587
|
+
let sawFinish = false
|
|
588
|
+
let sawUsage = false
|
|
589
|
+
let lastUsage: RawUsage | undefined
|
|
590
|
+
|
|
591
|
+
const fail = (message: string): never => {
|
|
592
|
+
throw new OpenRouterError(message)
|
|
593
|
+
}
|
|
594
|
+
|
|
595
|
+
// Fire a live-typing chunk to the caller's callback. Non-fatal by
|
|
596
|
+
// contract (types.ts): a throw from onStreamChunk must never fail the
|
|
597
|
+
// LLM call — the event feed is best-effort.
|
|
598
|
+
const emitChunk = (kind: "text" | "reasoning" | "tool", chunk: string): void => {
|
|
599
|
+
if (!request.onStreamChunk || chunk.length === 0) {
|
|
600
|
+
return
|
|
601
|
+
}
|
|
602
|
+
try {
|
|
603
|
+
request.onStreamChunk(kind, chunk)
|
|
604
|
+
} catch {
|
|
605
|
+
// Swallow: the dashboard's live-typing feed is best-effort.
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
while (!done) {
|
|
610
|
+
// Structural type: the lib is ES2022 (no DOM), so the undici
|
|
611
|
+
// ReadableStream read-result shape is spelled out rather than
|
|
612
|
+
// referenced by name.
|
|
613
|
+
let chunk: { done: boolean; value?: Uint8Array }
|
|
614
|
+
try {
|
|
615
|
+
chunk = await reader.read()
|
|
616
|
+
} catch (err) {
|
|
617
|
+
if (err instanceof Error && err.name === "AbortError") {
|
|
618
|
+
throw err
|
|
619
|
+
}
|
|
620
|
+
throw new OpenRouterError(
|
|
621
|
+
`Error reading OpenRouter stream: ${err instanceof Error ? err.message : String(err)}`,
|
|
622
|
+
)
|
|
623
|
+
}
|
|
624
|
+
if (chunk.done) {
|
|
625
|
+
done = true
|
|
626
|
+
break
|
|
627
|
+
}
|
|
628
|
+
buffer += decoder.decode(chunk.value, { stream: true })
|
|
629
|
+
const lines = buffer.split("\n")
|
|
630
|
+
buffer = lines.pop() ?? ""
|
|
631
|
+
|
|
632
|
+
for (const line of lines) {
|
|
633
|
+
const trimmed = line.trim()
|
|
634
|
+
if (!trimmed || !trimmed.startsWith("data:")) {
|
|
635
|
+
continue
|
|
636
|
+
}
|
|
637
|
+
const payload = trimmed.slice(5).trim()
|
|
638
|
+
if (payload === "[DONE]") {
|
|
639
|
+
done = true
|
|
640
|
+
break
|
|
641
|
+
}
|
|
642
|
+
if (payload === "") {
|
|
643
|
+
continue
|
|
644
|
+
}
|
|
645
|
+
let data:
|
|
646
|
+
| {
|
|
647
|
+
choices?: Array<{ delta?: { content?: string; reasoning?: string; tool_calls?: StreamToolCallDelta[] }; finish_reason?: string }>
|
|
648
|
+
usage?: RawUsage
|
|
649
|
+
error?: { message?: string }
|
|
650
|
+
}
|
|
651
|
+
| undefined
|
|
652
|
+
try {
|
|
653
|
+
data = JSON.parse(payload) as typeof data
|
|
654
|
+
} catch {
|
|
655
|
+
fail(`OpenRouter stream contained a non-JSON data line: ${payload.slice(0, 200)}`)
|
|
656
|
+
}
|
|
657
|
+
if (data?.error) {
|
|
658
|
+
fail(`OpenRouter stream error envelope: ${data.error.message ?? JSON.stringify(data.error)}`)
|
|
659
|
+
}
|
|
660
|
+
const choice = data?.choices?.[0]
|
|
661
|
+
if (choice?.finish_reason) {
|
|
662
|
+
sawFinish = true
|
|
663
|
+
}
|
|
664
|
+
const delta = choice?.delta
|
|
665
|
+
if (delta) {
|
|
666
|
+
if (typeof delta.content === "string") {
|
|
667
|
+
contentParts.push(delta.content)
|
|
668
|
+
emitChunk("text", delta.content)
|
|
669
|
+
}
|
|
670
|
+
if (typeof delta.reasoning === "string") {
|
|
671
|
+
reasoningParts.push(delta.reasoning)
|
|
672
|
+
emitChunk("reasoning", delta.reasoning)
|
|
673
|
+
}
|
|
674
|
+
if (Array.isArray(delta.tool_calls)) {
|
|
675
|
+
for (const tc of delta.tool_calls) {
|
|
676
|
+
const entry = toolCallsByIdx.get(tc.index) ?? { id: "", name: "", args: [] }
|
|
677
|
+
if (typeof tc.id === "string" && tc.id) {
|
|
678
|
+
entry.id = tc.id
|
|
679
|
+
}
|
|
680
|
+
if (typeof tc.function?.name === "string" && tc.function.name) {
|
|
681
|
+
entry.name = tc.function.name
|
|
682
|
+
}
|
|
683
|
+
if (typeof tc.function?.arguments === "string" && tc.function.arguments) {
|
|
684
|
+
entry.args.push(tc.function.arguments)
|
|
685
|
+
emitChunk("tool", tc.function.arguments)
|
|
686
|
+
}
|
|
687
|
+
toolCallsByIdx.set(tc.index, entry)
|
|
688
|
+
}
|
|
689
|
+
}
|
|
690
|
+
}
|
|
691
|
+
if (data?.usage) {
|
|
692
|
+
sawUsage = true
|
|
693
|
+
lastUsage = data.usage
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
const content = contentParts.join("")
|
|
699
|
+
const reasoning = reasoningParts.join("")
|
|
700
|
+
const toolCalls: ChatToolCall[] = [...toolCallsByIdx.entries()]
|
|
701
|
+
.sort(([a], [b]) => a - b)
|
|
702
|
+
.map(([, tc]) => ({
|
|
703
|
+
id: tc.id || `call_stream_${tc.name || "unknown"}`,
|
|
704
|
+
type: "function" as const,
|
|
705
|
+
function: { name: tc.name || "unknown_tool", arguments: tc.args.join("") },
|
|
706
|
+
}))
|
|
707
|
+
|
|
708
|
+
const message: ChatMessage = {
|
|
709
|
+
role: "assistant",
|
|
710
|
+
content: content === "" ? null : content,
|
|
711
|
+
...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
|
|
712
|
+
...(reasoning !== "" ? { reasoning } : {}),
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
if (!sawFinish && !toolCalls.length && content === "" && reasoning === "" && !sawUsage) {
|
|
716
|
+
// A stream that produced nothing at all (e.g. gateway cut) should
|
|
717
|
+
// surface as an error, not a silent empty assistant turn.
|
|
718
|
+
fail("OpenRouter stream ended with no content, reasoning, tool calls or usage")
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
return {
|
|
722
|
+
message,
|
|
723
|
+
usage: mapUsage(lastUsage),
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
/** A single `delta.tool_calls[]` entry as OpenRouter streams it. */
|
|
729
|
+
interface StreamToolCallDelta {
|
|
730
|
+
index: number
|
|
731
|
+
id?: string
|
|
732
|
+
function?: { name?: string; arguments?: string }
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
/**
|
|
736
|
+
* User-facing reasoning-effort levels for deepseek/* models (issue #30
|
|
737
|
+
* experiment). DeepSeek's native graded set is low/medium/high/max; OpenRouter's
|
|
738
|
+
* normalized `reasoning.effort` accepts "high" and "xhigh" (xhigh maps to the
|
|
739
|
+
* native max). "xhigh" is accepted as an alias for "max" so both vocabularies
|
|
740
|
+
* work. Anything else fails loudly — a typo must never silently keep the
|
|
741
|
+
* endpoint's undeclared default, which would poison the experiment's comparison.
|
|
742
|
+
*/
|
|
743
|
+
export const REASONING_EFFORT_LEVELS = ["low", "medium", "high", "max", "xhigh"] as const
|
|
744
|
+
|
|
745
|
+
/** The OpenRouter-normalized `reasoning.effort` values actually sent on the wire. */
|
|
746
|
+
export type WireReasoningEffort = "low" | "medium" | "high" | "xhigh"
|
|
747
|
+
|
|
748
|
+
/**
|
|
749
|
+
* Validate + normalize a configured reasoning-effort value to the
|
|
750
|
+
* OpenRouter-normalized wire value. `undefined`/empty means "no effort sent"
|
|
751
|
+
* (the endpoint's default applies). Throws on anything outside the known set.
|
|
752
|
+
*/
|
|
753
|
+
export function parseReasoningEffort(value: string | undefined): WireReasoningEffort | undefined {
|
|
754
|
+
if (value === undefined || value.trim() === "") {
|
|
755
|
+
return undefined
|
|
756
|
+
}
|
|
757
|
+
switch (value.trim().toLowerCase()) {
|
|
758
|
+
case "low":
|
|
759
|
+
return "low"
|
|
760
|
+
case "medium":
|
|
761
|
+
return "medium"
|
|
762
|
+
case "high":
|
|
763
|
+
return "high"
|
|
764
|
+
case "max":
|
|
765
|
+
case "xhigh":
|
|
766
|
+
// OpenRouter's normalized reasoning.effort has no "max" — xhigh is
|
|
767
|
+
// its alias for the native max level.
|
|
768
|
+
return "xhigh"
|
|
769
|
+
default:
|
|
770
|
+
throw new Error(
|
|
771
|
+
`reasoning effort must be one of ${REASONING_EFFORT_LEVELS.join("/")} (native DeepSeek levels, plus OpenRouter's normalized "xhigh" alias for "max"), got '${value}'`,
|
|
772
|
+
)
|
|
773
|
+
}
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
export function buildRequestBody(request: LlmRequest, model: string): Record<string, unknown> {
|
|
777
|
+
const body: Record<string, unknown> = {
|
|
778
|
+
model,
|
|
779
|
+
messages: request.messages,
|
|
780
|
+
}
|
|
781
|
+
if (request.tools && request.tools.length > 0) {
|
|
782
|
+
body.tools = request.tools satisfies ChatTool[]
|
|
783
|
+
}
|
|
784
|
+
if (request.temperature !== undefined) {
|
|
785
|
+
body.temperature = request.temperature
|
|
786
|
+
}
|
|
787
|
+
if (request.maxTokens !== undefined) {
|
|
788
|
+
body.max_tokens = request.maxTokens
|
|
789
|
+
}
|
|
790
|
+
// Opt-in SSE streaming (streaming-and-reasoning): `stream: true` makes
|
|
791
|
+
// OpenRouter return a text/event-stream instead of one JSON body, and
|
|
792
|
+
// `stream_options.include_usage: true` puts the final usage chunk (with
|
|
793
|
+
// prompt/completion/reasoning token counts) at the end of the stream —
|
|
794
|
+
// without it a streamed response carries NO usage data and cost
|
|
795
|
+
// accounting would silently under-report. Both are OpenRouter's own
|
|
796
|
+
// OpenAI-compatible conventions, verified live 2026-08-01.
|
|
797
|
+
if (request.stream) {
|
|
798
|
+
body.stream = true
|
|
799
|
+
body.stream_options = { include_usage: true }
|
|
800
|
+
}
|
|
801
|
+
// Pin routing to DeepSeek's own official endpoint for deepseek/* models,
|
|
802
|
+
// not whichever third-party host (DeepInfra, Baidu, Mancer, ...) OpenRouter's
|
|
803
|
+
// automatic price-based routing might otherwise pick behind the same model
|
|
804
|
+
// id. This matters beyond preference: those endpoints have meaningfully
|
|
805
|
+
// different prices — especially cache-read pricing, where the official
|
|
806
|
+
// endpoint's rate is roughly 10x cheaper than third-party hosts serving the
|
|
807
|
+
// same model — so DEFAULT_PRICING_TABLE's deepseek/* entries
|
|
808
|
+
// (src/budget/cost.ts) are only actually correct WITH this pin in place.
|
|
809
|
+
// `allow_fallbacks: false` means a request fails clearly if the official
|
|
810
|
+
// endpoint is down, rather than silently routing elsewhere at a different
|
|
811
|
+
// (unpriced-by-us) rate.
|
|
812
|
+
if (model.startsWith("deepseek/")) {
|
|
813
|
+
body.provider = { order: ["deepseek"], allow_fallbacks: false }
|
|
814
|
+
// Reasoning content (streaming-and-reasoning): request it explicitly for
|
|
815
|
+
// the only model family this harness pins. Verified live 2026-08-01 on
|
|
816
|
+
// deepseek/deepseek-v4-flash: `include_reasoning: true` is in every
|
|
817
|
+
// endpoint's `supported_parameters`; WITHOUT it the pinned official
|
|
818
|
+
// endpoint still reasons (the model thinks anyway) but `include_reasoning:
|
|
819
|
+
// false` suppresses the returned `reasoning` field — so this flag is
|
|
820
|
+
// load-bearing for actually getting the thinking text back. OpenRouter
|
|
821
|
+
// cleanly ignores the param for models that don't support it (probed with
|
|
822
|
+
// 200s on gpt-4o-mini/claude-3.7-sonnet when the model id resolves; the
|
|
823
|
+
// 404s seen were account/data-policy endpoint availability, not the
|
|
824
|
+
// param). Kept to the deepseek/ prefix rather than unconditional to match
|
|
825
|
+
// the existing pin boundary.
|
|
826
|
+
body.include_reasoning = true
|
|
827
|
+
// Graded reasoning effort (issue #30): DeepSeek's native low/medium/
|
|
828
|
+
// high/max is normalized by OpenRouter to `reasoning: { effort: "high" |
|
|
829
|
+
// "xhigh" }` ("xhigh" = max). Unset = send nothing = the endpoint's
|
|
830
|
+
// undeclared default, exactly as before. Same deepseek/ boundary as
|
|
831
|
+
// include_reasoning.
|
|
832
|
+
if (request.reasoningEffort !== undefined) {
|
|
833
|
+
const effort = parseReasoningEffort(request.reasoningEffort)
|
|
834
|
+
if (effort !== undefined) {
|
|
835
|
+
body.reasoning = { effort }
|
|
836
|
+
}
|
|
837
|
+
}
|
|
838
|
+
}
|
|
839
|
+
return body
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
interface RawUsage {
|
|
843
|
+
prompt_tokens?: number
|
|
844
|
+
completion_tokens?: number
|
|
845
|
+
total_tokens?: number
|
|
846
|
+
/** OpenRouter's pass-through of the underlying provider's cache accounting. */
|
|
847
|
+
prompt_tokens_details?: {
|
|
848
|
+
cached_tokens?: number
|
|
849
|
+
}
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
function mapUsage(raw: RawUsage | undefined): LlmResponse["usage"] {
|
|
853
|
+
if (!raw) {
|
|
854
|
+
return undefined
|
|
855
|
+
}
|
|
856
|
+
return {
|
|
857
|
+
promptTokens: raw.prompt_tokens,
|
|
858
|
+
completionTokens: raw.completion_tokens,
|
|
859
|
+
totalTokens: raw.total_tokens,
|
|
860
|
+
cachedTokens: raw.prompt_tokens_details?.cached_tokens,
|
|
861
|
+
}
|
|
862
|
+
}
|
|
863
|
+
|
|
864
|
+
/** Truncate a raw body to a bounded excerpt for error messages. */
|
|
865
|
+
function excerpt(body: string): string {
|
|
866
|
+
return body.length > 500 ? `${body.slice(0, 500)}…` : body
|
|
867
|
+
}
|
|
868
|
+
|