headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,868 @@
1
+ /**
2
+ * OpenRouter client.
3
+ *
4
+ * Uses the native `fetch` API (Node >= 18) — no axios / node-fetch / openai SDK.
5
+ *
6
+ * Chat completions: POST {OPENROUTER_BASE_URL}/api/v1/chat/completions
7
+ * Embeddings: POST {OPENROUTER_BASE_URL}/api/v1/embeddings
8
+ * - Base URL override: OPENROUTER_BASE_URL env var (default
9
+ * https://openrouter.ai) — lets tests/proxies point the client at a mock
10
+ * server (e.g. http://127.0.0.1:<port>).
11
+ * - Authorization: Bearer <HEADLESSCODE_OPENROUTER_API_KEY>
12
+ * - Optional headers from env: HTTP-Referer (OPENROUTER_HTTP_REFERER),
13
+ * X-Title (OPENROUTER_APP_TITLE) — recommended by OpenRouter for
14
+ * identifying the app and enabling higher rate limits.
15
+ *
16
+ * Chat model resolution order (per request): `request.model` → constructor
17
+ * `defaultModel` → `OPENROUTER_MODEL` env var → `deepseek/deepseek-v4-flash-0731`.
18
+ */
19
+
20
+ import type {
21
+ LlmClient,
22
+ LlmRequest,
23
+ LlmResponse,
24
+ ChatMessage,
25
+ ChatTool,
26
+ ChatToolCall,
27
+ } from "../engine/types.js"
28
+ import { parseEndpointPricing, type EndpointPricingEntry, type ModelPrice } from "../budget/cost.js"
29
+ import { captureTranscript, isTranscriptCaptureEnabled } from "./transcript-capture.js"
30
+
31
+ export const OPENROUTER_BASE_URL = "https://openrouter.ai"
32
+ export const DEFAULT_MODEL = "deepseek/deepseek-v4-flash-0731"
33
+
34
+ export interface OpenRouterClientOptions {
35
+ apiKey?: string
36
+ baseUrl?: string
37
+ defaultModel?: string
38
+ httpReferer?: string
39
+ appTitle?: string
40
+ }
41
+
42
+ /** Typed error for non-2xx responses or malformed payloads. */
43
+ export class OpenRouterError extends Error {
44
+ readonly status?: number
45
+ readonly body?: string
46
+
47
+ constructor(message: string, status?: number, body?: string) {
48
+ super(message)
49
+ this.name = "OpenRouterError"
50
+ this.status = status
51
+ this.body = body
52
+ }
53
+ }
54
+
55
+ /**
56
+ * Whether a `createChatCompletion` failure is worth ONE retry rather than
57
+ * failing the caller outright. Observed live (issue: harness provider-error
58
+ * resilience, 2026-08-08): a hard-pinned model (`allow_fallbacks: false`,
59
+ * see buildRequestBody's deepseek/* pin) has NO fallback to smooth over a
60
+ * blip on that one provider, so a transient hiccup that a normal
61
+ * multi-provider request would silently route around instead kills the
62
+ * request outright. Two shapes seen in one session:
63
+ * - HTTP 404 "No allowed providers are available for the selected model"
64
+ * — despite the 4xx status this is an AVAILABILITY signal (the pinned
65
+ * provider is temporarily down), not a real "this model/slug doesn't
66
+ * exist" error, which is the normal meaning of 404 elsewhere in this
67
+ * client (see fetchModelContextWindow's comment) — so it's retryable
68
+ * here specifically, by message content, not by status code alone.
69
+ * - HTTP 520 "Provider returned error" — an opaque upstream failure,
70
+ * retryable like any 5xx.
71
+ * Also retryable: 429 (rate limit) and network-transport failures (thrown
72
+ * with no `.status` — see the network-error catch below). NOT retryable:
73
+ * any other 4xx (auth failure, malformed request, unknown model) — those
74
+ * are deterministic and retrying would just fail the same way again.
75
+ */
76
+ export function isRetryableOpenRouterError(error: unknown): boolean {
77
+ if (!(error instanceof OpenRouterError)) {
78
+ return false
79
+ }
80
+ if (error.status === undefined) {
81
+ return true // network/transport failure — see the catch in createChatCompletion
82
+ }
83
+ if (error.status === 429 || error.status >= 500) {
84
+ return true
85
+ }
86
+ return error.status === 404 && /no allowed providers/i.test(error.message)
87
+ }
88
+
89
+ /**
90
+ * One embedding vector plus the model used to produce it (OpenRouter echoes
91
+ * the resolved model id on the response envelope).
92
+ */
93
+ export interface OpenRouterEmbedding {
94
+ model: string
95
+ /** One embedding per input string, in request order. */
96
+ embeddings: number[][]
97
+ /** Total prompt tokens consumed across all inputs in the request. */
98
+ promptTokens: number
99
+ /** Total tokens (prompt + completion; embeddings have no completion tokens). */
100
+ totalTokens: number
101
+ }
102
+
103
+ /**
104
+ * A shared HTTP helper: both the chat-completions and embeddings methods hit
105
+ * the same base URL with the same auth/app-identification headers, and share
106
+ * the same "surface the raw body, don't collapse 200-with-error-envelope into
107
+ * an uninformative message" error philosophy (see createChatCompletion).
108
+ */
109
+ export class OpenRouterClient implements LlmClient {
110
+ private readonly apiKey: string | undefined
111
+ private readonly baseUrl: string
112
+ private readonly defaultModel: string
113
+ private readonly httpReferer?: string
114
+ private readonly appTitle?: string
115
+
116
+ constructor(options: OpenRouterClientOptions = {}) {
117
+ this.apiKey = options.apiKey ?? process.env.HEADLESSCODE_OPENROUTER_API_KEY
118
+ this.baseUrl = (options.baseUrl ?? process.env.OPENROUTER_BASE_URL ?? OPENROUTER_BASE_URL).replace(/\/+$/, "")
119
+ this.defaultModel = options.defaultModel ?? process.env.OPENROUTER_MODEL ?? DEFAULT_MODEL
120
+ this.httpReferer = options.httpReferer ?? process.env.OPENROUTER_HTTP_REFERER
121
+ this.appTitle = options.appTitle ?? process.env.OPENROUTER_APP_TITLE
122
+ }
123
+
124
+ /**
125
+ * Resolve the model id to use for a request. The per-request model wins,
126
+ * then the client default (which itself falls back to env + built-in).
127
+ */
128
+ resolveModel(requestModel?: string): string {
129
+ return requestModel?.trim() || this.defaultModel
130
+ }
131
+
132
+ /** Shared request headers: auth + the app-identification pair OpenRouter recommends. */
133
+ private authHeaders(): Record<string, string> {
134
+ const headers: Record<string, string> = {
135
+ Authorization: `Bearer ${this.apiKey ?? ""}`,
136
+ "Content-Type": "application/json",
137
+ }
138
+ if (this.httpReferer) {
139
+ headers["HTTP-Referer"] = this.httpReferer
140
+ }
141
+ if (this.appTitle) {
142
+ headers["X-Title"] = this.appTitle
143
+ }
144
+ return headers
145
+ }
146
+
147
+ /**
148
+ * Embed a batch of text chunks via OpenRouter's embeddings endpoint.
149
+ *
150
+ * The API accepts an ARRAY of inputs in one call (verified live 2026-08-01:
151
+ * N inputs → N embeddings in a single request, with `usage.prompt_tokens`
152
+ * summed across all inputs), so a whole repo's chunk batch is sent as one
153
+ * HTTP request instead of one call per chunk. Embedding models are NOT in
154
+ * OpenRouter's public `/api/v1/models` listing, but the endpoint is live —
155
+ * see src/codesearch/embedder.ts's header comment for the full findings.
156
+ *
157
+ * `options.provider` (when given) pins routing via `extra_body.provider`
158
+ * (OpenRouter's `{"order": [...], "allow_fallbacks": bool}` convention,
159
+ * same shape buildRequestBody uses for the deepseek/* chat pin).
160
+ */
161
+ async embed(
162
+ inputs: string[],
163
+ model: string,
164
+ options: { signal?: AbortSignal; provider?: { order?: string[]; allowFallbacks?: boolean } } = {},
165
+ ): Promise<OpenRouterEmbedding> {
166
+ const { signal, provider } = options
167
+ if (!this.apiKey) {
168
+ throw new OpenRouterError(
169
+ "HEADLESSCODE_OPENROUTER_API_KEY is not set. Set the environment variable HEADLESSCODE_OPENROUTER_API_KEY (or pass apiKey to OpenRouterClient).",
170
+ )
171
+ }
172
+ if (inputs.length === 0) {
173
+ return { model, embeddings: [], promptTokens: 0, totalTokens: 0 }
174
+ }
175
+
176
+ const url = `${this.baseUrl}/api/v1/embeddings`
177
+ const body: Record<string, unknown> = { model, input: inputs }
178
+ if (provider) {
179
+ body.provider = provider
180
+ }
181
+
182
+ let response: Response
183
+ try {
184
+ response = await fetch(url, {
185
+ method: "POST",
186
+ headers: this.authHeaders(),
187
+ body: JSON.stringify(body),
188
+ signal,
189
+ })
190
+ } catch (err) {
191
+ if (err instanceof Error && err.name === "AbortError") {
192
+ throw err // caller-managed abort
193
+ }
194
+ throw new OpenRouterError(
195
+ `Network error calling OpenRouter embeddings: ${err instanceof Error ? err.message : String(err)}`,
196
+ )
197
+ }
198
+
199
+ if (!response.ok) {
200
+ const rawBody = await response.text().catch(() => "")
201
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
202
+ throw new OpenRouterError(
203
+ `OpenRouter embeddings returned HTTP ${response.status}: ${excerpt}`,
204
+ response.status,
205
+ excerpt,
206
+ )
207
+ }
208
+
209
+ // Same raw-body-first discipline as chat completions: 200 does not
210
+ // guarantee a well-formed data[] payload.
211
+ const rawBody = await response.text()
212
+ let data:
213
+ | {
214
+ data?: Array<{ embedding?: number[] | string }>
215
+ usage?: RawUsage
216
+ error?: { message?: string; code?: unknown }
217
+ }
218
+ | undefined
219
+ try {
220
+ data = JSON.parse(rawBody)
221
+ } catch {
222
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
223
+ throw new OpenRouterError(
224
+ `OpenRouter embeddings returned HTTP 200 with a non-JSON/unparseable body: ${excerpt || "(empty)"}`,
225
+ )
226
+ }
227
+
228
+ if (data?.error) {
229
+ throw new OpenRouterError(
230
+ `OpenRouter embeddings returned HTTP 200 with an error envelope: ${data.error.message ?? JSON.stringify(data.error)}`,
231
+ )
232
+ }
233
+
234
+ if (!data?.data || data.data.length !== inputs.length) {
235
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
236
+ throw new OpenRouterError(
237
+ `OpenRouter embeddings response contained ${data?.data?.length ?? 0} embeddings for ${inputs.length} inputs. Raw body: ${excerpt || "(empty)"}`,
238
+ )
239
+ }
240
+
241
+ const embeddings = data.data.map((item) => {
242
+ if (typeof item.embedding === "string") {
243
+ // base64-encoded float32 vector (some models/providers return this
244
+ // when requested or by default) — decode rather than storing a string.
245
+ const buf = Buffer.from(item.embedding, "base64")
246
+ const floats = new Float32Array(buf.buffer, buf.byteOffset, buf.byteLength / 4)
247
+ return Array.from(floats)
248
+ }
249
+ if (!item.embedding || !Array.isArray(item.embedding)) {
250
+ throw new OpenRouterError(
251
+ `OpenRouter embeddings response contained a data entry without a numeric embedding array. Raw body: ${excerpt(rawBody)}`,
252
+ )
253
+ }
254
+ return item.embedding
255
+ })
256
+
257
+ return {
258
+ model: typeof (data as { model?: unknown }).model === "string" ? (data as { model: string }).model : model,
259
+ embeddings,
260
+ promptTokens: data.usage?.prompt_tokens ?? 0,
261
+ totalTokens: data.usage?.total_tokens ?? data.usage?.prompt_tokens ?? 0,
262
+ }
263
+ }
264
+
265
+ /**
266
+ * Fetch BOTH the advertised context window and per-token pricing for a
267
+ * model from OpenRouter's `/api/v1/models/<id>/endpoints` endpoint — ONE
268
+ * fetch resolves both, so a session never pays for two round-trips to the
269
+ * same URL (the pricing capture is Part E of the central-store round; the
270
+ * context-window side is the pre-existing mechanism from
271
+ * plans/speed-and-context-efficiency.md).
272
+ *
273
+ * Two live-verified quirks are handled here:
274
+ * - The `<id>` path segment is the model's `provider/model` slug with a
275
+ * LITERAL slash — percent-encoding the whole string
276
+ * (`encodeURIComponent("deepseek/deepseek-v4-flash")`) yields
277
+ * `deepseek%2Fdeepseek-v4-flash` which OpenRouter 404s on. Encode each
278
+ * segment separately and rejoin with the literal `/`.
279
+ * - The response body is `{"data":{"endpoints":[{...}]}}`
280
+ * (per-endpoint data nested under `data.endpoints`), NOT the
281
+ * `{"data":[...]}` array some proxies return. Accept both shapes.
282
+ *
283
+ * The largest non-zero context_length across endpoints is the window. The
284
+ * pricing selection is delegated to parseEndpointPricing (src/budget/
285
+ * cost.ts): for the deepseek/* family the OFFICIAL DeepSeek endpoint's
286
+ * numbers are used (that is the price actually charged — this harness
287
+ * pins routing to it); for everything else the per-field max across
288
+ * endpoints is used (fail-closed for a cost guardrail). Returns
289
+ * `undefined` when the endpoint is unreachable/unauthorized or carries no
290
+ * usable data — callers fall back to their conservative defaults.
291
+ */
292
+ async fetchModelInfo(
293
+ model: string,
294
+ signal?: AbortSignal,
295
+ ): Promise<{ contextWindow?: number; price?: ModelPrice } | undefined> {
296
+ if (!this.apiKey) {
297
+ return undefined
298
+ }
299
+ // Encode each slug segment separately so the `provider/model` literal
300
+ // slash survives (see comment above — encodeURIComponent on the whole
301
+ // id makes OpenRouter return 404).
302
+ const id = model
303
+ .split("/")
304
+ .map((segment) => encodeURIComponent(segment))
305
+ .join("/")
306
+ const url = `${this.baseUrl}/api/v1/models/${id}/endpoints`
307
+
308
+ // This call happens at most ONCE per session (loop.ts caches the
309
+ // result), so a single transient hiccup would silently poison the
310
+ // whole session with the conservative defaults. Retry transport
311
+ // failures and 5xx/429 once with a short backoff; a deterministic 4xx
312
+ // (e.g. 404 from a wrong slug) is not retried.
313
+ let response: Response | undefined
314
+ for (let attempt = 0; attempt < 2; attempt++) {
315
+ try {
316
+ const res = await fetch(url, {
317
+ method: "GET",
318
+ headers: this.authHeaders(),
319
+ signal,
320
+ })
321
+ if (res.ok) {
322
+ response = res
323
+ break
324
+ }
325
+ if (!(res.status === 429 || res.status >= 500)) {
326
+ return undefined
327
+ }
328
+ } catch {
329
+ // transport failure — retried below
330
+ }
331
+ if (attempt === 0) {
332
+ await new Promise((r) => setTimeout(r, 250))
333
+ }
334
+ }
335
+ if (!response) {
336
+ return undefined
337
+ }
338
+
339
+ // Live shape: { data: { id, endpoints: [{ context_length, pricing, ... }] } }.
340
+ // Legacy/proxy shape: { data: [{ context_length }] }.
341
+ let data:
342
+ | {
343
+ data?:
344
+ | Array<{ context_length?: unknown; provider_name?: unknown; pricing?: unknown }>
345
+ | { endpoints?: Array<{ context_length?: unknown; provider_name?: unknown; pricing?: unknown }> }
346
+ }
347
+ | undefined
348
+ try {
349
+ data = JSON.parse(await response.text()) as typeof data
350
+ } catch {
351
+ return undefined
352
+ }
353
+
354
+ const endpoints = Array.isArray(data?.data)
355
+ ? data.data
356
+ : Array.isArray(data?.data?.endpoints)
357
+ ? data.data.endpoints
358
+ : []
359
+ let contextWindow: number | undefined
360
+ for (const ep of endpoints) {
361
+ const n = typeof ep?.context_length === "number" ? ep.context_length : undefined
362
+ if (typeof n === "number" && Number.isFinite(n) && n > 0) {
363
+ contextWindow = contextWindow === undefined ? n : Math.max(contextWindow, n)
364
+ }
365
+ }
366
+ const providerPreference = model.startsWith("deepseek/") ? "DeepSeek" : undefined
367
+ const price = parseEndpointPricing(endpoints as EndpointPricingEntry[], providerPreference)
368
+ return {
369
+ ...(contextWindow !== undefined ? { contextWindow } : {}),
370
+ ...(price !== undefined ? { price } : {}),
371
+ }
372
+ }
373
+
374
+ /**
375
+ * Fetch a model's advertised context window (tokens) from OpenRouter's
376
+ * `/api/v1/models/<id>/endpoints` endpoint (thin wrapper over
377
+ * fetchModelInfo — see that method's doc for the full contract). Returns
378
+ * `undefined` when the lookup fails or carries no usable number; the
379
+ * caller falls back to `DEFAULT_CONTEXT_WINDOW_TOKENS`.
380
+ */
381
+ async fetchModelContextWindow(model: string, signal?: AbortSignal): Promise<number | undefined> {
382
+ return (await this.fetchModelInfo(model, signal))?.contextWindow
383
+ }
384
+
385
+ async createChatCompletion(request: LlmRequest): Promise<LlmResponse> {
386
+ if (!this.apiKey) {
387
+ throw new OpenRouterError(
388
+ "HEADLESSCODE_OPENROUTER_API_KEY is not set. Set the environment variable HEADLESSCODE_OPENROUTER_API_KEY (or pass apiKey to OpenRouterClient).",
389
+ )
390
+ }
391
+
392
+ const model = this.resolveModel(request.model)
393
+ const url = `${this.baseUrl}/api/v1/chat/completions`
394
+ const startedAt = Date.now()
395
+
396
+ const headers = this.authHeaders()
397
+
398
+ const body = buildRequestBody(request, model)
399
+
400
+ // Opt-in SSE streaming (streaming-and-reasoning). When `request.stream`
401
+ // is true the response is parsed incrementally from the `data:` lines and
402
+ // the final message is assembled from the deltas; otherwise the legacy
403
+ // single blocking fetch path below runs unchanged (the default).
404
+ if (request.stream) {
405
+ return this.streamChatCompletion(request, url, headers, body)
406
+ }
407
+
408
+ try {
409
+ return await this.createChatCompletionNonStreaming(request, url, headers, body, model, startedAt)
410
+ } catch (err) {
411
+ if (isTranscriptCaptureEnabled()) {
412
+ captureTranscript(
413
+ { provider: "openrouter", model },
414
+ {
415
+ messages: request.messages,
416
+ tools: request.tools,
417
+ temperature: request.temperature,
418
+ error: err instanceof Error ? err.message : String(err),
419
+ durationMs: Date.now() - startedAt,
420
+ },
421
+ )
422
+ }
423
+ throw err
424
+ }
425
+ }
426
+
427
+ private async createChatCompletionNonStreaming(
428
+ request: LlmRequest,
429
+ url: string,
430
+ headers: Record<string, string>,
431
+ body: Record<string, unknown>,
432
+ model: string,
433
+ startedAt: number,
434
+ ): Promise<LlmResponse> {
435
+ let response: Response
436
+ try {
437
+ response = await fetch(url, {
438
+ method: "POST",
439
+ headers,
440
+ body: JSON.stringify(body),
441
+ signal: request.signal,
442
+ })
443
+ } catch (err) {
444
+ if (err instanceof Error && err.name === "AbortError") {
445
+ throw err // caller-managed abort (e.g. timeout in the loop)
446
+ }
447
+ throw new OpenRouterError(
448
+ `Network error calling OpenRouter: ${err instanceof Error ? err.message : String(err)}`,
449
+ )
450
+ }
451
+
452
+ if (!response.ok) {
453
+ const rawBody = await response.text().catch(() => "")
454
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
455
+ throw new OpenRouterError(
456
+ `OpenRouter returned HTTP ${response.status}: ${excerpt}`,
457
+ response.status,
458
+ excerpt,
459
+ )
460
+ }
461
+
462
+ // Read the raw body first (not response.json() directly): a 200 status
463
+ // doesn't guarantee a well-formed choices[] payload — OpenRouter/upstream
464
+ // providers can return HTTP 200 with an `{error: {...}}` envelope, an
465
+ // empty/truncated body (e.g. a gateway cutting the connection on a slow
466
+ // generation), or a `choices[0]` with `finish_reason` set but no
467
+ // `message`. Surfacing the raw body distinguishes these cases instead of
468
+ // collapsing them all into one uninformative "no choices[0].message".
469
+ const rawBody = await response.text()
470
+ let data:
471
+ | {
472
+ choices?: Array<{ message?: ChatMessage; finish_reason?: string }>
473
+ usage?: RawUsage
474
+ error?: { message?: string; code?: unknown }
475
+ }
476
+ | undefined
477
+ try {
478
+ data = JSON.parse(rawBody)
479
+ } catch {
480
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
481
+ throw new OpenRouterError(`OpenRouter returned HTTP 200 with a non-JSON/unparseable body: ${excerpt || "(empty)"}`)
482
+ }
483
+
484
+ if (data?.error) {
485
+ throw new OpenRouterError(
486
+ `OpenRouter returned HTTP 200 with an error envelope: ${data.error.message ?? JSON.stringify(data.error)}`,
487
+ )
488
+ }
489
+
490
+ const message = data?.choices?.[0]?.message
491
+ if (!data || !message) {
492
+ const finishReason = data?.choices?.[0]?.finish_reason
493
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
494
+ throw new OpenRouterError(
495
+ `OpenRouter response contained no choices[0].message` +
496
+ (finishReason ? ` (finish_reason: ${finishReason})` : "") +
497
+ `. Raw body: ${excerpt || "(empty)"}`,
498
+ )
499
+ }
500
+
501
+ const result: LlmResponse = {
502
+ message,
503
+ usage: mapUsage(data.usage),
504
+ }
505
+ if (isTranscriptCaptureEnabled()) {
506
+ captureTranscript(
507
+ { provider: "openrouter", model },
508
+ { messages: request.messages, tools: request.tools, temperature: request.temperature, response: result, durationMs: Date.now() - startedAt },
509
+ )
510
+ }
511
+ return result
512
+ }
513
+
514
+ /**
515
+ * Opt-in SSE streaming path (streaming-and-reasoning). `body` is the same
516
+ * request body the blocking path would send, plus `stream: true` and
517
+ * `stream_options: { include_usage: true }` (the usage chunk at the end of
518
+ * the stream is what makes cost accounting work — verified live 2026-08-01:
519
+ * the final `data:` chunk carries `usage` with the same fields as a
520
+ * non-streamed response).
521
+ *
522
+ * Assembles the final `LlmResponse` message from the streamed deltas so the
523
+ * caller sees exactly what a non-streamed equivalent response would have
524
+ * produced. Text, reasoning and tool calls are all reassembled (tool-call
525
+ * arguments arrive as partial JSON across many chunks — see the probe
526
+ * findings in the task notes; each chunk's `delta.tool_calls[i]` carries
527
+ * `index`, and the id/name arrive once on the first chunk for that index,
528
+ * with subsequent chunks carrying only partial `function.arguments`).
529
+ *
530
+ * The caller's AbortSignal (the loop's llmTimeoutMs timer) is passed to the
531
+ * fetch AND honored while reading the stream body: a mid-stream stall still
532
+ * aborts (the reader throws AbortError), so a stalled stream cannot hang the
533
+ * session past the timeout.
534
+ */
535
+ private async streamChatCompletion(
536
+ request: LlmRequest,
537
+ url: string,
538
+ headers: Record<string, string>,
539
+ body: Record<string, unknown>,
540
+ ): Promise<LlmResponse> {
541
+ const streamBody: Record<string, unknown> = {
542
+ ...body,
543
+ }
544
+
545
+ let response: Response
546
+ try {
547
+ response = await fetch(url, {
548
+ method: "POST",
549
+ headers,
550
+ body: JSON.stringify(streamBody),
551
+ signal: request.signal,
552
+ })
553
+ } catch (err) {
554
+ if (err instanceof Error && err.name === "AbortError") {
555
+ throw err
556
+ }
557
+ throw new OpenRouterError(
558
+ `Network error calling OpenRouter (stream): ${err instanceof Error ? err.message : String(err)}`,
559
+ )
560
+ }
561
+
562
+ if (!response.ok) {
563
+ const rawBody = await response.text().catch(() => "")
564
+ const excerpt = rawBody.length > 500 ? `${rawBody.slice(0, 500)}…` : rawBody
565
+ throw new OpenRouterError(
566
+ `OpenRouter returned HTTP ${response.status}: ${excerpt}`,
567
+ response.status,
568
+ excerpt,
569
+ )
570
+ }
571
+
572
+ if (!response.body) {
573
+ throw new OpenRouterError("OpenRouter returned a stream response with no body")
574
+ }
575
+
576
+ const reader = response.body.getReader()
577
+ const decoder = new TextDecoder()
578
+ let buffer = ""
579
+ let done = false
580
+
581
+ // Final assembled message. `reasoning` concatenates `delta.reasoning`
582
+ // strings; `content` concatenates `delta.content`; tool calls are keyed
583
+ // by their stream index and merged after the stream ends.
584
+ const contentParts: string[] = []
585
+ const reasoningParts: string[] = []
586
+ const toolCallsByIdx = new Map<number, { id: string; name: string; args: string[] }>()
587
+ let sawFinish = false
588
+ let sawUsage = false
589
+ let lastUsage: RawUsage | undefined
590
+
591
+ const fail = (message: string): never => {
592
+ throw new OpenRouterError(message)
593
+ }
594
+
595
+ // Fire a live-typing chunk to the caller's callback. Non-fatal by
596
+ // contract (types.ts): a throw from onStreamChunk must never fail the
597
+ // LLM call — the event feed is best-effort.
598
+ const emitChunk = (kind: "text" | "reasoning" | "tool", chunk: string): void => {
599
+ if (!request.onStreamChunk || chunk.length === 0) {
600
+ return
601
+ }
602
+ try {
603
+ request.onStreamChunk(kind, chunk)
604
+ } catch {
605
+ // Swallow: the dashboard's live-typing feed is best-effort.
606
+ }
607
+ }
608
+
609
+ while (!done) {
610
+ // Structural type: the lib is ES2022 (no DOM), so the undici
611
+ // ReadableStream read-result shape is spelled out rather than
612
+ // referenced by name.
613
+ let chunk: { done: boolean; value?: Uint8Array }
614
+ try {
615
+ chunk = await reader.read()
616
+ } catch (err) {
617
+ if (err instanceof Error && err.name === "AbortError") {
618
+ throw err
619
+ }
620
+ throw new OpenRouterError(
621
+ `Error reading OpenRouter stream: ${err instanceof Error ? err.message : String(err)}`,
622
+ )
623
+ }
624
+ if (chunk.done) {
625
+ done = true
626
+ break
627
+ }
628
+ buffer += decoder.decode(chunk.value, { stream: true })
629
+ const lines = buffer.split("\n")
630
+ buffer = lines.pop() ?? ""
631
+
632
+ for (const line of lines) {
633
+ const trimmed = line.trim()
634
+ if (!trimmed || !trimmed.startsWith("data:")) {
635
+ continue
636
+ }
637
+ const payload = trimmed.slice(5).trim()
638
+ if (payload === "[DONE]") {
639
+ done = true
640
+ break
641
+ }
642
+ if (payload === "") {
643
+ continue
644
+ }
645
+ let data:
646
+ | {
647
+ choices?: Array<{ delta?: { content?: string; reasoning?: string; tool_calls?: StreamToolCallDelta[] }; finish_reason?: string }>
648
+ usage?: RawUsage
649
+ error?: { message?: string }
650
+ }
651
+ | undefined
652
+ try {
653
+ data = JSON.parse(payload) as typeof data
654
+ } catch {
655
+ fail(`OpenRouter stream contained a non-JSON data line: ${payload.slice(0, 200)}`)
656
+ }
657
+ if (data?.error) {
658
+ fail(`OpenRouter stream error envelope: ${data.error.message ?? JSON.stringify(data.error)}`)
659
+ }
660
+ const choice = data?.choices?.[0]
661
+ if (choice?.finish_reason) {
662
+ sawFinish = true
663
+ }
664
+ const delta = choice?.delta
665
+ if (delta) {
666
+ if (typeof delta.content === "string") {
667
+ contentParts.push(delta.content)
668
+ emitChunk("text", delta.content)
669
+ }
670
+ if (typeof delta.reasoning === "string") {
671
+ reasoningParts.push(delta.reasoning)
672
+ emitChunk("reasoning", delta.reasoning)
673
+ }
674
+ if (Array.isArray(delta.tool_calls)) {
675
+ for (const tc of delta.tool_calls) {
676
+ const entry = toolCallsByIdx.get(tc.index) ?? { id: "", name: "", args: [] }
677
+ if (typeof tc.id === "string" && tc.id) {
678
+ entry.id = tc.id
679
+ }
680
+ if (typeof tc.function?.name === "string" && tc.function.name) {
681
+ entry.name = tc.function.name
682
+ }
683
+ if (typeof tc.function?.arguments === "string" && tc.function.arguments) {
684
+ entry.args.push(tc.function.arguments)
685
+ emitChunk("tool", tc.function.arguments)
686
+ }
687
+ toolCallsByIdx.set(tc.index, entry)
688
+ }
689
+ }
690
+ }
691
+ if (data?.usage) {
692
+ sawUsage = true
693
+ lastUsage = data.usage
694
+ }
695
+ }
696
+ }
697
+
698
+ const content = contentParts.join("")
699
+ const reasoning = reasoningParts.join("")
700
+ const toolCalls: ChatToolCall[] = [...toolCallsByIdx.entries()]
701
+ .sort(([a], [b]) => a - b)
702
+ .map(([, tc]) => ({
703
+ id: tc.id || `call_stream_${tc.name || "unknown"}`,
704
+ type: "function" as const,
705
+ function: { name: tc.name || "unknown_tool", arguments: tc.args.join("") },
706
+ }))
707
+
708
+ const message: ChatMessage = {
709
+ role: "assistant",
710
+ content: content === "" ? null : content,
711
+ ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
712
+ ...(reasoning !== "" ? { reasoning } : {}),
713
+ }
714
+
715
+ if (!sawFinish && !toolCalls.length && content === "" && reasoning === "" && !sawUsage) {
716
+ // A stream that produced nothing at all (e.g. gateway cut) should
717
+ // surface as an error, not a silent empty assistant turn.
718
+ fail("OpenRouter stream ended with no content, reasoning, tool calls or usage")
719
+ }
720
+
721
+ return {
722
+ message,
723
+ usage: mapUsage(lastUsage),
724
+ }
725
+ }
726
+ }
727
+
728
+ /** A single `delta.tool_calls[]` entry as OpenRouter streams it. */
729
+ interface StreamToolCallDelta {
730
+ index: number
731
+ id?: string
732
+ function?: { name?: string; arguments?: string }
733
+ }
734
+
735
+ /**
736
+ * User-facing reasoning-effort levels for deepseek/* models (issue #30
737
+ * experiment). DeepSeek's native graded set is low/medium/high/max; OpenRouter's
738
+ * normalized `reasoning.effort` accepts "high" and "xhigh" (xhigh maps to the
739
+ * native max). "xhigh" is accepted as an alias for "max" so both vocabularies
740
+ * work. Anything else fails loudly — a typo must never silently keep the
741
+ * endpoint's undeclared default, which would poison the experiment's comparison.
742
+ */
743
+ export const REASONING_EFFORT_LEVELS = ["low", "medium", "high", "max", "xhigh"] as const
744
+
745
+ /** The OpenRouter-normalized `reasoning.effort` values actually sent on the wire. */
746
+ export type WireReasoningEffort = "low" | "medium" | "high" | "xhigh"
747
+
748
+ /**
749
+ * Validate + normalize a configured reasoning-effort value to the
750
+ * OpenRouter-normalized wire value. `undefined`/empty means "no effort sent"
751
+ * (the endpoint's default applies). Throws on anything outside the known set.
752
+ */
753
+ export function parseReasoningEffort(value: string | undefined): WireReasoningEffort | undefined {
754
+ if (value === undefined || value.trim() === "") {
755
+ return undefined
756
+ }
757
+ switch (value.trim().toLowerCase()) {
758
+ case "low":
759
+ return "low"
760
+ case "medium":
761
+ return "medium"
762
+ case "high":
763
+ return "high"
764
+ case "max":
765
+ case "xhigh":
766
+ // OpenRouter's normalized reasoning.effort has no "max" — xhigh is
767
+ // its alias for the native max level.
768
+ return "xhigh"
769
+ default:
770
+ throw new Error(
771
+ `reasoning effort must be one of ${REASONING_EFFORT_LEVELS.join("/")} (native DeepSeek levels, plus OpenRouter's normalized "xhigh" alias for "max"), got '${value}'`,
772
+ )
773
+ }
774
+ }
775
+
776
+ export function buildRequestBody(request: LlmRequest, model: string): Record<string, unknown> {
777
+ const body: Record<string, unknown> = {
778
+ model,
779
+ messages: request.messages,
780
+ }
781
+ if (request.tools && request.tools.length > 0) {
782
+ body.tools = request.tools satisfies ChatTool[]
783
+ }
784
+ if (request.temperature !== undefined) {
785
+ body.temperature = request.temperature
786
+ }
787
+ if (request.maxTokens !== undefined) {
788
+ body.max_tokens = request.maxTokens
789
+ }
790
+ // Opt-in SSE streaming (streaming-and-reasoning): `stream: true` makes
791
+ // OpenRouter return a text/event-stream instead of one JSON body, and
792
+ // `stream_options.include_usage: true` puts the final usage chunk (with
793
+ // prompt/completion/reasoning token counts) at the end of the stream —
794
+ // without it a streamed response carries NO usage data and cost
795
+ // accounting would silently under-report. Both are OpenRouter's own
796
+ // OpenAI-compatible conventions, verified live 2026-08-01.
797
+ if (request.stream) {
798
+ body.stream = true
799
+ body.stream_options = { include_usage: true }
800
+ }
801
+ // Pin routing to DeepSeek's own official endpoint for deepseek/* models,
802
+ // not whichever third-party host (DeepInfra, Baidu, Mancer, ...) OpenRouter's
803
+ // automatic price-based routing might otherwise pick behind the same model
804
+ // id. This matters beyond preference: those endpoints have meaningfully
805
+ // different prices — especially cache-read pricing, where the official
806
+ // endpoint's rate is roughly 10x cheaper than third-party hosts serving the
807
+ // same model — so DEFAULT_PRICING_TABLE's deepseek/* entries
808
+ // (src/budget/cost.ts) are only actually correct WITH this pin in place.
809
+ // `allow_fallbacks: false` means a request fails clearly if the official
810
+ // endpoint is down, rather than silently routing elsewhere at a different
811
+ // (unpriced-by-us) rate.
812
+ if (model.startsWith("deepseek/")) {
813
+ body.provider = { order: ["deepseek"], allow_fallbacks: false }
814
+ // Reasoning content (streaming-and-reasoning): request it explicitly for
815
+ // the only model family this harness pins. Verified live 2026-08-01 on
816
+ // deepseek/deepseek-v4-flash: `include_reasoning: true` is in every
817
+ // endpoint's `supported_parameters`; WITHOUT it the pinned official
818
+ // endpoint still reasons (the model thinks anyway) but `include_reasoning:
819
+ // false` suppresses the returned `reasoning` field — so this flag is
820
+ // load-bearing for actually getting the thinking text back. OpenRouter
821
+ // cleanly ignores the param for models that don't support it (probed with
822
+ // 200s on gpt-4o-mini/claude-3.7-sonnet when the model id resolves; the
823
+ // 404s seen were account/data-policy endpoint availability, not the
824
+ // param). Kept to the deepseek/ prefix rather than unconditional to match
825
+ // the existing pin boundary.
826
+ body.include_reasoning = true
827
+ // Graded reasoning effort (issue #30): DeepSeek's native low/medium/
828
+ // high/max is normalized by OpenRouter to `reasoning: { effort: "high" |
829
+ // "xhigh" }` ("xhigh" = max). Unset = send nothing = the endpoint's
830
+ // undeclared default, exactly as before. Same deepseek/ boundary as
831
+ // include_reasoning.
832
+ if (request.reasoningEffort !== undefined) {
833
+ const effort = parseReasoningEffort(request.reasoningEffort)
834
+ if (effort !== undefined) {
835
+ body.reasoning = { effort }
836
+ }
837
+ }
838
+ }
839
+ return body
840
+ }
841
+
842
+ interface RawUsage {
843
+ prompt_tokens?: number
844
+ completion_tokens?: number
845
+ total_tokens?: number
846
+ /** OpenRouter's pass-through of the underlying provider's cache accounting. */
847
+ prompt_tokens_details?: {
848
+ cached_tokens?: number
849
+ }
850
+ }
851
+
852
+ function mapUsage(raw: RawUsage | undefined): LlmResponse["usage"] {
853
+ if (!raw) {
854
+ return undefined
855
+ }
856
+ return {
857
+ promptTokens: raw.prompt_tokens,
858
+ completionTokens: raw.completion_tokens,
859
+ totalTokens: raw.total_tokens,
860
+ cachedTokens: raw.prompt_tokens_details?.cached_tokens,
861
+ }
862
+ }
863
+
864
+ /** Truncate a raw body to a bounded excerpt for error messages. */
865
+ function excerpt(body: string): string {
866
+ return body.length > 500 ? `${body.slice(0, 500)}…` : body
867
+ }
868
+