github-router 0.3.153 → 0.3.163

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/main.js CHANGED
@@ -1,16 +1,18 @@
1
1
  #!/usr/bin/env node
2
- import { $ as logStreamError, $t as GITHUB_API_BASE_URL, At as DEFAULT_PORT, B as toolbeltEnabled, Bt as tryRefreshAndRetry, C as repoFingerprint, Ct as shouldUseInsecureTls, D as trustRepo, Dt as DEFAULT_CLAUDE_MODEL_FALLBACKS, E as stopReviewStateDir, Et as toolbeltPathOverride, Ft as getPackageVersion, G as searchWeb, Gt as isNullish, H as vscodeRipgrepPath, Ht as cacheModels, It as withInstallLock, J as buildAdvisorStream, Jt as sleep, K as ADVISOR_INTERNAL_TOOL_NAME, Kt as resolveCodexModel, Lt as setupCopilotToken, Mt as UPSTREAM_INACTIVITY_TIMEOUT_MS, Nt as generateRandomPort, O as resolveSealedGate, Ot as DEFAULT_CODEX_MODEL, Pt as pickClaudeDefault, Q as isControllerClosedError, Qt as forwardError, R as availableToolCommands, Rt as setupGitHubAgentToken, S as isSubagentContext, St as extractZipMember, T as stopGateEnabledForRepo, Tt as collapsePathKeys, U as TOOLBELT_TOOLS, Ut as cacheVSCodeVersion, V as toolbeltSkipSet, Vt as cacheCopilotVersion, W as assetFor, Wt as filterBetaHeader, X as isAdvisorRequested, Xt as fetchWithTransientRetry, Y as injectAdvisorTool, Yt as getModels, Z as buildOpenAIErrorEvent, Zt as HTTPError, _ as stopReviewEnabled, _t as parseJsonOrDiagnose, a as buildPeerAwarenessSnippet, at as browseAgentEnabled, b as fileLastPromptStore, bt as provisionAndIndexColbert, c as buildSessionBindHookCommand, ct as standInToolEnabled, d as decideStopHook, dt as createMessages, en as copilotBaseUrl, et as readIteratorWithTimeout, f as fileBlockBudget, ft as getTokenCount, g as stopGateId, gt as readResponseBodyCapped, h as stopGateDisabled, ht as MAX_RESPONSE_BODY_BYTES, i as buildAgentPrompt, it as agentToolsEnabled, jt as UPSTREAM_FETCH_TIMEOUT_MS, k as liveExec, kt as DEFAULT_CODEX_MODEL_FALLBACKS, l as buildStopHookCommand, lt as workerToolsEnabled, m as launchBaselineKey, mt as createChatCompletions, n as MCP_GROUPS, nn as githubHeaders, nt as handleMcpDelete, o as personasFor, ot as browserToolsEnabled, p as injectStopHookIntoSettingsFile, pt as createResponses, q as ADVISOR_TOOL_INSTRUCTIONS, qt as resolveModel, r as assertMcpToolSurfaceConsistent, rn as state, rt as handleMcpPost, s as buildArtifactOpenHookCommand, st as fleetToolsEnabled, t as GROUP_META, tn as copilotHeaders, tt as relayAnthropicStream, u as captureLaunchBaseline, ut as countTokens, v as fileBaselineStore, vt as provisionBrowserAssets, w as repoRoot, wt as ArtifactClient, x as fileReviewDebounce, xt as extractTarGzMember, y as fileFindingsStore, yt as hasSupportedBrowserInstalled, z as buildToolbeltAwareness, zt as setupGitHubToken } from "./peer-mcp-personas-DMM1akDa.js";
3
- import { a as removeOwnClaudeConfigMirror, i as isUnderClaudeConfigMirror, l as writeArtifactCredsToMirror, n as ensureClaudeConfigMirror, r as ensurePaths, t as PATHS, u as writeRuntimeFileSecure } from "./paths-Bt7sqiVr.js";
4
- import { c as killManagedTree, d as runCommandCapture, f as runCommandVoid, l as parseBoolEnv, s as killChildProcessTree, u as resolveExecutable } from "./lifecycle-VTQI28wT.js";
5
- import { a as sweepRegistry } from "./lifecycle-C4k0pEvn.js";
2
+ import { $ as buildOpenAIErrorEvent, $t as resolveModel, At as ArtifactClient, B as buildToolbeltAwareness, Bt as pickClaudeDefault, C as repoFingerprint, Ct as parseJsonOrDiagnose, D as trustRepo, Dt as extractTarGzMember, E as stopReviewStateDir, Et as provisionAndIndexColbert, Ft as DEFAULT_CODEX_MODEL_FALLBACKS, G as assetFor, Gt as setupGitHubToken, H as toolbeltSkipSet, Ht as withInstallLock, It as DEFAULT_PORT, J as ADVISOR_TOOL_INSTRUCTIONS, Jt as cacheModels, K as searchWeb, Kt as tryRefreshAndRetry, Lt as UPSTREAM_FETCH_TIMEOUT_MS, Mt as toolbeltPathOverride, Nt as DEFAULT_CLAUDE_MODEL_FALLBACKS, O as resolveSealedGate, Ot as extractZipMember, Pt as DEFAULT_CODEX_MODEL, Q as buildAnthropicErrorEvent, Qt as resolveCodexModel, Rt as UPSTREAM_INACTIVITY_TIMEOUT_MS, S as isSubagentContext, St as readResponseBodyCapped, T as stopGateEnabledForRepo, Tt as hasSupportedBrowserInstalled, U as vscodeRipgrepPath, Ut as setupCopilotToken, V as toolbeltEnabled, Vt as getPackageVersion, W as TOOLBELT_TOOLS, Wt as setupGitHubAgentToken, X as injectAdvisorTool, Xt as filterBetaHeader, Y as buildAdvisorStream, Yt as cacheVSCodeVersion, Z as isAdvisorRequested, Zt as isNullish, _ as stopReviewEnabled, _t as resolveMcpToolTimeoutMs, a as buildPeerAwarenessSnippet, an as GITHUB_API_BASE_URL, at as handleMcpPost, b as fileLastPromptStore, bt as createChatCompletions, c as buildSessionBindHookCommand, cn as githubHeaders, ct as browserToolsEnabled, d as decideStopHook, dt as standInToolEnabled, en as sleep, et as isControllerClosedError, f as fileBlockBudget, ft as workerToolsEnabled, g as stopGateId, gt as assembleResponsesPayload, h as stopGateDisabled, ht as getTokenCount, i as buildAgentPrompt, in as forwardError, it as handleMcpDelete, jt as collapsePathKeys, k as liveExec, kt as shouldUseInsecureTls, l as buildStopHookCommand, ln as state, lt as fleetToolsEnabled, m as launchBaselineKey, mt as createMessages, n as MCP_GROUPS, nn as fetchWithTransientRetry, nt as readIteratorWithTimeout, o as personasFor, on as copilotBaseUrl, ot as agentToolsEnabled, p as injectStopHookIntoSettingsFile, pt as countTokens, q as ADVISOR_INTERNAL_TOOL_NAME, qt as cacheCopilotVersion, r as assertMcpToolSurfaceConsistent, rn as HTTPError, rt as relayAnthropicStream, s as buildArtifactOpenHookCommand, sn as copilotHeaders, st as browseAgentEnabled, t as GROUP_META, tn as getModels, tt as logStreamError, u as captureLaunchBaseline, ut as implementerSubagentModel, v as fileBaselineStore, vt as pickEndpoint, w as repoRoot, wt as provisionBrowserAssets, x as fileReviewDebounce, xt as MAX_RESPONSE_BODY_BYTES, y as fileFindingsStore, yt as createResponses, z as availableToolCommands, zt as generateRandomPort } from "./peer-mcp-personas-Bfq3FPKc.js";
3
+ import { a as removeOwnClaudeConfigMirror, i as isUnderClaudeConfigMirror, l as writeArtifactCredsToMirror, n as ensureClaudeConfigMirror, r as ensurePaths, t as PATHS, u as writeRuntimeFileSecure } from "./paths-CTlT1nTo.js";
4
+ import { c as killManagedTree, d as runCommandCapture, f as runCommandVoid, l as parseBoolEnv, s as killChildProcessTree, u as resolveExecutable } from "./lifecycle-ChPBRt6K.js";
5
+ import { a as sweepRegistry } from "./lifecycle-BUXxiltc.js";
6
6
  import { defineCommand, runMain } from "citty";
7
7
  import consola from "consola";
8
8
  import { createHash, randomBytes, randomUUID } from "node:crypto";
9
9
  import fs, { chmod, copyFile, link, mkdir, readFile, readdir, rename, rm, writeFile } from "node:fs/promises";
10
10
  import os, { homedir, tmpdir } from "node:os";
11
+ import * as nodePath$1 from "node:path";
11
12
  import nodePath from "node:path";
12
13
  import process$1 from "node:process";
13
14
  import { execFileSync, spawn } from "node:child_process";
15
+ import * as fs$2 from "node:fs";
14
16
  import fs$1, { existsSync, mkdirSync, promises, readFileSync, realpathSync, renameSync, unlinkSync, writeFileSync } from "node:fs";
15
17
  import { Agent, ProxyAgent, setGlobalDispatcher } from "undici";
16
18
  import { Writable } from "node:stream";
@@ -815,7 +817,7 @@ function launchChild(target, server$1, options = {}) {
815
817
  * Frozen contract for the NON-BLOCKING workers surface.
816
818
  *
817
819
  * The `workers` MCP tools (`explore`/`implement`/`review`/`plan`/`test`, and
818
- * `browse` when the browse agent is enabled) BLOCK the caller for up to 30 min
820
+ * `browse` when the browse agent is enabled) BLOCK the caller for up to 6h
819
821
  * (`runWorkerAgent`). The MAIN Claude Code agent must never block on one, so a
820
822
  * per-mode `worker-*` DISPATCHER SUBAGENT — which Claude Code runs in the
821
823
  * background and reports on via a completion notification — is the only
@@ -971,14 +973,15 @@ function dispatcherPrompt(mode, workersKey) {
971
973
  `# Subagent: ${dispatcherAgentName(mode)}`,
972
974
  "",
973
975
  `You are a thin DISPATCHER for the \`${mode}\` worker. You run in the background so the`,
974
- "lead agent's turn is never blocked while the (up-to-30-minute) worker runs.",
976
+ "lead agent's turn is never blocked while the (up-to-6-hour) worker runs.",
975
977
  "",
976
978
  "## Your only job",
977
979
  "",
978
980
  `Call the \`${tool}\` tool EXACTLY ONCE, passing through the fields from the lead's brief:`,
979
981
  " - `prompt`: the lead's worker brief, copied verbatim",
980
982
  " - `workspace` (optional): absolute path, if the lead specified one",
981
- " - `model` / `thinking` (optional): only if the lead specified them" + (mode === "implement" || mode === "test" ? "\n - `worktree` (optional): pass `true` if the lead asked for isolated-worktree execution" : ""),
983
+ " - `model` / `thinking` (optional): only if the lead specified them",
984
+ " - `maxWallClockMs` (optional): per-call wall-clock budget in ms, if the lead specified one" + (mode === "implement" || mode === "test" ? "\n - `worktree` (optional): pass `true` if the lead asked for isolated-worktree execution" : ""),
982
985
  "",
983
986
  "When the tool returns, output its result VERBATIM as your final message. That final",
984
987
  "message is what the lead receives in the completion notification — it IS the result.",
@@ -1199,6 +1202,11 @@ function buildPeerAgentDefinitions(opts) {
1199
1202
  codexCli: opts.codexCli,
1200
1203
  geminiAvailable: opts.geminiAvailable
1201
1204
  });
1205
+ if (opts.implementerModel && opts.implementerModel.length > 0) out.implementer = {
1206
+ description: "Bounded implementation subagent running gpt-5.5 (strong non-Claude coder, high reasoning). Use for well-scoped coding tasks — edits, small features, fixes — you want implemented in an integrated subagent. Model is overridable at spawn.",
1207
+ prompt: "You are a bounded implementation subagent for well-scoped coding tasks. Implement the requested change surgically, matching the surrounding code style and minimizing unrelated churn. Use the dedicated Edit/Write/Read tools for file changes and Grep/Glob for search; reserve Bash for running builds, tests, and git — do not shell out (sed/awk/python/here-docs) to read or edit files. Verify with the project's build or tests where applicable. Do the work yourself — do not spawn further subagents. Report exactly what changed and any risks.",
1208
+ model: opts.implementerModel
1209
+ };
1202
1210
  if (opts.workerToolsAvailable) {
1203
1211
  const workersKey = workersKeyOf(opts.groupKeys);
1204
1212
  for (const mode of activeDispatchModes({ browse: opts.browseAvailable === true })) out[dispatcherAgentName(mode)] = {
@@ -1265,6 +1273,7 @@ function buildAgentMd(spec) {
1265
1273
  `name: ${spec.name}`,
1266
1274
  `description: ${escapeYamlString(spec.description)}`
1267
1275
  ];
1276
+ if (spec.model) lines.push(`model: ${escapeYamlString(spec.model)}`);
1268
1277
  if (spec.tools && spec.tools.length > 0) lines.push(`tools: [${spec.tools.map((t) => JSON.stringify(t)).join(", ")}]`);
1269
1278
  lines.push("---", "", spec.prompt, "");
1270
1279
  return lines.join("\n");
@@ -1299,6 +1308,7 @@ async function writePeerAgentMdFiles(agents, opts) {
1299
1308
  name: name$1,
1300
1309
  description: def.description,
1301
1310
  prompt: def.prompt,
1311
+ model: def.model,
1302
1312
  tools: def.tools
1303
1313
  }));
1304
1314
  paths.push(filePath);
@@ -1493,6 +1503,7 @@ async function writePeerMcpRuntimeFiles(serverUrl, opts) {
1493
1503
  groupKeys: opts.groupKeys,
1494
1504
  workerToolsAvailable: opts.workerToolsAvailable,
1495
1505
  browseAvailable: opts.browseAvailable,
1506
+ implementerModel: opts.implementerModel,
1496
1507
  nonce,
1497
1508
  codexHome
1498
1509
  });
@@ -2278,7 +2289,7 @@ async function discoverGateCommands(cwd, opts) {
2278
2289
  if (files.length === 0) return null;
2279
2290
  let result;
2280
2291
  try {
2281
- const { runWorkerAgent } = await import("./engine-BP5EnZ-n.js");
2292
+ const { runWorkerAgent } = await import("./engine-DHFa2_hK.js");
2282
2293
  result = await runWorkerAgent({
2283
2294
  mode: "explore",
2284
2295
  workspace: root,
@@ -2978,14 +2989,14 @@ const WORKER_SKILL = {
2978
2989
  name: "gh-worker",
2979
2990
  md: `---
2980
2991
  name: gh-worker
2981
- description: How to run github-router workers without blocking your turn. Workers (explore/implement/review/plan/test) can run up to 30 minutes; dispatch the matching worker-* background subagent so you get a completion notification instead of waiting. Use whenever you would reach for a worker.
2992
+ description: How to run github-router workers without blocking your turn. Workers (explore/implement/review/plan/test) can run up to 6 hours; dispatch the matching worker-* background subagent so you get a completion notification instead of waiting. Use whenever you would reach for a worker.
2982
2993
  user-invocable: true
2983
2994
  ---
2984
2995
 
2985
2996
  # gh-worker: non-blocking workers
2986
2997
 
2987
- Worker tasks (explore, implement, review, plan, test) can run for up to 30
2988
- minutes. In this session they are NON-BLOCKING BY DESIGN: you dispatch a
2998
+ Worker tasks (explore, implement, review, plan, test) can run for up to 6 hours.
2999
+ In this session they are NON-BLOCKING BY DESIGN: you dispatch a
2989
3000
  background \`worker-*\` subagent, get control back immediately, and receive the
2990
3001
  worker's result as a completion notification when it finishes. Your turn is
2991
3002
  never blocked waiting on a worker, and the worker's tool output never fills your
@@ -4084,7 +4095,7 @@ function initProxyFromEnv() {
4084
4095
  //#endregion
4085
4096
  //#region package.json
4086
4097
  var name = "github-router";
4087
- var version$1 = "0.3.153";
4098
+ var version$1 = "0.3.163";
4088
4099
 
4089
4100
  //#endregion
4090
4101
  //#region src/lib/approval.ts
@@ -4681,144 +4692,1583 @@ function sanitizeAnthropicBody(rawBody) {
4681
4692
  }
4682
4693
 
4683
4694
  //#endregion
4684
- //#region src/routes/messages/handler.ts
4685
- const isWebSearchTool$1 = (tool) => typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search";
4686
- /**
4687
- * Extract whitelisted beta headers from the incoming request to forward
4688
- * to the Copilot API. VS Code sends these to enable extended features
4689
- * like thinking, context management, and advanced tool use.
4690
- */
4691
- function extractBetaHeaders(c) {
4692
- const headers = {};
4693
- const anthropicBeta = c.req.header("anthropic-beta");
4694
- if (anthropicBeta) {
4695
- const filtered = filterBetaHeader(anthropicBeta);
4696
- if (filtered) headers["anthropic-beta"] = filtered;
4697
- }
4698
- return headers;
4695
+ //#region src/lib/anthropic-translate/anthropic-sse.ts
4696
+ function makeMessageId() {
4697
+ return `msg_${randomUUID().replace(/-/g, "")}`;
4698
+ }
4699
+ function makeMessageStart(id, model, usage = {}) {
4700
+ return {
4701
+ type: "message_start",
4702
+ message: {
4703
+ id,
4704
+ type: "message",
4705
+ role: "assistant",
4706
+ model,
4707
+ content: [],
4708
+ stop_reason: null,
4709
+ stop_sequence: null,
4710
+ usage: {
4711
+ input_tokens: usage.input_tokens ?? 0,
4712
+ output_tokens: usage.output_tokens ?? 0,
4713
+ cache_read_input_tokens: usage.cache_read_input_tokens ?? 0,
4714
+ cache_creation_input_tokens: usage.cache_creation_input_tokens ?? 0
4715
+ }
4716
+ }
4717
+ };
4718
+ }
4719
+ function makeContentBlockStart(index, block) {
4720
+ return {
4721
+ type: "content_block_start",
4722
+ index,
4723
+ content_block: block
4724
+ };
4725
+ }
4726
+ function makeTextDelta(index, text) {
4727
+ return {
4728
+ type: "content_block_delta",
4729
+ index,
4730
+ delta: {
4731
+ type: "text_delta",
4732
+ text
4733
+ }
4734
+ };
4735
+ }
4736
+ function makeInputJsonDelta(index, partialJson) {
4737
+ return {
4738
+ type: "content_block_delta",
4739
+ index,
4740
+ delta: {
4741
+ type: "input_json_delta",
4742
+ partial_json: partialJson
4743
+ }
4744
+ };
4745
+ }
4746
+ function makeThinkingDelta(index, thinking) {
4747
+ return {
4748
+ type: "content_block_delta",
4749
+ index,
4750
+ delta: {
4751
+ type: "thinking_delta",
4752
+ thinking
4753
+ }
4754
+ };
4755
+ }
4756
+ function makeSignatureDelta(index, signature) {
4757
+ return {
4758
+ type: "content_block_delta",
4759
+ index,
4760
+ delta: {
4761
+ type: "signature_delta",
4762
+ signature
4763
+ }
4764
+ };
4765
+ }
4766
+ function makeContentBlockStop(index) {
4767
+ return {
4768
+ type: "content_block_stop",
4769
+ index
4770
+ };
4771
+ }
4772
+ function makeMessageDelta(stopReason, stopSequence, usage) {
4773
+ return {
4774
+ type: "message_delta",
4775
+ delta: {
4776
+ stop_reason: stopReason,
4777
+ stop_sequence: stopSequence
4778
+ },
4779
+ usage: {
4780
+ input_tokens: usage.input_tokens ?? 0,
4781
+ output_tokens: usage.output_tokens ?? 0,
4782
+ cache_read_input_tokens: usage.cache_read_input_tokens ?? 0,
4783
+ cache_creation_input_tokens: usage.cache_creation_input_tokens ?? 0
4784
+ }
4785
+ };
4786
+ }
4787
+ function makeMessageStop() {
4788
+ return { type: "message_stop" };
4789
+ }
4790
+ /** Serialize one event to the Anthropic SSE wire form (`event:` + `data:`). */
4791
+ function serializeAnthropicEvent(ev) {
4792
+ return `event: ${ev.type}\ndata: ${JSON.stringify(ev)}\n\n`;
4699
4793
  }
4700
4794
  /**
4701
- * Extract the text content from the last user message for web search.
4702
- * Handles both string content and content block arrays (multimodal).
4795
+ * Wrap a generator of Anthropic stream events into a byte `ReadableStream`.
4796
+ * Pull-based; guards the consumer-cancel race on both `enqueue` and `close`;
4797
+ * converts a mid-stream generator throw into a terminal `event: error` frame.
4798
+ * On consumer cancel it invokes `onCancel` (abort the upstream fetch) and
4799
+ * `return()`s the generator so its `finally` tears down the upstream reader.
4703
4800
  */
4704
- function extractUserQuery$1(messages) {
4705
- for (let i = messages.length - 1; i >= 0; i--) {
4706
- const msg = messages[i];
4707
- if (msg.role === "user") {
4708
- if (typeof msg.content === "string") return msg.content;
4709
- if (Array.isArray(msg.content)) {
4710
- const textBlock = msg.content.find((block) => block.type === "text");
4711
- if (textBlock?.text) return textBlock.text;
4801
+ function anthropicSseStreamFromEvents(events, opts) {
4802
+ const enc = new TextEncoder();
4803
+ let consumerCancelled = false;
4804
+ let finished = false;
4805
+ const safeClose = (controller) => {
4806
+ try {
4807
+ controller.close();
4808
+ } catch {}
4809
+ };
4810
+ return new ReadableStream({
4811
+ async pull(controller) {
4812
+ if (consumerCancelled || finished) {
4813
+ safeClose(controller);
4814
+ return;
4815
+ }
4816
+ let res;
4817
+ try {
4818
+ res = await events.next();
4819
+ } catch (err) {
4820
+ finished = true;
4821
+ if (consumerCancelled) {
4822
+ safeClose(controller);
4823
+ return;
4824
+ }
4825
+ const name$1 = err instanceof Error ? err.name : "Error";
4826
+ const message = err instanceof Error ? err.message : String(err);
4827
+ consola.error(`Anthropic-translate stream interrupted at ${opts.routePath}: ${name$1}: ${JSON.stringify(message)}`);
4828
+ try {
4829
+ controller.enqueue(enc.encode(buildAnthropicErrorEvent(name$1, message)));
4830
+ } catch (enqueueError) {
4831
+ if (!isControllerClosedError(enqueueError)) consola.warn(`Could not deliver error event to consumer at ${opts.routePath}: ${enqueueError instanceof Error ? enqueueError.message : String(enqueueError)}`);
4832
+ }
4833
+ safeClose(controller);
4834
+ return;
4835
+ }
4836
+ if (consumerCancelled) {
4837
+ safeClose(controller);
4838
+ return;
4839
+ }
4840
+ if (res.done) {
4841
+ finished = true;
4842
+ safeClose(controller);
4843
+ return;
4844
+ }
4845
+ try {
4846
+ controller.enqueue(enc.encode(serializeAnthropicEvent(res.value)));
4847
+ } catch (err) {
4848
+ if (isControllerClosedError(err)) {
4849
+ consumerCancelled = true;
4850
+ return;
4851
+ }
4852
+ throw err;
4712
4853
  }
4854
+ },
4855
+ cancel() {
4856
+ consumerCancelled = true;
4857
+ finished = true;
4858
+ opts.onCancel?.();
4859
+ events.return?.(void 0);
4713
4860
  }
4714
- }
4861
+ });
4715
4862
  }
4863
+
4864
+ //#endregion
4865
+ //#region src/lib/reasoning-effort.ts
4716
4866
  /**
4717
- * Check if any user message contains tool_result content blocks,
4718
- * indicating a follow-up turn where we should skip web search.
4719
- * In Anthropic format, tool results are content blocks inside user messages,
4720
- * NOT separate role: "tool" messages like in OpenAI format.
4867
+ * Reasoning-effort bucketing + clamping shared by the `/v1/messages` handler
4868
+ * (adaptive-thinking translation) and the Anthropic-translation shim
4869
+ * (thinking-budget Responses `reasoning.effort`).
4870
+ *
4871
+ * Extracted from `src/routes/messages/handler.ts` so a `~/lib/*` module can
4872
+ * depend on it without importing route code (and without forming a
4873
+ * handler → shim → handler import cycle). `handler.ts` re-exports these for
4874
+ * backward compatibility with existing imports/tests.
4721
4875
  */
4722
- function hasToolResultContent(messages) {
4723
- return messages.some((msg) => Array.isArray(msg.content) && msg.content.some((block) => block.type === "tool_result"));
4724
- }
4876
+ const EFFORT_ORDER = [
4877
+ "low",
4878
+ "medium",
4879
+ "high",
4880
+ "xhigh"
4881
+ ];
4725
4882
  /**
4726
- * Inject web search results into the Anthropic system field.
4727
- * Handles three cases: absent, string, or array of content blocks.
4728
- * When array, prepends without cache_control to preserve existing directives.
4883
+ * Bucket a thinking budget into a Copilot reasoning-effort string.
4884
+ * `<2000`→low, `<8000`→medium, `<24000`→high, else→xhigh.
4885
+ * Defaults missing/non-numeric budgets to 8000 ("high").
4729
4886
  */
4730
- function injectSearchResults(body, searchContext) {
4731
- if (body.system === void 0 || body.system === null) body.system = searchContext;
4732
- else if (typeof body.system === "string") body.system = `${searchContext}\n\n${body.system}`;
4733
- else if (Array.isArray(body.system)) body.system = [{
4734
- type: "text",
4735
- text: searchContext
4736
- }, ...body.system];
4887
+ function bucketEffort(budget) {
4888
+ const n = typeof budget === "number" && Number.isFinite(budget) ? budget : 8e3;
4889
+ if (n < 2e3) return "low";
4890
+ if (n < 8e3) return "medium";
4891
+ if (n < 24e3) return "high";
4892
+ return "xhigh";
4737
4893
  }
4738
4894
  /**
4739
- * Strip web_search tools from the request and clean up tool_choice.
4740
- * Returns the modified body object.
4895
+ * Clamp a bucketed effort to the closest value in `supported`. Ties resolve to
4896
+ * the lower-tier option (per EFFORT_ORDER).
4897
+ *
4898
+ * Iterates EFFORT_ORDER (canonical low→xhigh) so the first match on a given
4899
+ * distance is always the lower-tier value, regardless of input order in
4900
+ * `supported`.
4741
4901
  */
4742
- function stripWebSearchTool(body) {
4743
- if (!body.tools) return;
4744
- body.tools = body.tools.filter((tool) => !isWebSearchTool$1(tool));
4745
- if (body.tools.length === 0) {
4746
- body.tools = void 0;
4747
- body.tool_choice = void 0;
4748
- } else if (body.tool_choice && typeof body.tool_choice === "object" && body.tool_choice.type === "tool") {
4749
- const choiceName = body.tool_choice.name;
4750
- if (choiceName && !body.tools.some((tool) => tool.name === choiceName)) body.tool_choice = { type: "auto" };
4902
+ function clampEffort(bucketed, supported) {
4903
+ if (supported.includes(bucketed)) return bucketed;
4904
+ const targetIdx = EFFORT_ORDER.indexOf(bucketed);
4905
+ let best;
4906
+ let bestDist = Infinity;
4907
+ for (let i = 0; i < EFFORT_ORDER.length; i++) {
4908
+ const value = EFFORT_ORDER[i];
4909
+ if (!supported.includes(value)) continue;
4910
+ const dist = Math.abs(i - targetIdx);
4911
+ if (dist < bestDist) {
4912
+ bestDist = dist;
4913
+ best = value;
4914
+ }
4915
+ }
4916
+ return best ?? bucketed;
4917
+ }
4918
+
4919
+ //#endregion
4920
+ //#region src/lib/anthropic-translate/anthropic-request.ts
4921
+ /**
4922
+ * Steering appended to `instructions` for shim-routed (non-Claude) models when
4923
+ * the request carries Claude Code's native file tools. gpt-5.5 and other
4924
+ * OpenAI/Gemini-lineage models receive the Edit/Write tool definitions verbatim
4925
+ * (the shim never mangles them), but their base prior is to script file ops in
4926
+ * Python/Bash rather than call the dedicated tools. Claude models never reach
4927
+ * this code path (they fall through to the /v1/messages passthrough), so this is
4928
+ * automatically scoped to the models that need the nudge. Strong PREFERENCE, not
4929
+ * a Bash ban — running builds/tests/git still belongs in Bash.
4930
+ */
4931
+ const FILE_TOOL_GUIDANCE = `<file_tools>
4932
+ You have dedicated tools for files: use Read to read a file, Edit to modify an existing file, and Write to create one. Prefer them over shell for reading or editing. Do NOT shell out (cat, sed, awk, echo >, here-docs, or python/one-off scripts) to read, search, or rewrite file contents when a dedicated tool exists — the dedicated tools are safer and produce reviewable diffs. Use Grep/Glob to search rather than shell grep/find. Reserve Bash for commands that have no dedicated tool: builds, tests, git, package managers, and running programs.
4933
+ </file_tools>`;
4934
+ /**
4935
+ * Append `FILE_TOOL_GUIDANCE` to the flattened system `instructions` iff the
4936
+ * request carries Claude Code's canonical `Edit` or `Write` tool. The exact
4937
+ * capitalized-name match is deliberately precise: it fires for a Claude Code
4938
+ * editing session but not for arbitrary MCP tools like `write_file`, and not for
4939
+ * non-editing chats (so a plain gpt-5.5 conversation is not polluted). The block
4940
+ * is appended AFTER the existing instructions (end-of-prompt recency) and the
4941
+ * original system text is preserved, never replaced. Opt out with
4942
+ * `GH_ROUTER_DISABLE_SHIM_TOOL_STEERING=1`.
4943
+ */
4944
+ function appendFileToolGuidance(instructions, tools) {
4945
+ if (parseBoolEnv(process.env.GH_ROUTER_DISABLE_SHIM_TOOL_STEERING) === true) return instructions;
4946
+ if (!tools?.some((t) => t.name === "Edit" || t.name === "Write")) return instructions;
4947
+ return instructions && instructions.length > 0 ? `${instructions}\n\n${FILE_TOOL_GUIDANCE}` : FILE_TOOL_GUIDANCE;
4948
+ }
4949
+ /** Flatten Anthropic `system` (string | array of text blocks) into a string. */
4950
+ function flattenSystem(system) {
4951
+ if (typeof system === "string") return system.length > 0 ? system : void 0;
4952
+ if (Array.isArray(system)) {
4953
+ let s = "";
4954
+ for (const block of system) if (block && typeof block === "object" && block.type === "text") {
4955
+ const t = block.text;
4956
+ if (typeof t === "string") s += t;
4957
+ }
4958
+ return s.length > 0 ? s : void 0;
4959
+ }
4960
+ }
4961
+ /**
4962
+ * Parse an Anthropic `tool_result.content` (string | block array) into the
4963
+ * plain-text `output` for the Responses `function_call_output` (a string-only
4964
+ * item) PLUS any image parts found in the content. A `function_call_output`
4965
+ * cannot carry images, so the caller emits the extracted images as a follow-up
4966
+ * user message (Claude Code browser screenshots/observations arrive this way).
4967
+ * `isError` (the tool_result `is_error` flag) is preserved by prefixing the
4968
+ * text so the model still learns the tool call failed.
4969
+ */
4970
+ function parseToolResultContent(content, isError) {
4971
+ const images = [];
4972
+ let text = "";
4973
+ if (typeof content === "string") text = content;
4974
+ else if (Array.isArray(content)) for (const block of content) {
4975
+ if (!block || typeof block !== "object") continue;
4976
+ const b = block;
4977
+ if (b.type === "text" && typeof b.text === "string") text += b.text;
4978
+ else if (b.type === "image") {
4979
+ const img = anthropicImageToNeutral(b.source);
4980
+ if (img) images.push(img);
4981
+ }
4982
+ }
4983
+ if (images.length > 0 && text.length === 0) text = "[image result below]";
4984
+ if (isError) text = text.length > 0 ? `[tool error] ${text}` : "[tool error]";
4985
+ return {
4986
+ output: text,
4987
+ images
4988
+ };
4989
+ }
4990
+ /** Map an Anthropic `image` block source to a neutral image part. */
4991
+ function anthropicImageToNeutral(source) {
4992
+ if (!source || typeof source !== "object") return null;
4993
+ if (source.type === "url" && typeof source.url === "string") return {
4994
+ type: "image",
4995
+ url: source.url
4996
+ };
4997
+ if (source.type === "base64" && typeof source.data === "string") return {
4998
+ type: "image",
4999
+ mimeType: typeof source.media_type === "string" ? source.media_type : "image/png",
5000
+ data: source.data
5001
+ };
5002
+ return null;
5003
+ }
5004
+ /** Concatenate the text of an Anthropic `document` `content`-source block array. */
5005
+ function joinDocumentContentText(content) {
5006
+ if (!Array.isArray(content)) return "";
5007
+ let text = "";
5008
+ for (const block of content) if (block && typeof block === "object") {
5009
+ const b = block;
5010
+ if (b.type === "text" && typeof b.text === "string") text += b.text;
5011
+ }
5012
+ return text;
5013
+ }
5014
+ /**
5015
+ * Map an Anthropic `document` block to a neutral content part.
5016
+ * - base64 source → neutral `document` (mimeType + data) → Responses
5017
+ * `input_file` with `file_data`; on the chat path → an inline text note
5018
+ * (Copilot's `/chat/completions` rejects file parts).
5019
+ * - url source → neutral `document` (url) → Responses `input_file.file_url`.
5020
+ * - text source (a plain-text document) → the doc's text folded into a `text`
5021
+ * part, so the model sees it on BOTH paths.
5022
+ * - content source (content-block document) → its text blocks folded into a
5023
+ * `text` part.
5024
+ * Missing/invalid fields (unknown source type, `file`-id references Copilot has
5025
+ * no Files API for, empty text) yield null and are dropped.
5026
+ */
5027
+ function anthropicDocumentToNeutral(b) {
5028
+ const source = b.source;
5029
+ if (!source || typeof source !== "object") return null;
5030
+ const s = source;
5031
+ const filename = typeof b.title === "string" && b.title.length > 0 ? b.title : "document.pdf";
5032
+ if (s.type === "base64" && typeof s.data === "string") return {
5033
+ type: "document",
5034
+ mimeType: typeof s.media_type === "string" ? s.media_type : "application/pdf",
5035
+ data: s.data,
5036
+ filename
5037
+ };
5038
+ if (s.type === "url" && typeof s.url === "string") return {
5039
+ type: "document",
5040
+ url: s.url,
5041
+ filename
5042
+ };
5043
+ if (s.type === "text" && typeof s.data === "string") return s.data.length > 0 ? {
5044
+ type: "text",
5045
+ text: s.data
5046
+ } : null;
5047
+ if (s.type === "content") {
5048
+ const text = joinDocumentContentText(s.content);
5049
+ return text.length > 0 ? {
5050
+ type: "text",
5051
+ text
5052
+ } : null;
4751
5053
  }
5054
+ return null;
4752
5055
  }
4753
5056
  /**
4754
- * Process web search if the request contains a web_search tool.
4755
- * Performs the search, injects results into system, and strips the tool.
4756
- * Returns the (possibly modified) body string to forward.
4757
- */
4758
- async function processWebSearch(rawBody) {
4759
- if (!rawBody.includes("web_search")) return rawBody;
4760
- let body;
4761
- try {
4762
- body = JSON.parse(rawBody);
4763
- } catch {
4764
- return rawBody;
5057
+ * Convert one Anthropic message into zero-or-more neutral messages. A user
5058
+ * message with `tool_result` blocks fans out: text/image content becomes a
5059
+ * user message and each tool_result becomes its own `toolResult` message,
5060
+ * emitted in wire order so a function_call_output never precedes its text.
5061
+ */
5062
+ function anthropicMessageToNeutral(msg) {
5063
+ const role = msg.role;
5064
+ const content = msg.content;
5065
+ if (role === "assistant") {
5066
+ const parts = [];
5067
+ if (typeof content === "string") {
5068
+ if (content.length > 0) parts.push({
5069
+ type: "text",
5070
+ text: content
5071
+ });
5072
+ } else if (Array.isArray(content)) for (const block of content) {
5073
+ if (!block || typeof block !== "object") continue;
5074
+ const b = block;
5075
+ if (b.type === "text" && typeof b.text === "string") parts.push({
5076
+ type: "text",
5077
+ text: b.text
5078
+ });
5079
+ else if (b.type === "tool_use") parts.push({
5080
+ type: "toolCall",
5081
+ id: typeof b.id === "string" ? b.id : "",
5082
+ name: typeof b.name === "string" ? b.name : "",
5083
+ arguments: b.input ?? {}
5084
+ });
5085
+ }
5086
+ return [{
5087
+ role: "assistant",
5088
+ content: parts
5089
+ }];
4765
5090
  }
4766
- if (!body.tools?.some((tool) => isWebSearchTool$1(tool))) return rawBody;
4767
- const query = hasToolResultContent(body.messages ?? []) ? void 0 : extractUserQuery$1(body.messages ?? []);
4768
- if (query) try {
4769
- const results = await searchWeb(query);
4770
- const searchContext = [
4771
- "[Web Search Results]",
4772
- results.content,
4773
- "",
4774
- results.references.map((r) => `- [${r.title}](${r.url})`).join("\n"),
4775
- "[End Web Search Results]"
4776
- ].join("\n");
4777
- injectSearchResults(body, searchContext);
4778
- } catch (error) {
4779
- consola.warn("Web search failed, continuing without results:", error);
5091
+ const out = [];
5092
+ let userParts = [];
5093
+ const flushUser = () => {
5094
+ if (userParts.length === 0) return;
5095
+ out.push({
5096
+ role: "user",
5097
+ content: userParts
5098
+ });
5099
+ userParts = [];
5100
+ };
5101
+ if (typeof content === "string") {
5102
+ if (content.length > 0) out.push({
5103
+ role: "user",
5104
+ content
5105
+ });
5106
+ return out;
5107
+ }
5108
+ if (Array.isArray(content)) for (const block of content) {
5109
+ if (!block || typeof block !== "object") continue;
5110
+ const b = block;
5111
+ if (b.type === "text" && typeof b.text === "string") userParts.push({
5112
+ type: "text",
5113
+ text: b.text
5114
+ });
5115
+ else if (b.type === "image") {
5116
+ const img = anthropicImageToNeutral(b.source);
5117
+ if (img) userParts.push(img);
5118
+ } else if (b.type === "document") {
5119
+ const doc = anthropicDocumentToNeutral(b);
5120
+ if (doc) userParts.push(doc);
5121
+ } else if (b.type === "tool_result") {
5122
+ flushUser();
5123
+ const { output, images } = parseToolResultContent(b.content, b.is_error === true);
5124
+ out.push({
5125
+ role: "toolResult",
5126
+ toolCallId: typeof b.tool_use_id === "string" ? b.tool_use_id : "",
5127
+ output
5128
+ });
5129
+ if (images.length > 0) out.push({
5130
+ role: "user",
5131
+ content: images
5132
+ });
5133
+ }
4780
5134
  }
4781
- stripWebSearchTool(body);
4782
- return JSON.stringify(body);
5135
+ flushUser();
5136
+ return out;
4783
5137
  }
4784
- async function handleCompletion(c) {
4785
- const startTime = Date.now();
4786
- await checkRateLimit(state);
4787
- const rawBody = await c.req.text();
4788
- const debugEnabled = consola.level >= 4;
4789
- if (debugEnabled) consola.debug("Anthropic request body:", rawBody.slice(0, 2e3));
4790
- if (process.env.GH_ROUTER_LOG_FIELDS === "1") {
4791
- let parsedForLog = void 0;
4792
- try {
4793
- parsedForLog = JSON.parse(rawBody);
4794
- } catch {}
4795
- logRequestFields({
4796
- path: c.req.path,
4797
- body: parsedForLog,
4798
- betaHeader: c.req.header("anthropic-beta")
5138
+ function parseTools(tools) {
5139
+ if (!Array.isArray(tools) || tools.length === 0) return void 0;
5140
+ const out = [];
5141
+ for (const tool of tools) {
5142
+ if (!tool || typeof tool !== "object") continue;
5143
+ const t = tool;
5144
+ if (typeof t.name !== "string" || t.name.length === 0) continue;
5145
+ const schema = t.input_schema ?? t.parameters;
5146
+ out.push({
5147
+ name: t.name,
5148
+ description: typeof t.description === "string" ? t.description : void 0,
5149
+ parameters: schema && typeof schema === "object" ? schema : {
5150
+ type: "object",
5151
+ properties: {}
5152
+ }
4799
5153
  });
4800
5154
  }
4801
- if (state.manualApprove) await awaitApproval();
4802
- const betaHeaders = extractBetaHeaders(c);
4803
- const advisorEnabled = isAdvisorRequested(c.req.header("anthropic-beta"));
4804
- let finalBody = await processWebSearch(rawBody);
4805
- finalBody = sanitizeAnthropicBody(finalBody);
4806
- if (advisorEnabled) {
4807
- finalBody = injectAdvisorTool(finalBody);
4808
- consola.info("ADVISOR enabled for this request — injecting __anthropic_advisor tool; will translate tool_use → server_tool_use{advisor} on the SSE stream");
5155
+ return out.length > 0 ? out : void 0;
5156
+ }
5157
+ function parseToolChoice(toolChoice) {
5158
+ if (!toolChoice || typeof toolChoice !== "object") return void 0;
5159
+ const tc = toolChoice;
5160
+ switch (tc.type) {
5161
+ case "auto": return "auto";
5162
+ case "any": return "required";
5163
+ case "none": return "none";
5164
+ case "tool": return typeof tc.name === "string" && tc.name.length > 0 ? {
5165
+ type: "function",
5166
+ name: tc.name
5167
+ } : void 0;
5168
+ default: return;
4809
5169
  }
4810
- if (finalBody.includes("\"mcp_servers\"")) try {
4811
- const probe = JSON.parse(finalBody);
4812
- if (Array.isArray(probe.mcp_servers) && probe.mcp_servers.length > 0) return c.json({
4813
- type: "error",
4814
- error: {
4815
- type: "invalid_request_error",
4816
- message: "Inline `mcp_servers` body field is not supported by github-router. Configure remote MCP servers as local stdio entries in `~/.claude/mcp.json` instead — Claude Code will spawn them locally and the proxy passes their tool calls through transparently. (https://docs.claude.com/en/docs/claude-code/mcp)"
4817
- }
4818
- }, 400);
4819
- } catch {}
5170
+ }
5171
+ /**
5172
+ * Anthropic carries `disable_parallel_tool_use` on the `tool_choice` object.
5173
+ * Returns `false` (the wire signal to disable parallel tool calls) only when it
5174
+ * is explicitly `true`; `undefined` otherwise, so the payload builders omit the
5175
+ * field rather than ever sending `parallel_tool_calls: true`.
5176
+ */
5177
+ function parseDisableParallelToolUse(toolChoice) {
5178
+ if (!toolChoice || typeof toolChoice !== "object") return void 0;
5179
+ return toolChoice.disable_parallel_tool_use === true ? false : void 0;
5180
+ }
5181
+ /** Default absent Anthropic `thinking` to high effort, clamped by the model.
5182
+ * Returns undefined for a model that advertises NO `reasoning_effort` allowlist
5183
+ * — such a model may not support reasoning at all, so forcing an effort could
5184
+ * 400; leaving it unset preserves the pre-default safe behavior for that case. */
5185
+ function defaultReasoningEffort(model) {
5186
+ const supported = model?.capabilities?.supports?.reasoning_effort;
5187
+ return Array.isArray(supported) && supported.length > 0 ? clampEffort("high", supported) : void 0;
5188
+ }
5189
+ /**
5190
+ * Map Anthropic `thinking` to a Responses reasoning effort, clamped to the
5191
+ * model's `reasoning_effort` allowlist. Returns undefined when thinking is
5192
+ * absent/disabled/non-enabled; the absent default is applied at the call site.
5193
+ */
5194
+ function parseReasoningEffort(thinking, model) {
5195
+ if (!thinking || typeof thinking !== "object") return void 0;
5196
+ const t = thinking;
5197
+ if (t.type !== "enabled") return void 0;
5198
+ const bucketed = bucketEffort(t.budget_tokens);
5199
+ const supported = model?.capabilities?.supports?.reasoning_effort;
5200
+ return Array.isArray(supported) && supported.length > 0 ? clampEffort(bucketed, supported) : bucketed;
5201
+ }
5202
+ /**
5203
+ * Parse an already-JSON-parsed Anthropic Messages body into the neutral shape.
5204
+ * `resolvedModel` is the catalog id the request will run on; `model` its
5205
+ * catalog entry (for the reasoning-effort allowlist).
5206
+ */
5207
+ function parseAnthropicRequest(body, resolvedModel, model) {
5208
+ const messages = [];
5209
+ if (Array.isArray(body.messages)) {
5210
+ for (const msg of body.messages) if (msg && typeof msg === "object") for (const n of anthropicMessageToNeutral(msg)) messages.push(n);
5211
+ }
5212
+ const maxTokens = typeof body.max_tokens === "number" && body.max_tokens > 0 ? body.max_tokens : void 0;
5213
+ const stopSequences = Array.isArray(body.stop_sequences) ? body.stop_sequences.filter((s) => typeof s === "string") : void 0;
5214
+ const tools = parseTools(body.tools);
5215
+ return {
5216
+ model: resolvedModel,
5217
+ instructions: appendFileToolGuidance(flattenSystem(body.system), tools),
5218
+ messages,
5219
+ tools,
5220
+ toolChoice: parseToolChoice(body.tool_choice),
5221
+ parallelToolCalls: parseDisableParallelToolUse(body.tool_choice),
5222
+ reasoningEffort: body.thinking === void 0 ? defaultReasoningEffort(model) : parseReasoningEffort(body.thinking, model),
5223
+ maxOutputTokens: maxTokens,
5224
+ stopSequences: stopSequences && stopSequences.length > 0 ? stopSequences : void 0,
5225
+ stream: body.stream === true
5226
+ };
5227
+ }
5228
+ /** Build the Copilot `/responses` payload from a parsed Anthropic request. */
5229
+ function parsedToResponsesPayload(parsed) {
5230
+ return assembleResponsesPayload({
5231
+ model: parsed.model,
5232
+ instructions: parsed.instructions,
5233
+ messages: parsed.messages,
5234
+ tools: parsed.tools,
5235
+ toolChoice: parsed.toolChoice,
5236
+ reasoningEffort: parsed.reasoningEffort,
5237
+ maxOutputTokens: parsed.maxOutputTokens,
5238
+ stopSequences: parsed.stopSequences,
5239
+ parallelToolCalls: parsed.parallelToolCalls,
5240
+ stream: parsed.stream
5241
+ });
5242
+ }
5243
+
5244
+ //#endregion
5245
+ //#region src/lib/anthropic-translate/chat-request.ts
5246
+ /** Build the `data:` URI (base64) or return the verbatim URL for an image part. */
5247
+ function imageUrlFor(part) {
5248
+ if (typeof part.url === "string" && part.url.length > 0) return part.url;
5249
+ return `data:${part.mimeType ?? "image/png"};base64,${part.data ?? ""}`;
5250
+ }
5251
+ /**
5252
+ * A brief inline note standing in for a document on the chat path. Copilot's
5253
+ * `/chat/completions` rejects file content parts (`type` must be `image_url` or
5254
+ * `text` → HTTP 400), so a document (PDF) can't be forwarded to a Gemini model;
5255
+ * the note keeps the document from being silently dropped and tells the model
5256
+ * one was provided but is unavailable, instead of 400ing the request.
5257
+ *
5258
+ * The note is wrapped in leading + trailing newlines so it is always DELIMITED
5259
+ * from adjacent user text — in the string-collapse branch it can't glue onto a
5260
+ * neighboring text run (`...[model]what is this?`), and in the content-parts
5261
+ * branch it stands as its own line. Regular text-to-text concatenation is left
5262
+ * untouched (only the note carries the delimiter), so wire order is preserved.
5263
+ */
5264
+ function documentNote(part) {
5265
+ return `\n[document "${part.filename ?? "document"}" attached but not supported for this model]\n`;
5266
+ }
5267
+ /**
5268
+ * A user turn: plain string when there are no images; otherwise OpenAI content
5269
+ * parts (`text` / `image_url`). Mirrors `neutralUserToResponses` so the two
5270
+ * shims encode a multimodal user turn the same way.
5271
+ */
5272
+ function neutralUserToChat(m) {
5273
+ if (typeof m.content === "string") return {
5274
+ role: "user",
5275
+ content: m.content
5276
+ };
5277
+ if (!m.content.some((c) => c.type === "image")) {
5278
+ let text = "";
5279
+ for (const c of m.content) if (c.type === "text") text += c.text;
5280
+ else if (c.type === "document") text += documentNote(c);
5281
+ return {
5282
+ role: "user",
5283
+ content: text
5284
+ };
5285
+ }
5286
+ const parts = [];
5287
+ for (const c of m.content) if (c.type === "text") parts.push({
5288
+ type: "text",
5289
+ text: c.text
5290
+ });
5291
+ else if (c.type === "image") parts.push({
5292
+ type: "image_url",
5293
+ image_url: { url: imageUrlFor(c) }
5294
+ });
5295
+ else if (c.type === "document") parts.push({
5296
+ type: "text",
5297
+ text: documentNote(c)
5298
+ });
5299
+ return {
5300
+ role: "user",
5301
+ content: parts
5302
+ };
5303
+ }
5304
+ /**
5305
+ * An assistant turn: text parts collapse into `content`, tool_use parts become
5306
+ * OpenAI `tool_calls`. Unlike the Responses path (which flushes text before each
5307
+ * call to preserve interleaving), the chat wire shape carries all text on
5308
+ * `content` and all calls on `tool_calls`, so ordering within the turn is not
5309
+ * representable — matching how OpenAI itself echoes an assistant turn. When the
5310
+ * turn is tool-calls-only, `content` is `null` (OpenAI convention).
5311
+ */
5312
+ function neutralAssistantToChat(m) {
5313
+ let text = "";
5314
+ const toolCalls = [];
5315
+ for (const c of m.content) if (c.type === "text") text += c.text;
5316
+ else if (c.type === "toolCall") toolCalls.push({
5317
+ id: c.id,
5318
+ type: "function",
5319
+ function: {
5320
+ name: c.name,
5321
+ arguments: JSON.stringify(c.arguments ?? {})
5322
+ }
5323
+ });
5324
+ const msg = {
5325
+ role: "assistant",
5326
+ content: text.length > 0 ? text : toolCalls.length > 0 ? null : ""
5327
+ };
5328
+ if (toolCalls.length > 0) msg.tool_calls = toolCalls;
5329
+ return msg;
5330
+ }
5331
+ /** Translate one neutral message into a single chat/completions message. */
5332
+ function neutralMessageToChat(m) {
5333
+ if (m.role === "user") return neutralUserToChat(m);
5334
+ if (m.role === "assistant") return neutralAssistantToChat(m);
5335
+ return {
5336
+ role: "tool",
5337
+ tool_call_id: m.toolCallId,
5338
+ content: m.output
5339
+ };
5340
+ }
5341
+ function neutralToolsToChat(tools) {
5342
+ if (!tools || tools.length === 0) return void 0;
5343
+ return tools.map((t) => ({
5344
+ type: "function",
5345
+ function: {
5346
+ name: t.name,
5347
+ description: t.description,
5348
+ parameters: t.parameters ?? {
5349
+ type: "object",
5350
+ properties: {}
5351
+ }
5352
+ }
5353
+ }));
5354
+ }
5355
+ /**
5356
+ * Convert the parsed (Responses-flat) `tool_choice` to the chat NESTED form.
5357
+ * A forced tool is `{type:"function", function:{name}}` on chat/completions —
5358
+ * distinct from the Responses flat `{type:"function", name}`. `"auto"` /
5359
+ * `"required"` / `"none"` pass through unchanged.
5360
+ */
5361
+ function toolChoiceToChat(tc) {
5362
+ if (tc === void 0) return void 0;
5363
+ if (typeof tc === "string") return tc === "auto" || tc === "none" || tc === "required" ? tc : void 0;
5364
+ if (tc.type === "function" && typeof tc.name === "string" && tc.name.length > 0) return {
5365
+ type: "function",
5366
+ function: { name: tc.name }
5367
+ };
5368
+ }
5369
+ /** Assemble the `/chat/completions` payload from a parsed Anthropic request. */
5370
+ function parsedToChatPayload(parsed) {
5371
+ const messages = [];
5372
+ if (parsed.instructions) messages.push({
5373
+ role: "system",
5374
+ content: parsed.instructions
5375
+ });
5376
+ for (const m of parsed.messages) messages.push(neutralMessageToChat(m));
5377
+ const payload = {
5378
+ model: parsed.model,
5379
+ messages,
5380
+ stream: parsed.stream
5381
+ };
5382
+ const tools = neutralToolsToChat(parsed.tools);
5383
+ if (tools && tools.length > 0) {
5384
+ payload.tools = tools;
5385
+ payload.tool_choice = toolChoiceToChat(parsed.toolChoice) ?? "auto";
5386
+ }
5387
+ if (parsed.reasoningEffort && parsed.reasoningEffort !== "off") payload.reasoning_effort = parsed.reasoningEffort;
5388
+ if (typeof parsed.maxOutputTokens === "number" && parsed.maxOutputTokens > 0) payload.max_tokens = parsed.maxOutputTokens;
5389
+ if (parsed.stopSequences && parsed.stopSequences.length > 0) payload.stop = parsed.stopSequences;
5390
+ if (parsed.parallelToolCalls === false) payload.parallel_tool_calls = false;
5391
+ return payload;
5392
+ }
5393
+
5394
+ //#endregion
5395
+ //#region src/lib/anthropic-translate/chat-egress.ts
5396
+ /** Synthesize a matchable Anthropic tool_use id when the upstream id is absent. */
5397
+ function makeToolUseId$1() {
5398
+ return `toolu_${randomUUID().replace(/-/g, "")}`;
5399
+ }
5400
+ function parseToolArgs$1(raw) {
5401
+ if (typeof raw !== "string" || raw.trim().length === 0) return {};
5402
+ try {
5403
+ const parsed = JSON.parse(raw);
5404
+ if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return parsed;
5405
+ } catch {}
5406
+ return {};
5407
+ }
5408
+ function anthropicUsageFromChat(u) {
5409
+ if (!u) return {};
5410
+ return {
5411
+ input_tokens: u.prompt_tokens ?? 0,
5412
+ output_tokens: u.completion_tokens ?? 0,
5413
+ cache_read_input_tokens: u.prompt_tokens_details?.cached_tokens ?? 0,
5414
+ cache_creation_input_tokens: 0
5415
+ };
5416
+ }
5417
+ /**
5418
+ * Map a chat/completions `finish_reason` to an Anthropic stop_reason. A
5419
+ * truncated (`length`) response is `max_tokens` even when a partial tool call
5420
+ * is present — the response was cut — mirroring the Responses egress precedence.
5421
+ * `tool_calls` (or any buffered tool) → `tool_use`; everything else (`stop`,
5422
+ * `content_filter`, null) → `end_turn`.
5423
+ */
5424
+ function chatStopReason(finishReason, sawTool) {
5425
+ if (finishReason === "length") return "max_tokens";
5426
+ if (finishReason === "tool_calls" || sawTool) return "tool_use";
5427
+ return "end_turn";
5428
+ }
5429
+ /**
5430
+ * Map a non-streaming chat/completions object to an Anthropic Messages object.
5431
+ * The first choice's `message.content` becomes a text block (when non-empty)
5432
+ * and each `message.tool_calls[]` becomes a tool_use block.
5433
+ */
5434
+ function chatResponseToAnthropicMessage(resp, modelId) {
5435
+ const choice = resp.choices?.[0];
5436
+ const content = [];
5437
+ let sawTool = false;
5438
+ const message = choice?.message;
5439
+ if (message) {
5440
+ if (typeof message.content === "string" && message.content.length > 0) content.push({
5441
+ type: "text",
5442
+ text: message.content
5443
+ });
5444
+ if (Array.isArray(message.tool_calls)) for (const tc of message.tool_calls) {
5445
+ sawTool = true;
5446
+ const rawId = typeof tc.id === "string" ? tc.id : "";
5447
+ content.push({
5448
+ type: "tool_use",
5449
+ id: rawId.length > 0 ? rawId : makeToolUseId$1(),
5450
+ name: typeof tc.function?.name === "string" ? tc.function.name : "",
5451
+ input: parseToolArgs$1(tc.function?.arguments)
5452
+ });
5453
+ }
5454
+ }
5455
+ const usage = anthropicUsageFromChat(resp.usage);
5456
+ const stopReason = chatStopReason(choice?.finish_reason ?? null, sawTool);
5457
+ return {
5458
+ id: typeof resp.id === "string" && resp.id.length > 0 ? `msg_${resp.id}` : makeMessageId(),
5459
+ type: "message",
5460
+ role: "assistant",
5461
+ model: modelId,
5462
+ content,
5463
+ stop_reason: stopReason,
5464
+ stop_sequence: null,
5465
+ usage: {
5466
+ input_tokens: usage.input_tokens ?? 0,
5467
+ output_tokens: usage.output_tokens ?? 0,
5468
+ cache_read_input_tokens: usage.cache_read_input_tokens ?? 0,
5469
+ cache_creation_input_tokens: usage.cache_creation_input_tokens ?? 0
5470
+ }
5471
+ };
5472
+ }
5473
+ /**
5474
+ * Streaming synthesizer: consume a chat/completions SSE iterable, yield the
5475
+ * Anthropic event sequence. Emits `message_start` first, streams text live,
5476
+ * buffers tool calls per OpenAI array index and flushes them atomically at
5477
+ * end-of-stream (in numeric index order), then a terminal `message_delta`
5478
+ * (accumulated usage + stop_reason) and `message_stop`. The `[DONE]` sentinel
5479
+ * is the authoritative clean-end marker: a stream that ends WITHOUT it is
5480
+ * treated as truncated and throws so the stream adapter can emit a terminal
5481
+ * `event: error`. A clean `[DONE]` carrying no chunk-level `finish_reason`
5482
+ * still completes cleanly (stop_reason from any buffered tool, else `end_turn`).
5483
+ */
5484
+ async function* synthAnthropicFromChat(upstream, opts) {
5485
+ const messageId = opts.messageId ?? makeMessageId();
5486
+ let nextIndex = 0;
5487
+ let activeTextIndex = null;
5488
+ const toolByIndex = /* @__PURE__ */ new Map();
5489
+ let usageIn = 0;
5490
+ let usageOut = 0;
5491
+ let usageCacheRead = 0;
5492
+ let finishReason = null;
5493
+ let sawDone = false;
5494
+ yield makeMessageStart(messageId, opts.modelId);
5495
+ for await (const evt of upstream) {
5496
+ const data = evt?.data;
5497
+ if (data == null) continue;
5498
+ if (data === "[DONE]") {
5499
+ sawDone = true;
5500
+ break;
5501
+ }
5502
+ let chunk;
5503
+ try {
5504
+ chunk = JSON.parse(data);
5505
+ } catch {
5506
+ continue;
5507
+ }
5508
+ if (chunk.usage) {
5509
+ usageIn = Math.max(usageIn, chunk.usage.prompt_tokens ?? 0);
5510
+ usageOut = Math.max(usageOut, chunk.usage.completion_tokens ?? 0);
5511
+ usageCacheRead = Math.max(usageCacheRead, chunk.usage.prompt_tokens_details?.cached_tokens ?? 0);
5512
+ }
5513
+ const choice = chunk.choices?.[0];
5514
+ if (!choice) continue;
5515
+ const delta = choice.delta;
5516
+ if (delta && typeof delta.content === "string" && delta.content.length > 0) {
5517
+ if (activeTextIndex == null) {
5518
+ activeTextIndex = nextIndex++;
5519
+ yield makeContentBlockStart(activeTextIndex, {
5520
+ type: "text",
5521
+ text: ""
5522
+ });
5523
+ }
5524
+ yield makeTextDelta(activeTextIndex, delta.content);
5525
+ }
5526
+ if (delta && Array.isArray(delta.tool_calls) && delta.tool_calls.length > 0) {
5527
+ if (activeTextIndex != null) {
5528
+ yield makeContentBlockStop(activeTextIndex);
5529
+ activeTextIndex = null;
5530
+ }
5531
+ for (const tcd of delta.tool_calls) {
5532
+ if (tcd == null || typeof tcd.index !== "number") continue;
5533
+ let entry = toolByIndex.get(tcd.index);
5534
+ if (!entry) {
5535
+ entry = {
5536
+ id: "",
5537
+ name: "",
5538
+ args: ""
5539
+ };
5540
+ toolByIndex.set(tcd.index, entry);
5541
+ }
5542
+ if (typeof tcd.id === "string" && tcd.id.length > 0) entry.id = tcd.id;
5543
+ const name$1 = tcd.function?.name;
5544
+ if (typeof name$1 === "string" && name$1.length > 0) entry.name = name$1;
5545
+ const argDelta = tcd.function?.arguments;
5546
+ if (typeof argDelta === "string" && argDelta.length > 0) entry.args += argDelta;
5547
+ }
5548
+ }
5549
+ if (choice.finish_reason != null) finishReason = choice.finish_reason;
5550
+ }
5551
+ if (!sawDone) throw new Error("chat stream ended without a [DONE] sentinel (truncated)");
5552
+ if (activeTextIndex != null) {
5553
+ yield makeContentBlockStop(activeTextIndex);
5554
+ activeTextIndex = null;
5555
+ }
5556
+ let sawTool = false;
5557
+ const orderedTools = [...toolByIndex.entries()].sort((a, b) => a[0] - b[0]);
5558
+ for (const [, entry] of orderedTools) {
5559
+ sawTool = true;
5560
+ const index = nextIndex++;
5561
+ yield makeContentBlockStart(index, {
5562
+ type: "tool_use",
5563
+ id: entry.id.length > 0 ? entry.id : makeToolUseId$1(),
5564
+ name: entry.name,
5565
+ input: {}
5566
+ });
5567
+ yield makeInputJsonDelta(index, JSON.stringify(parseToolArgs$1(entry.args)));
5568
+ yield makeContentBlockStop(index);
5569
+ }
5570
+ yield makeMessageDelta(chatStopReason(finishReason, sawTool), null, {
5571
+ input_tokens: usageIn,
5572
+ output_tokens: usageOut,
5573
+ cache_read_input_tokens: usageCacheRead,
5574
+ cache_creation_input_tokens: 0
5575
+ });
5576
+ yield makeMessageStop();
5577
+ }
5578
+
5579
+ //#endregion
5580
+ //#region src/lib/anthropic-translate/responses-egress.ts
5581
+ /**
5582
+ * Stable map key for a `/responses` output item: prefer `output_index`
5583
+ * (constant per item), fall back to the opaque id only when absent. Namespaced
5584
+ * so a numeric index and a string id can never collide.
5585
+ */
5586
+ function responsesKey(outputIndex, fallbackId) {
5587
+ if (typeof outputIndex === "number") return `oi:${outputIndex}`;
5588
+ if (typeof fallbackId === "string" && fallbackId.length > 0) return `id:${fallbackId}`;
5589
+ }
5590
+ /** Synthesize a matchable Anthropic tool_use id when the upstream id is absent. */
5591
+ function makeToolUseId() {
5592
+ return `toolu_${randomUUID().replace(/-/g, "")}`;
5593
+ }
5594
+ /** First non-empty string among the candidates, or "" when none qualifies. */
5595
+ function firstNonEmpty(...vals) {
5596
+ for (const v of vals) if (typeof v === "string" && v.length > 0) return v;
5597
+ return "";
5598
+ }
5599
+ function anthropicUsageFromResponses(u) {
5600
+ if (!u) return {};
5601
+ return {
5602
+ input_tokens: u.input_tokens ?? 0,
5603
+ output_tokens: u.output_tokens ?? 0,
5604
+ cache_read_input_tokens: u.input_tokens_details?.cached_tokens ?? 0,
5605
+ cache_creation_input_tokens: 0
5606
+ };
5607
+ }
5608
+ function parseToolArgs(raw) {
5609
+ if (typeof raw !== "string" || raw.trim().length === 0) return {};
5610
+ try {
5611
+ const parsed = JSON.parse(raw);
5612
+ if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) return parsed;
5613
+ } catch {}
5614
+ return {};
5615
+ }
5616
+ /**
5617
+ * Map a non-streaming Responses object to an Anthropic Messages object.
5618
+ *
5619
+ * stop_reason precedence for a completed non-streaming response:
5620
+ * an incomplete/max-output response is `max_tokens` even if a partial tool call
5621
+ * is present (the response was truncated), else a function_call → `tool_use`,
5622
+ * else `end_turn`.
5623
+ */
5624
+ function responsesResponseToAnthropicMessage(resp, modelId) {
5625
+ const output = Array.isArray(resp.output) ? resp.output : [];
5626
+ const content = [];
5627
+ let sawToolUse = false;
5628
+ for (const rawItem of output) {
5629
+ if (!rawItem || typeof rawItem !== "object") continue;
5630
+ const item = rawItem;
5631
+ if (item.type === "message") {
5632
+ let text = "";
5633
+ if (Array.isArray(item.content)) {
5634
+ for (const part of item.content) if (part && typeof part === "object" && part.type === "output_text" && typeof part.text === "string") text += part.text;
5635
+ }
5636
+ if (text.length > 0) content.push({
5637
+ type: "text",
5638
+ text
5639
+ });
5640
+ } else if (item.type === "function_call") {
5641
+ sawToolUse = true;
5642
+ const rawId = typeof item.call_id === "string" && item.call_id.length > 0 ? item.call_id : typeof item.id === "string" && item.id.length > 0 ? item.id : "";
5643
+ content.push({
5644
+ type: "tool_use",
5645
+ id: rawId.length > 0 ? rawId : makeToolUseId(),
5646
+ name: typeof item.name === "string" ? item.name : "",
5647
+ input: parseToolArgs(item.arguments)
5648
+ });
5649
+ } else if (item.type === "reasoning") {
5650
+ let thinking = "";
5651
+ if (Array.isArray(item.summary)) {
5652
+ for (const part of item.summary) if (part && typeof part === "object" && typeof part.text === "string") thinking += part.text;
5653
+ }
5654
+ if (thinking.length > 0) content.push({
5655
+ type: "thinking",
5656
+ thinking,
5657
+ signature: typeof item.encrypted_content === "string" ? item.encrypted_content : ""
5658
+ });
5659
+ }
5660
+ }
5661
+ const usage = anthropicUsageFromResponses(resp.usage);
5662
+ const stopReason = resp.status === "incomplete" && resp.incomplete_details?.reason === "max_output_tokens" ? "max_tokens" : sawToolUse ? "tool_use" : "end_turn";
5663
+ return {
5664
+ id: typeof resp.id === "string" && resp.id.length > 0 ? `msg_${resp.id}` : makeMessageId(),
5665
+ type: "message",
5666
+ role: "assistant",
5667
+ model: modelId,
5668
+ content,
5669
+ stop_reason: stopReason,
5670
+ stop_sequence: null,
5671
+ usage: {
5672
+ input_tokens: usage.input_tokens ?? 0,
5673
+ output_tokens: usage.output_tokens ?? 0,
5674
+ cache_read_input_tokens: usage.cache_read_input_tokens ?? 0,
5675
+ cache_creation_input_tokens: usage.cache_creation_input_tokens ?? 0
5676
+ }
5677
+ };
5678
+ }
5679
+ /**
5680
+ * Streaming synthesizer: consume a `/responses` SSE iterable, yield the
5681
+ * Anthropic event sequence. Emits `message_start` first, then content blocks in
5682
+ * item order, then a terminal `message_delta` (accumulated usage + stop_reason)
5683
+ * and `message_stop`. A `response.failed` throws so the stream adapter can emit
5684
+ * a terminal `event: error`; a stream that ends WITHOUT a terminal event
5685
+ * (`completed`/`incomplete`/`failed`) is treated as truncated and throws too.
5686
+ */
5687
+ async function* synthAnthropicFromResponses(upstream, opts) {
5688
+ const messageId = opts.messageId ?? makeMessageId();
5689
+ const q = [];
5690
+ let nextIndex = 0;
5691
+ let current = null;
5692
+ const toolByKey = /* @__PURE__ */ new Map();
5693
+ const thinkingByKey = /* @__PURE__ */ new Map();
5694
+ const textByKey = /* @__PURE__ */ new Map();
5695
+ let usageIn = 0;
5696
+ let usageOut = 0;
5697
+ let usageCacheRead = 0;
5698
+ let sawTool = false;
5699
+ let hitMaxTokens = false;
5700
+ let sawTerminal = false;
5701
+ const closeCurrent = () => {
5702
+ if (!current) return;
5703
+ q.push(makeContentBlockStop(current.index));
5704
+ current = null;
5705
+ };
5706
+ const currentKind = () => current ? current.kind : null;
5707
+ const currentIndex = () => current ? current.index : null;
5708
+ const ensureTextState = (key) => {
5709
+ const existing = textByKey.get(key);
5710
+ if (existing != null && current?.kind === "text" && current.index === existing.index) return existing;
5711
+ closeCurrent();
5712
+ const index = nextIndex++;
5713
+ const state$1 = {
5714
+ index,
5715
+ emitted: ""
5716
+ };
5717
+ textByKey.set(key, state$1);
5718
+ current = {
5719
+ index,
5720
+ kind: "text"
5721
+ };
5722
+ q.push(makeContentBlockStart(index, {
5723
+ type: "text",
5724
+ text: ""
5725
+ }));
5726
+ return state$1;
5727
+ };
5728
+ const ensureThinking = (key) => {
5729
+ if (current?.kind === "thinking" && thinkingByKey.get(key) === current.index) return current.index;
5730
+ closeCurrent();
5731
+ const index = nextIndex++;
5732
+ current = {
5733
+ index,
5734
+ kind: "thinking"
5735
+ };
5736
+ thinkingByKey.set(key, index);
5737
+ q.push(makeContentBlockStart(index, {
5738
+ type: "thinking",
5739
+ thinking: ""
5740
+ }));
5741
+ return index;
5742
+ };
5743
+ const emitTool = (t) => {
5744
+ if (t.emitted) return;
5745
+ closeCurrent();
5746
+ const index = nextIndex++;
5747
+ q.push(makeContentBlockStart(index, {
5748
+ type: "tool_use",
5749
+ id: t.id,
5750
+ name: t.name,
5751
+ input: {}
5752
+ }));
5753
+ const args = t.argsBuffer.length > 0 ? t.argsBuffer : "{}";
5754
+ q.push(makeInputJsonDelta(index, args));
5755
+ q.push(makeContentBlockStop(index));
5756
+ t.emitted = true;
5757
+ sawTool = true;
5758
+ };
5759
+ q.push(makeMessageStart(messageId, opts.modelId));
5760
+ for (const e of q) yield e;
5761
+ q.length = 0;
5762
+ for await (const evt of upstream) {
5763
+ const data = evt?.data;
5764
+ if (data == null) continue;
5765
+ if (data === "[DONE]") break;
5766
+ let ev;
5767
+ try {
5768
+ ev = JSON.parse(data);
5769
+ } catch {
5770
+ continue;
5771
+ }
5772
+ switch (ev.type) {
5773
+ case "response.output_text.delta": {
5774
+ const d = ev.delta;
5775
+ if (typeof d !== "string" || d.length === 0) break;
5776
+ const state$1 = ensureTextState(responsesKey(ev.output_index, ev.item_id) ?? "text");
5777
+ state$1.emitted += d;
5778
+ q.push(makeTextDelta(state$1.index, d));
5779
+ break;
5780
+ }
5781
+ case "response.output_text.done": {
5782
+ const key = responsesKey(ev.output_index, ev.item_id) ?? "text";
5783
+ const fullText = typeof ev.text === "string" ? ev.text : "";
5784
+ const existing = textByKey.get(key);
5785
+ if (existing == null) {
5786
+ if (fullText.length > 0) {
5787
+ const state$1 = ensureTextState(key);
5788
+ state$1.emitted = fullText;
5789
+ q.push(makeTextDelta(state$1.index, fullText));
5790
+ closeCurrent();
5791
+ }
5792
+ } else if (currentKind() === "text" && currentIndex() === existing.index) {
5793
+ if (fullText.length > existing.emitted.length && fullText.startsWith(existing.emitted)) {
5794
+ q.push(makeTextDelta(existing.index, fullText.slice(existing.emitted.length)));
5795
+ existing.emitted = fullText;
5796
+ }
5797
+ closeCurrent();
5798
+ }
5799
+ break;
5800
+ }
5801
+ case "response.reasoning_summary_text.delta":
5802
+ case "response.reasoning_text.delta": {
5803
+ const d = ev.delta;
5804
+ if (typeof d !== "string" || d.length === 0) break;
5805
+ const key = responsesKey(ev.output_index, ev.item_id) ?? "reasoning";
5806
+ q.push(makeThinkingDelta(ensureThinking(key), d));
5807
+ break;
5808
+ }
5809
+ case "response.reasoning_summary_text.done":
5810
+ case "response.reasoning_text.done": break;
5811
+ case "response.output_item.added": {
5812
+ const item = ev.item;
5813
+ if (item?.type === "function_call") {
5814
+ const key = responsesKey(ev.output_index, item.id);
5815
+ if (key == null || toolByKey.has(key)) break;
5816
+ const toolId = firstNonEmpty(item.call_id, item.id);
5817
+ toolByKey.set(key, {
5818
+ id: toolId.length > 0 ? toolId : makeToolUseId(),
5819
+ name: item.name ?? "",
5820
+ argsBuffer: "",
5821
+ emitted: false
5822
+ });
5823
+ sawTool = true;
5824
+ }
5825
+ break;
5826
+ }
5827
+ case "response.function_call_arguments.delta": {
5828
+ const key = responsesKey(ev.output_index, ev.item_id);
5829
+ if (key == null) break;
5830
+ const t = toolByKey.get(key);
5831
+ if (!t || t.emitted) break;
5832
+ const d = ev.delta;
5833
+ if (typeof d !== "string" || d.length === 0) break;
5834
+ t.argsBuffer += d;
5835
+ break;
5836
+ }
5837
+ case "response.function_call_arguments.done": {
5838
+ const key = responsesKey(ev.output_index, ev.item_id);
5839
+ if (key == null) break;
5840
+ const t = toolByKey.get(key);
5841
+ if (!t || t.emitted) break;
5842
+ if (typeof ev.arguments === "string") t.argsBuffer = ev.arguments;
5843
+ break;
5844
+ }
5845
+ case "response.output_item.done": {
5846
+ const item = ev.item;
5847
+ if (item?.type === "function_call") {
5848
+ const key = responsesKey(ev.output_index, item.id);
5849
+ if (key == null) break;
5850
+ const t = toolByKey.get(key);
5851
+ if (!t || t.emitted) break;
5852
+ const doneId = firstNonEmpty(item.call_id, item.id);
5853
+ if (doneId.length > 0) t.id = doneId;
5854
+ if (typeof item.name === "string" && item.name.length > 0) t.name = item.name;
5855
+ if (typeof item.arguments === "string") t.argsBuffer = item.arguments;
5856
+ emitTool(t);
5857
+ } else if (item?.type === "reasoning") {
5858
+ const key = responsesKey(ev.output_index, item.id);
5859
+ const idx = key != null ? thinkingByKey.get(key) : void 0;
5860
+ if (idx != null && currentKind() === "thinking" && currentIndex() === idx) {
5861
+ if (typeof item.encrypted_content === "string" && item.encrypted_content.length > 0) q.push(makeSignatureDelta(idx, item.encrypted_content));
5862
+ closeCurrent();
5863
+ }
5864
+ }
5865
+ break;
5866
+ }
5867
+ case "response.completed":
5868
+ case "response.incomplete": {
5869
+ sawTerminal = true;
5870
+ const u = ev.response?.usage;
5871
+ if (u) {
5872
+ usageIn = Math.max(usageIn, u.input_tokens ?? 0);
5873
+ usageOut = Math.max(usageOut, u.output_tokens ?? 0);
5874
+ usageCacheRead = Math.max(usageCacheRead, u.input_tokens_details?.cached_tokens ?? 0);
5875
+ }
5876
+ if (ev.type === "response.incomplete" && ev.response?.incomplete_details?.reason === "max_output_tokens") hitMaxTokens = true;
5877
+ break;
5878
+ }
5879
+ case "response.failed": throw new Error(ev.response?.error?.message ?? "response.failed");
5880
+ default: break;
5881
+ }
5882
+ for (const e of q) yield e;
5883
+ q.length = 0;
5884
+ }
5885
+ if (!sawTerminal) throw new Error("responses stream ended without a terminal event (truncated)");
5886
+ closeCurrent();
5887
+ for (const t of toolByKey.values()) if (!t.emitted) emitTool(t);
5888
+ const stopReason = hitMaxTokens ? "max_tokens" : sawTool ? "tool_use" : "end_turn";
5889
+ q.push(makeMessageDelta(stopReason, null, {
5890
+ input_tokens: usageIn,
5891
+ output_tokens: usageOut,
5892
+ cache_read_input_tokens: usageCacheRead,
5893
+ cache_creation_input_tokens: 0
5894
+ }));
5895
+ q.push(makeMessageStop());
5896
+ for (const e of q) yield e;
5897
+ q.length = 0;
5898
+ }
5899
+
5900
+ //#endregion
5901
+ //#region src/lib/anthropic-translate/classifier.ts
5902
+ /**
5903
+ * Match "claude"/"anthropic" as a delimiter-bounded segment anywhere in a model
5904
+ * id: at the start, or after a `/ _ . : -` path/version delimiter, and followed
5905
+ * by end-of-string, another such delimiter, or a digit. This catches catalog
5906
+ * aliases like `github/claude-3-7-sonnet` or `anthropic/…` whose vendor is
5907
+ * "github" and whose family is empty — where the token only surfaces mid-id —
5908
+ * while NOT firing on incidental substrings like `notclaude`. Deliberately
5909
+ * over-inclusive at the boundaries: we fail CLOSED toward Claude (→ passthrough)
5910
+ * so a real Claude model can never be diverted to the non-Claude shim.
5911
+ */
5912
+ const CLAUDE_ID_RE = /(^|[/_.:-])(claude|anthropic)(?=$|[/_.:-]|\d)/i;
5913
+ /**
5914
+ * True when the target is a Claude / Anthropic model. Matches on any of:
5915
+ * catalog vendor containing "anthropic", capability family containing "claude",
5916
+ * or a Claude/anthropic path-segment (see `CLAUDE_ID_RE`) in ANY id we hold —
5917
+ * the resolved id (`modelId`), the pre-resolution request id (`originalModelId`),
5918
+ * or the catalog entry's own id (`model.id`). Conservative by design: when in
5919
+ * doubt it returns true so a Claude request can never be diverted to the shim.
5920
+ */
5921
+ function isClaudeModel(modelId, model, originalModelId) {
5922
+ if (model) {
5923
+ if ((model.vendor?.toLowerCase() ?? "").includes("anthropic")) return true;
5924
+ if ((model.capabilities?.family?.toLowerCase() ?? "").includes("claude")) return true;
5925
+ }
5926
+ return [
5927
+ modelId,
5928
+ originalModelId,
5929
+ model?.id
5930
+ ].some((id) => typeof id === "string" && CLAUDE_ID_RE.test(id));
5931
+ }
5932
+ /**
5933
+ * Decide the route for a resolved model id + its catalog entry.
5934
+ *
5935
+ * - No model id, or a Claude model → "claude-passthrough" (existing behaviour).
5936
+ * - A non-Claude model whose catalog endpoint is `/responses` → "responses-shim".
5937
+ * - A non-Claude model whose catalog endpoint is `/chat/completions` (gemini and
5938
+ * any chat-default model) → "chat-shim".
5939
+ * - A non-Claude model absent from the catalog (so we can't confirm an endpoint)
5940
+ * → "claude-passthrough" (unchanged; we don't divert what we can't classify).
5941
+ *
5942
+ * `originalModelId` is the optional pre-resolution request id; when supplied it
5943
+ * is checked for Claude-likeness alongside the resolved id so an alias that
5944
+ * resolves to a non-Claude-looking id can't slip past.
5945
+ */
5946
+ function classifyMessagesRoute(modelId, model, originalModelId) {
5947
+ if (!modelId) return "claude-passthrough";
5948
+ if (isClaudeModel(modelId, model, originalModelId)) return "claude-passthrough";
5949
+ if (!model) return "claude-passthrough";
5950
+ const endpoint = pickEndpoint(model);
5951
+ if (endpoint === "responses") return "responses-shim";
5952
+ if (endpoint === "chat") return "chat-shim";
5953
+ return "claude-passthrough";
5954
+ }
5955
+
5956
+ //#endregion
5957
+ //#region src/lib/anthropic-translate/index.ts
5958
+ const STREAM_HEADERS = {
5959
+ "content-type": "text/event-stream",
5960
+ "cache-control": "no-cache",
5961
+ "transfer-encoding": "chunked",
5962
+ connection: "keep-alive"
5963
+ };
5964
+ function isAsyncIterable(x) {
5965
+ return x != null && typeof x[Symbol.asyncIterator] === "function";
5966
+ }
5967
+ /**
5968
+ * Handle a `/v1/messages` request targeting a non-Claude `/responses` model.
5969
+ * Returns a streaming or non-streaming Anthropic-format Response. Upstream
5970
+ * non-2xx / abort errors are thrown (as HTTPError) and handled by the route's
5971
+ * `forwardError`, exactly like the passthrough path.
5972
+ */
5973
+ async function handleNonClaudeResponses(c, opts) {
5974
+ const routePath = c.req.path;
5975
+ let body;
5976
+ try {
5977
+ body = JSON.parse(opts.rawBody);
5978
+ } catch {
5979
+ return c.json({
5980
+ type: "error",
5981
+ error: {
5982
+ type: "invalid_request_error",
5983
+ message: "Request body is not valid JSON"
5984
+ }
5985
+ }, 400);
5986
+ }
5987
+ const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
5988
+ const payload = parsedToResponsesPayload(parsed);
5989
+ if (consola.level >= 4) consola.debug(`Anthropic-translate → /responses model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
5990
+ if (parsed.stream) {
5991
+ const aborter = new AbortController();
5992
+ const result = await createResponses(payload, opts.model?.requestHeaders, aborter.signal, true);
5993
+ if (!isAsyncIterable(result)) throw new Error("Upstream /responses did not return an SSE stream (stream: true expected)");
5994
+ logRequest({
5995
+ method: "POST",
5996
+ path: routePath,
5997
+ model: opts.originalModel,
5998
+ resolvedModel: opts.modelId,
5999
+ status: 200,
6000
+ streaming: true
6001
+ }, opts.model, opts.startTime);
6002
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromResponses(result, { modelId: opts.modelId }), {
6003
+ routePath,
6004
+ onCancel: () => aborter.abort()
6005
+ });
6006
+ return new Response(stream, {
6007
+ status: 200,
6008
+ headers: STREAM_HEADERS
6009
+ });
6010
+ }
6011
+ const anthropic = responsesResponseToAnthropicMessage(await createResponses(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
6012
+ logRequest({
6013
+ method: "POST",
6014
+ path: routePath,
6015
+ model: opts.originalModel,
6016
+ resolvedModel: opts.modelId,
6017
+ inputTokens: anthropic.usage.input_tokens,
6018
+ outputTokens: anthropic.usage.output_tokens,
6019
+ status: 200
6020
+ }, opts.model, opts.startTime);
6021
+ return c.json(anthropic, 200);
6022
+ }
6023
+ /**
6024
+ * Handle a `/v1/messages` request targeting a non-Claude `/chat/completions`
6025
+ * model (gemini). Twin of `handleNonClaudeResponses` — same lifecycle, abort,
6026
+ * and logging contract, but assembles a chat/completions payload and translates
6027
+ * the chat response (object or SSE) back to the Anthropic wire shape.
6028
+ */
6029
+ async function handleNonClaudeChat(c, opts) {
6030
+ const routePath = c.req.path;
6031
+ let body;
6032
+ try {
6033
+ body = JSON.parse(opts.rawBody);
6034
+ } catch {
6035
+ return c.json({
6036
+ type: "error",
6037
+ error: {
6038
+ type: "invalid_request_error",
6039
+ message: "Request body is not valid JSON"
6040
+ }
6041
+ }, 400);
6042
+ }
6043
+ const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
6044
+ const payload = parsedToChatPayload(parsed);
6045
+ if (consola.level >= 4) consola.debug(`Anthropic-translate → /chat/completions model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
6046
+ if (parsed.stream) {
6047
+ const aborter = new AbortController();
6048
+ const result = await createChatCompletions(payload, opts.model?.requestHeaders, aborter.signal, true);
6049
+ if (!isAsyncIterable(result)) throw new Error("Upstream /chat/completions did not return an SSE stream (stream: true expected)");
6050
+ logRequest({
6051
+ method: "POST",
6052
+ path: routePath,
6053
+ model: opts.originalModel,
6054
+ resolvedModel: opts.modelId,
6055
+ status: 200,
6056
+ streaming: true
6057
+ }, opts.model, opts.startTime);
6058
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromChat(result, { modelId: opts.modelId }), {
6059
+ routePath,
6060
+ onCancel: () => aborter.abort()
6061
+ });
6062
+ return new Response(stream, {
6063
+ status: 200,
6064
+ headers: STREAM_HEADERS
6065
+ });
6066
+ }
6067
+ const anthropic = chatResponseToAnthropicMessage(await createChatCompletions(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
6068
+ logRequest({
6069
+ method: "POST",
6070
+ path: routePath,
6071
+ model: opts.originalModel,
6072
+ resolvedModel: opts.modelId,
6073
+ inputTokens: anthropic.usage.input_tokens,
6074
+ outputTokens: anthropic.usage.output_tokens,
6075
+ status: 200
6076
+ }, opts.model, opts.startTime);
6077
+ return c.json(anthropic, 200);
6078
+ }
6079
+
6080
+ //#endregion
6081
+ //#region src/routes/messages/handler.ts
6082
+ const isWebSearchTool$1 = (tool) => !!tool && typeof tool === "object" && (typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search");
6083
+ /**
6084
+ * Extract whitelisted beta headers from the incoming request to forward
6085
+ * to the Copilot API. VS Code sends these to enable extended features
6086
+ * like thinking, context management, and advanced tool use.
6087
+ */
6088
+ function extractBetaHeaders(c) {
6089
+ const headers = {};
6090
+ const anthropicBeta = c.req.header("anthropic-beta");
6091
+ if (anthropicBeta) {
6092
+ const filtered = filterBetaHeader(anthropicBeta);
6093
+ if (filtered) headers["anthropic-beta"] = filtered;
6094
+ }
6095
+ return headers;
6096
+ }
6097
+ /**
6098
+ * Extract the text content from the last user message for web search.
6099
+ * Handles both string content and content block arrays (multimodal).
6100
+ */
6101
+ function extractUserQuery$1(messages) {
6102
+ for (let i = messages.length - 1; i >= 0; i--) {
6103
+ const msg = messages[i];
6104
+ if (msg.role === "user") {
6105
+ if (typeof msg.content === "string") return msg.content;
6106
+ if (Array.isArray(msg.content)) {
6107
+ const textBlock = msg.content.find((block) => block.type === "text");
6108
+ if (textBlock?.text) return textBlock.text;
6109
+ }
6110
+ }
6111
+ }
6112
+ }
6113
+ /**
6114
+ * Check if any user message contains tool_result content blocks,
6115
+ * indicating a follow-up turn where we should skip web search.
6116
+ * In Anthropic format, tool results are content blocks inside user messages,
6117
+ * NOT separate role: "tool" messages like in OpenAI format.
6118
+ */
6119
+ function hasToolResultContent(messages) {
6120
+ return messages.some((msg) => Array.isArray(msg.content) && msg.content.some((block) => block.type === "tool_result"));
6121
+ }
6122
+ /**
6123
+ * Inject web search results into the Anthropic system field.
6124
+ * Handles three cases: absent, string, or array of content blocks.
6125
+ * When array, prepends without cache_control to preserve existing directives.
6126
+ */
6127
+ function injectSearchResults(body, searchContext) {
6128
+ if (body.system === void 0 || body.system === null) body.system = searchContext;
6129
+ else if (typeof body.system === "string") body.system = `${searchContext}\n\n${body.system}`;
6130
+ else if (Array.isArray(body.system)) body.system = [{
6131
+ type: "text",
6132
+ text: searchContext
6133
+ }, ...body.system];
6134
+ }
6135
+ /**
6136
+ * Strip web_search tools from the request and clean up tool_choice.
6137
+ * Returns the modified body object.
6138
+ */
6139
+ function stripWebSearchTool(body) {
6140
+ if (!body.tools) return;
6141
+ const tools = body.tools.filter((tool) => !isWebSearchTool$1(tool));
6142
+ body.tools = tools;
6143
+ if (tools.length === 0) {
6144
+ body.tools = void 0;
6145
+ body.tool_choice = void 0;
6146
+ } else if (body.tool_choice && typeof body.tool_choice === "object" && body.tool_choice.type === "tool") {
6147
+ const choiceName = body.tool_choice.name;
6148
+ if (choiceName && !tools.some((tool) => tool && typeof tool === "object" && tool.name === choiceName)) body.tool_choice = { type: "auto" };
6149
+ }
6150
+ }
6151
+ /**
6152
+ * Strip the injected `__anthropic_advisor` tool (and any Anthropic-native
6153
+ * `advisor_*` typed tool) from a request body. Used ONLY on the non-Claude
6154
+ * shim path: ADVISOR's server-side translate-loop (buildAdvisorStream) lives
6155
+ * on the native /v1/messages route, so a non-Claude model has no handler for
6156
+ * the tool and it must be removed before forwarding — otherwise the model
6157
+ * could emit a tool_use that nothing fulfils. Mirrors stripWebSearchTool's
6158
+ * tool_choice cleanup. Returns the original string (same reference) when
6159
+ * nothing was removed.
6160
+ */
6161
+ function stripAdvisorTool(rawBody) {
6162
+ let body;
6163
+ try {
6164
+ body = JSON.parse(rawBody);
6165
+ } catch {
6166
+ return rawBody;
6167
+ }
6168
+ if (!Array.isArray(body.tools)) return rawBody;
6169
+ const original = body.tools;
6170
+ const tools = original.filter((tool) => {
6171
+ if (typeof tool !== "object" || tool === null) return true;
6172
+ if (tool.name === ADVISOR_INTERNAL_TOOL_NAME) return false;
6173
+ const type = tool.type;
6174
+ return typeof type !== "string" || !type.startsWith("advisor_");
6175
+ });
6176
+ if (tools.length === original.length) return rawBody;
6177
+ if (tools.length === 0) {
6178
+ body.tools = void 0;
6179
+ body.tool_choice = void 0;
6180
+ } else {
6181
+ body.tools = tools;
6182
+ if (body.tool_choice && typeof body.tool_choice === "object" && body.tool_choice.type === "tool") {
6183
+ const choiceName = body.tool_choice.name;
6184
+ if (choiceName && !tools.some((tool) => tool && typeof tool === "object" && tool.name === choiceName)) body.tool_choice = { type: "auto" };
6185
+ }
6186
+ }
6187
+ return JSON.stringify(body);
6188
+ }
6189
+ /**
6190
+ * Process web search if the request contains a web_search tool.
6191
+ * Performs the search, injects results into system, and strips the tool.
6192
+ * Returns the (possibly modified) body string to forward.
6193
+ */
6194
+ async function processWebSearch(rawBody) {
6195
+ if (!rawBody.includes("web_search")) return rawBody;
6196
+ let body;
6197
+ try {
6198
+ body = JSON.parse(rawBody);
6199
+ } catch {
6200
+ return rawBody;
6201
+ }
6202
+ if (!body.tools?.some((tool) => isWebSearchTool$1(tool))) return rawBody;
6203
+ const messages = body.messages ?? [];
6204
+ const query = hasToolResultContent(messages) ? void 0 : extractUserQuery$1(messages);
6205
+ if (query) try {
6206
+ const results = await searchWeb(query);
6207
+ const searchContext = [
6208
+ "[Web Search Results]",
6209
+ results.content,
6210
+ "",
6211
+ results.references.map((r) => `- [${r.title}](${r.url})`).join("\n"),
6212
+ "[End Web Search Results]"
6213
+ ].join("\n");
6214
+ injectSearchResults(body, searchContext);
6215
+ } catch (error) {
6216
+ consola.warn("Web search failed, continuing without results:", error);
6217
+ }
6218
+ stripWebSearchTool(body);
6219
+ return JSON.stringify(body);
6220
+ }
6221
+ async function handleCompletion(c) {
6222
+ const startTime = Date.now();
6223
+ await checkRateLimit(state);
6224
+ const rawBody = await c.req.text();
6225
+ const debugEnabled = consola.level >= 4;
6226
+ if (debugEnabled) consola.debug("Anthropic request body:", rawBody.slice(0, 2e3));
6227
+ if (process.env.GH_ROUTER_LOG_FIELDS === "1") {
6228
+ let parsedForLog = void 0;
6229
+ try {
6230
+ parsedForLog = JSON.parse(rawBody);
6231
+ } catch {}
6232
+ logRequestFields({
6233
+ path: c.req.path,
6234
+ body: parsedForLog,
6235
+ betaHeader: c.req.header("anthropic-beta")
6236
+ });
6237
+ }
6238
+ if (state.manualApprove) await awaitApproval();
6239
+ const betaHeaders = extractBetaHeaders(c);
6240
+ const advisorEnabled = isAdvisorRequested(c.req.header("anthropic-beta"));
6241
+ let finalBody = await processWebSearch(rawBody);
6242
+ finalBody = sanitizeAnthropicBody(finalBody);
6243
+ if (advisorEnabled) {
6244
+ finalBody = injectAdvisorTool(finalBody);
6245
+ consola.info("ADVISOR enabled for this request — injecting __anthropic_advisor tool; will translate tool_use → server_tool_use{advisor} on the SSE stream");
6246
+ }
6247
+ if (finalBody.includes("\"mcp_servers\"")) try {
6248
+ const probe = JSON.parse(finalBody);
6249
+ if (Array.isArray(probe.mcp_servers) && probe.mcp_servers.length > 0) return c.json({
6250
+ type: "error",
6251
+ error: {
6252
+ type: "invalid_request_error",
6253
+ message: "Inline `mcp_servers` body field is not supported by github-router. Configure remote MCP servers as local stdio entries in `~/.claude/mcp.json` instead — Claude Code will spawn them locally and the proxy passes their tool calls through transparently. (https://docs.claude.com/en/docs/claude-code/mcp)"
6254
+ }
6255
+ }, 400);
6256
+ } catch {}
4820
6257
  const { body: resolvedBody, originalModel, resolvedModel, selectedModel } = resolveModelInBody$1(finalBody);
4821
6258
  const modelId = resolvedModel ?? originalModel;
6259
+ const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel);
6260
+ if (messagesRoute !== "claude-passthrough") {
6261
+ const shimBody = stripAdvisorTool(resolvedBody);
6262
+ if (advisorEnabled) consola.info("ADVISOR requested with a non-Claude model — stripping the injected __anthropic_advisor tool and proceeding without advisor (Claude-only feature; gracefully degraded).");
6263
+ const shimOpts = {
6264
+ rawBody: shimBody,
6265
+ modelId,
6266
+ model: selectedModel,
6267
+ originalModel,
6268
+ startTime
6269
+ };
6270
+ return messagesRoute === "chat-shim" ? handleNonClaudeChat(c, shimOpts) : handleNonClaudeResponses(c, shimOpts);
6271
+ }
4822
6272
  if (modelId) logEndpointMismatch(modelId, "/v1/messages");
4823
6273
  const effectiveBetas = applyDefaultBetas(betaHeaders, resolvedModel ?? originalModel);
4824
6274
  const advisorAborter = advisorEnabled ? new AbortController() : void 0;
@@ -4900,13 +6350,14 @@ async function handleCompletion(c) {
4900
6350
  const cappedResult = await readResponseBodyCapped(response, c.req.path, MAX_RESPONSE_BODY_BYTES);
4901
6351
  if (!cappedResult.ok) return c.json(cappedResult.errorResponse, cappedResult.status);
4902
6352
  const responseBody = cappedResult.value;
6353
+ const usage = responseBody.usage;
4903
6354
  logRequest({
4904
6355
  method: "POST",
4905
6356
  path: c.req.path,
4906
6357
  model: originalModel,
4907
6358
  resolvedModel,
4908
- inputTokens: responseBody.usage?.input_tokens,
4909
- outputTokens: responseBody.usage?.output_tokens,
6359
+ inputTokens: usage?.input_tokens,
6360
+ outputTokens: usage?.output_tokens,
4910
6361
  status: response.status
4911
6362
  }, selectedModel, startTime);
4912
6363
  if (debugEnabled) consola.debug("Non-streaming response from Copilot /v1/messages:", JSON.stringify(responseBody).slice(0, 2e3));
@@ -4953,48 +6404,6 @@ function resolveModelInBody$1(rawBody) {
4953
6404
  selectedModel
4954
6405
  };
4955
6406
  }
4956
- const EFFORT_ORDER = [
4957
- "low",
4958
- "medium",
4959
- "high",
4960
- "xhigh"
4961
- ];
4962
- /**
4963
- * Bucket a thinking budget into a Copilot reasoning-effort string.
4964
- * `<2000`→low, `<8000`→medium, `<24000`→high, else→xhigh.
4965
- * Defaults missing/non-numeric budgets to 8000 ("high").
4966
- */
4967
- function bucketEffort(budget) {
4968
- const n = typeof budget === "number" && Number.isFinite(budget) ? budget : 8e3;
4969
- if (n < 2e3) return "low";
4970
- if (n < 8e3) return "medium";
4971
- if (n < 24e3) return "high";
4972
- return "xhigh";
4973
- }
4974
- /**
4975
- * Clamp a bucketed effort to the closest value in `supported`. Ties
4976
- * resolve to the lower-tier option (per EFFORT_ORDER).
4977
- *
4978
- * Iterates EFFORT_ORDER (canonical low→xhigh) so the first match on a
4979
- * given distance is always the lower-tier value, regardless of input
4980
- * order in `supported`.
4981
- */
4982
- function clampEffort(bucketed, supported) {
4983
- if (supported.includes(bucketed)) return bucketed;
4984
- const targetIdx = EFFORT_ORDER.indexOf(bucketed);
4985
- let best;
4986
- let bestDist = Infinity;
4987
- for (let i = 0; i < EFFORT_ORDER.length; i++) {
4988
- const value = EFFORT_ORDER[i];
4989
- if (!supported.includes(value)) continue;
4990
- const dist = Math.abs(i - targetIdx);
4991
- if (dist < bestDist) {
4992
- bestDist = dist;
4993
- best = value;
4994
- }
4995
- }
4996
- return best ?? bucketed;
4997
- }
4998
6407
  /**
4999
6408
  * Clamp `body.output_config.effort` to the model's
5000
6409
  * `capabilities.supports.reasoning_effort` allowlist. Mutates `body`
@@ -5054,8 +6463,9 @@ function translateThinking(body, model) {
5054
6463
  if (!model?.capabilities?.supports?.adaptive_thinking) return false;
5055
6464
  const thinking = body.thinking;
5056
6465
  if (!thinking || typeof thinking !== "object") return false;
5057
- if (thinking.type !== "enabled") return false;
5058
- const bucketed = bucketEffort(thinking.budget_tokens);
6466
+ const t = thinking;
6467
+ if (t.type !== "enabled") return false;
6468
+ const bucketed = bucketEffort(t.budget_tokens);
5059
6469
  const supported = model.capabilities.supports.reasoning_effort;
5060
6470
  const effort = Array.isArray(supported) && supported.length > 0 ? clampEffort(bucketed, supported) : bucketed;
5061
6471
  body.thinking = { type: "adaptive" };
@@ -5077,9 +6487,10 @@ function translateThinking(body, model) {
5077
6487
  function sanitizeCacheControl$1(body) {
5078
6488
  let stripped = false;
5079
6489
  function stripScope(block) {
5080
- if (block.cache_control?.scope !== void 0) {
5081
- delete block.cache_control.scope;
5082
- if (Object.keys(block.cache_control).length === 0) delete block.cache_control;
6490
+ const cc = block.cache_control;
6491
+ if (cc?.scope !== void 0) {
6492
+ delete cc.scope;
6493
+ if (Object.keys(cc).length === 0) delete block.cache_control;
5083
6494
  stripped = true;
5084
6495
  }
5085
6496
  }
@@ -6006,6 +7417,134 @@ function parseSharedArgs(args) {
6006
7417
  };
6007
7418
  }
6008
7419
  /**
7420
+ * Non-Claude models we surface as first-class, selectable rows in Claude
7421
+ * Code's model picker (Phase 3 of native-non-claude-models). The main
7422
+ * agent loop runs on them through the `/v1/messages` translation shim
7423
+ * (`src/lib/anthropic-translate/*`, branched in `routes/messages/handler.ts`)
7424
+ * that forwards non-Claude targets to Copilot `/responses` (gpt) or
7425
+ * `/chat/completions` (gemini). The exact gemini id is
7426
+ * `gemini-3.1-pro-preview` (preview slug — NOT `gemini-3.1-pro`).
7427
+ *
7428
+ * Display labels only: the gateway-model cache schema Claude Code reads is
7429
+ * `{id, display_name?}` per model — there is NO per-model context-window
7430
+ * field, so context accounting for a selected row uses Claude Code's
7431
+ * default window (safe under-accounting: it compacts earlier than the real
7432
+ * 1M/400k window, never overflows). See `seedGatewayModelCache`.
7433
+ */
7434
+ const NATIVE_NON_CLAUDE_MODELS = [
7435
+ {
7436
+ id: "gpt-5.5",
7437
+ displayName: "GPT-5.5"
7438
+ },
7439
+ {
7440
+ id: "gpt-5.3-codex",
7441
+ displayName: "GPT-5.3 Codex"
7442
+ },
7443
+ {
7444
+ id: "gemini-3.5-flash",
7445
+ displayName: "Gemini 3.5 Flash"
7446
+ },
7447
+ {
7448
+ id: "gemini-3.1-pro-preview",
7449
+ displayName: "Gemini 3.1 Pro (preview)"
7450
+ }
7451
+ ];
7452
+ /**
7453
+ * The subset of `NATIVE_NON_CLAUDE_MODELS` actually present in the live
7454
+ * Copilot catalog. License tiers differ (gpt-5.5 needs
7455
+ * pro_plus/business/enterprise/max; gemini-3.5-flash is absent on
7456
+ * edu/individual_trial), so a model missing from the catalog is silently
7457
+ * dropped — the caller then neither enables discovery nor writes a cache
7458
+ * for it, and lesser tiers see the unchanged picker. Pure (reads
7459
+ * `state.models`), so it is unit-testable without side effects.
7460
+ */
7461
+ function nativeSelectableModelsInCatalog() {
7462
+ const catalog = state.models?.data;
7463
+ if (!catalog || catalog.length === 0) return [];
7464
+ const present = new Set(catalog.map((m) => m.id));
7465
+ return NATIVE_NON_CLAUDE_MODELS.filter((m) => present.has(m.id)).map((m) => ({
7466
+ id: m.id,
7467
+ display_name: m.displayName
7468
+ }));
7469
+ }
7470
+ /**
7471
+ * Pre-seed Claude Code's gateway-model discovery cache so the non-Claude
7472
+ * models appear as selectable picker rows WITHOUT the network fetch.
7473
+ *
7474
+ * Verified against the installed Claude Code build (2.1.201): the picker
7475
+ * builder reads `<CLAUDE_CONFIG_DIR>/cache/gateway-models.json`
7476
+ * (schema `{baseUrl: string, fetchedAt: number, models: [{id, display_name?}]}`)
7477
+ * and — when gateway discovery is enabled (first-party auth mode +
7478
+ * `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` set + a non-`api.anthropic.com`
7479
+ * `ANTHROPIC_BASE_URL`, all true for the proxy) — maps each cached model to
7480
+ * a picker row `{value: id, label: display_name}`. Critically, the
7481
+ * cache-READ path applies NO id filter; the `/^(claude|anthropic)/i` filter
7482
+ * lives ONLY in the network-FETCH path. So a pre-seeded cache can carry the
7483
+ * real Copilot ids (`gpt-5.5`, `gemini-3.1-pro-preview`, …) — no `claude-*`
7484
+ * alias needed — and selecting a row sends that real id, which
7485
+ * `resolveModel()` exact-matches and the `/v1/messages` shim routes.
7486
+ *
7487
+ * The network fetch never overwrites this seed: it bails when nonessential
7488
+ * traffic is disabled, and the proxy ALWAYS sets
7489
+ * `CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1`, so the fetch returns before
7490
+ * it can write. The seed is therefore authoritative for the session.
7491
+ *
7492
+ * `baseUrl` MUST equal the `ANTHROPIC_BASE_URL` Claude Code sees
7493
+ * (`serverUrl`) or the cache is discarded. `configDir` defaults to
7494
+ * `PATHS.CLAUDE_CONFIG_DIR` — the same dir the proxy points
7495
+ * `CLAUDE_CONFIG_DIR` at — so the write target and Claude Code's read
7496
+ * target are identical by construction.
7497
+ *
7498
+ * Best-effort: every failure is swallowed — a missing picker row must never
7499
+ * break launch. This is coupled to Claude Code's internal cache path/schema;
7500
+ * if a future build changes them the read simply ignores the seed and the
7501
+ * rows don't appear (graceful degradation). Returns whether a file was
7502
+ * written (for tests/observability).
7503
+ *
7504
+ * The write is atomic (temp file in the same dir + rename) so a Claude Code
7505
+ * read can never observe a torn/partial JSON (which its safeParse would
7506
+ * reject, dropping the rows). Rename-over-existing is atomic on POSIX and
7507
+ * Windows (libuv MoveFileEx REPLACE_EXISTING).
7508
+ */
7509
+ function seedGatewayModelCache(serverUrl, models$1, configDir = PATHS.CLAUDE_CONFIG_DIR) {
7510
+ if (models$1.length === 0) return false;
7511
+ const cacheDir = nodePath$1.join(configDir, "cache");
7512
+ const target = nodePath$1.join(cacheDir, "gateway-models.json");
7513
+ const tmp = nodePath$1.join(cacheDir, `gateway-models.json.${process.pid}.${Math.random().toString(36).slice(2)}.tmp`);
7514
+ try {
7515
+ fs$2.mkdirSync(cacheDir, { recursive: true });
7516
+ const payload = {
7517
+ baseUrl: serverUrl,
7518
+ fetchedAt: Date.now(),
7519
+ models: models$1.map((m) => ({
7520
+ id: m.id,
7521
+ display_name: m.display_name
7522
+ }))
7523
+ };
7524
+ fs$2.writeFileSync(tmp, JSON.stringify(payload), "utf-8");
7525
+ fs$2.renameSync(tmp, target);
7526
+ return true;
7527
+ } catch {
7528
+ try {
7529
+ fs$2.rmSync(tmp, { force: true });
7530
+ } catch {}
7531
+ return false;
7532
+ }
7533
+ }
7534
+ /**
7535
+ * Remove any seeded gateway-model cache. Called when the current catalog
7536
+ * carries none of the target models, so a user who has pinned
7537
+ * `CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY` on cannot surface rows for
7538
+ * models that are no longer available. Best-effort (per-launch config dirs
7539
+ * make a stale file rare, but this closes the pinned-port + catalog-change
7540
+ * seam). Never throws.
7541
+ */
7542
+ function clearGatewayModelCache(configDir = PATHS.CLAUDE_CONFIG_DIR) {
7543
+ try {
7544
+ fs$2.rmSync(nodePath$1.join(configDir, "cache", "gateway-models.json"), { force: true });
7545
+ } catch {}
7546
+ }
7547
+ /**
6009
7548
  * Build environment variables for Claude Code.
6010
7549
  *
6011
7550
  * The parent env is sanitized of every key in `STRIPPED_PARENT_ENV_KEYS`
@@ -6049,16 +7588,17 @@ function getClaudeCodeEnvVars(serverUrl, model) {
6049
7588
  const vars = {
6050
7589
  ANTHROPIC_BASE_URL: serverUrl,
6051
7590
  CLAUDE_CONFIG_DIR: PATHS.CLAUDE_CONFIG_DIR,
6052
- MCP_TIMEOUT: "2100000",
6053
- MCP_TOOL_TIMEOUT: "2100000",
6054
7591
  DISABLE_NON_ESSENTIAL_MODEL_CALLS: "1",
6055
7592
  CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC: "1",
6056
7593
  DISABLE_TELEMETRY: "1"
6057
7594
  };
6058
7595
  if (model) vars.ANTHROPIC_MODEL = model;
6059
- if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = "claude-sonnet-4-6";
6060
- if (process.env.ANTHROPIC_DEFAULT_SONNET_MODEL === void 0) vars.ANTHROPIC_DEFAULT_SONNET_MODEL = "claude-sonnet-4-6";
6061
- if (process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL === void 0) vars.ANTHROPIC_DEFAULT_HAIKU_MODEL = "claude-haiku-4-5";
7596
+ const mcpToolTimeoutMs = String(resolveMcpToolTimeoutMs());
7597
+ if (process.env.MCP_TIMEOUT === void 0) vars.MCP_TIMEOUT = mcpToolTimeoutMs;
7598
+ if (process.env.MCP_TOOL_TIMEOUT === void 0) vars.MCP_TOOL_TIMEOUT = mcpToolTimeoutMs;
7599
+ if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = "claude-sonnet-5";
7600
+ if (process.env.ANTHROPIC_DEFAULT_SONNET_MODEL === void 0) vars.ANTHROPIC_DEFAULT_SONNET_MODEL = "claude-sonnet-5";
7601
+ if (process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL === void 0) vars.ANTHROPIC_DEFAULT_HAIKU_MODEL = "claude-sonnet-5";
6062
7602
  if (process.env.ANTHROPIC_DEFAULT_OPUS_MODEL === void 0) vars.ANTHROPIC_DEFAULT_OPUS_MODEL = "claude-opus-4-8";
6063
7603
  if (process.env.CLAUDE_CODE_PLAN_V2_AGENT_COUNT === void 0) vars.CLAUDE_CODE_PLAN_V2_AGENT_COUNT = "7";
6064
7604
  for (const key of [
@@ -6068,6 +7608,10 @@ function getClaudeCodeEnvVars(serverUrl, model) {
6068
7608
  "CLAUDE_CODE_ENABLE_FINE_GRAINED_TOOL_STREAMING",
6069
7609
  "CLAUDE_CODE_ENABLE_TASKS"
6070
7610
  ]) if (process.env[key] === void 0) vars[key] = "1";
7611
+ const nativeModels = nativeSelectableModelsInCatalog();
7612
+ if (nativeModels.length > 0) {
7613
+ if (seedGatewayModelCache(serverUrl, nativeModels) && process.env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0 && vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0) vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY = "1";
7614
+ } else clearGatewayModelCache();
6071
7615
  if (toolbeltEnabled()) Object.assign(vars, toolbeltPathOverride(process.env, PATHS.TOOLBELT_BIN_DIR));
6072
7616
  return vars;
6073
7617
  }
@@ -6095,60 +7639,116 @@ function getCodexEnvVars(serverUrl) {
6095
7639
 
6096
7640
  //#endregion
6097
7641
  //#region src/claude.ts
7642
+ const claudeArgs = {
7643
+ ...sharedServerArgs,
7644
+ model: {
7645
+ alias: "m",
7646
+ type: "string",
7647
+ description: "Override the default model for Claude Code. Accepts a full slug (e.g. claude-opus-4-7) or an Opus family shorthand (e.g. 4.7, 4.8, 4.6) which expands to the best variant for that family — adding the [1m] suffix when a 1M-context backend is in the catalog."
7648
+ },
7649
+ "codex-mcp": {
7650
+ type: "boolean",
7651
+ default: true,
7652
+ description: "Wire peer-model MCP personas (codex-critic, codex-reviewer, gemini-critic) into the spawned Claude Code session"
7653
+ },
7654
+ "codex-cli": {
7655
+ type: "boolean",
7656
+ default: false,
7657
+ description: "Add a `codex mcp-server` stdio backend so codex-implementer can mutate files. Requires codex CLI 0.129+; gracefully falls back to HTTP-only if absent."
7658
+ },
7659
+ "codex-mcp-only": {
7660
+ type: "boolean",
7661
+ default: false,
7662
+ description: "Pass --strict-mcp-config to claude code so only github-router's MCP servers are loaded (hides user's existing MCP servers)"
7663
+ },
7664
+ stealth: {
7665
+ type: "boolean",
7666
+ default: false,
7667
+ description: "Opt back into VS Code-only beta header filtering. Loses leverage features (task budgets, token-efficient tools, prompt caching, etc.) but minimizes the wire-fingerprint difference from VS Code Copilot Chat. By default the `claude` subcommand enables extended/leverage betas because the spawned Claude Code already identifies itself via UA and other headers — partial stealth doesn't buy much."
7668
+ },
7669
+ "trust-gate": {
7670
+ type: "boolean",
7671
+ default: false,
7672
+ description: "Explicitly record consent for the structural Stop-gate in THIS repo (pinned to the repo's root-commit). The gate is ON BY DEFAULT when a harness is detected (consent-by-launching), so this is now mostly redundant; it stays for explicit/scripted use. Disable the gate entirely with GH_ROUTER_DISABLE_STOP_GATE=1."
7673
+ },
7674
+ "no-stop-gate": {
7675
+ type: "boolean",
7676
+ default: false,
7677
+ description: "Disable the structural Stop-gate for THIS session (same effect as GH_ROUTER_DISABLE_STOP_GATE=1). Intended for driven/automated sessions where a blocking Stop hook would hang the turn-end while a fleet driver waits."
7678
+ },
7679
+ "auto-update": {
7680
+ type: "boolean",
7681
+ default: true,
7682
+ description: "Check for and install the latest Claude Code on launch via `claude update` (throttled to once per hour via ~/.local/share/github-router/last-update-check). `claude update` respects the real install method (native installer or npm), so it never creates a conflicting second install; builds too old to support it fall back to `npm install -g @anthropic-ai/claude-code@latest`. Set to false (--no-auto-update) to check and warn only. Falls back gracefully if claude/npm/network unavailable."
7683
+ },
7684
+ "update-check": {
7685
+ type: "boolean",
7686
+ default: true,
7687
+ description: "Check the npm registry for a newer Claude Code version on launch and warn if stale (non-blocking ~500ms cost). Set to false (--no-update-check) to skip the check entirely (useful for offline/CI). Independent from --auto-update: --no-update-check implies no auto-install (nothing to install since we never check)."
7688
+ }
7689
+ };
7690
+ /**
7691
+ * Build the argv to forward to the spawned `claude` child from citty's
7692
+ * rawArgs (every token after the `claude` subcommand). citty is non-strict,
7693
+ * so an unknown flag such as `--print`/`-p`… `--output-format` is absorbed
7694
+ * into the parsed `args` object AND its value swallowed, instead of landing
7695
+ * in `args._`; forwarding only `args._` therefore drops headless flags unless
7696
+ * the user wrapped them in `--`. Here we walk rawArgs and forward every token
7697
+ * that is NOT one of github-router's OWN declared flags (or that flag's
7698
+ * consumed value). Everything after a literal `--` is forwarded verbatim,
7699
+ * preserving the prior explicit-passthrough behavior.
7700
+ *
7701
+ * A child flag whose NAME collides with a github-router flag (`-p`/`--port`,
7702
+ * `-v`/`--verbose`, `-m`/`--model`, `-a`/`--account-type`, `-r`/`--rate-limit`,
7703
+ * `-g`/`--github-token`) is owned by github-router; forward it to the child
7704
+ * explicitly after `--` (e.g. `github-router claude -- -p`). Every other Claude
7705
+ * flag (`--print`, `--output-format`, `--resume`, `--continue`, …) flows
7706
+ * through automatically.
7707
+ */
7708
+ function collectChildPassthroughArgs(rawArgs, argsDef) {
7709
+ const known = /* @__PURE__ */ new Set();
7710
+ const stringTyped = /* @__PURE__ */ new Set();
7711
+ for (const [name$1, def] of Object.entries(argsDef)) {
7712
+ const rawAlias = "alias" in def ? def.alias : void 0;
7713
+ const aliases = rawAlias === void 0 ? [] : Array.isArray(rawAlias) ? rawAlias : [rawAlias];
7714
+ for (const n of [name$1, ...aliases]) {
7715
+ known.add(n);
7716
+ if (def.type === "string") stringTyped.add(n);
7717
+ }
7718
+ }
7719
+ const forwarded = [];
7720
+ for (let i = 0; i < rawArgs.length; i++) {
7721
+ const tok = rawArgs[i];
7722
+ if (tok === "--") {
7723
+ forwarded.push(...rawArgs.slice(i + 1));
7724
+ break;
7725
+ }
7726
+ if (tok === "-" || !tok.startsWith("-")) {
7727
+ forwarded.push(tok);
7728
+ continue;
7729
+ }
7730
+ const doubleDash = tok.startsWith("--");
7731
+ const afterDashes = tok.slice(doubleDash ? 2 : 1);
7732
+ const eq = afterDashes.indexOf("=");
7733
+ const rawName = eq >= 0 ? afterDashes.slice(0, eq) : afterDashes;
7734
+ const hasInlineValue = eq >= 0;
7735
+ const negated = doubleDash && rawName.startsWith("no-");
7736
+ const baseName = negated ? rawName.slice(3) : rawName;
7737
+ if (!(known.has(rawName) || negated && known.has(baseName))) {
7738
+ forwarded.push(tok);
7739
+ continue;
7740
+ }
7741
+ if (!negated && !hasInlineValue && stringTyped.has(rawName) && i + 1 < rawArgs.length && rawArgs[i + 1] !== "--" && !rawArgs[i + 1].startsWith("-")) i++;
7742
+ }
7743
+ return forwarded;
7744
+ }
6098
7745
  const claude = defineCommand({
6099
7746
  meta: {
6100
7747
  name: "claude",
6101
7748
  description: "Start the proxy server and launch Claude Code"
6102
7749
  },
6103
- args: {
6104
- ...sharedServerArgs,
6105
- model: {
6106
- alias: "m",
6107
- type: "string",
6108
- description: "Override the default model for Claude Code. Accepts a full slug (e.g. claude-opus-4-7) or an Opus family shorthand (e.g. 4.7, 4.8, 4.6) which expands to the best variant for that family — adding the [1m] suffix when a 1M-context backend is in the catalog."
6109
- },
6110
- "codex-mcp": {
6111
- type: "boolean",
6112
- default: true,
6113
- description: "Wire peer-model MCP personas (codex-critic, codex-reviewer, gemini-critic) into the spawned Claude Code session"
6114
- },
6115
- "codex-cli": {
6116
- type: "boolean",
6117
- default: false,
6118
- description: "Add a `codex mcp-server` stdio backend so codex-implementer can mutate files. Requires codex CLI 0.129+; gracefully falls back to HTTP-only if absent."
6119
- },
6120
- "codex-mcp-only": {
6121
- type: "boolean",
6122
- default: false,
6123
- description: "Pass --strict-mcp-config to claude code so only github-router's MCP servers are loaded (hides user's existing MCP servers)"
6124
- },
6125
- stealth: {
6126
- type: "boolean",
6127
- default: false,
6128
- description: "Opt back into VS Code-only beta header filtering. Loses leverage features (task budgets, token-efficient tools, prompt caching, etc.) but minimizes the wire-fingerprint difference from VS Code Copilot Chat. By default the `claude` subcommand enables extended/leverage betas because the spawned Claude Code already identifies itself via UA and other headers — partial stealth doesn't buy much."
6129
- },
6130
- "trust-gate": {
6131
- type: "boolean",
6132
- default: false,
6133
- description: "Explicitly record consent for the structural Stop-gate in THIS repo (pinned to the repo's root-commit). The gate is ON BY DEFAULT when a harness is detected (consent-by-launching), so this is now mostly redundant; it stays for explicit/scripted use. Disable the gate entirely with GH_ROUTER_DISABLE_STOP_GATE=1."
6134
- },
6135
- "no-stop-gate": {
6136
- type: "boolean",
6137
- default: false,
6138
- description: "Disable the structural Stop-gate for THIS session (same effect as GH_ROUTER_DISABLE_STOP_GATE=1). Intended for driven/automated sessions where a blocking Stop hook would hang the turn-end while a fleet driver waits."
6139
- },
6140
- "auto-update": {
6141
- type: "boolean",
6142
- default: true,
6143
- description: "Check for and install the latest Claude Code on launch via `claude update` (throttled to once per hour via ~/.local/share/github-router/last-update-check). `claude update` respects the real install method (native installer or npm), so it never creates a conflicting second install; builds too old to support it fall back to `npm install -g @anthropic-ai/claude-code@latest`. Set to false (--no-auto-update) to check and warn only. Falls back gracefully if claude/npm/network unavailable."
6144
- },
6145
- "update-check": {
6146
- type: "boolean",
6147
- default: true,
6148
- description: "Check the npm registry for a newer Claude Code version on launch and warn if stale (non-blocking ~500ms cost). Set to false (--no-update-check) to skip the check entirely (useful for offline/CI). Independent from --auto-update: --no-update-check implies no auto-install (nothing to install since we never check)."
6149
- }
6150
- },
6151
- async run({ args }) {
7750
+ args: claudeArgs,
7751
+ async run({ args, rawArgs }) {
6152
7752
  if (!process$1.stdout.isTTY) {
6153
7753
  consola.error("The claude subcommand requires a TTY (interactive terminal).");
6154
7754
  process$1.exit(1);
@@ -6217,7 +7817,7 @@ const claude = defineCommand({
6217
7817
  const banner = chosenSlug === resolvedSlug ? chosenSlug : `${chosenSlug} → ${resolvedSlug}`;
6218
7818
  process$1.stderr.write(`Server ready on ${serverUrl}, launching Claude Code (${banner})...\n`);
6219
7819
  const envVars = getClaudeCodeEnvVars(serverUrl, chosenSlug);
6220
- const extraArgs = args._ ?? [];
7820
+ const extraArgs = collectChildPassthroughArgs(rawArgs, claudeArgs);
6221
7821
  if (toolbeltEnabled()) {
6222
7822
  provisionToolbelt().catch((err) => consola.debug("Toolbelt provisioning failed:", err));
6223
7823
  const toolbeltLine = buildToolbeltAwareness(availableToolCommands());
@@ -6260,7 +7860,8 @@ const claude = defineCommand({
6260
7860
  geminiAvailable,
6261
7861
  groupKeys,
6262
7862
  workerToolsAvailable: workerToolsEnabled(),
6263
- browseAvailable: browseAgentEnabled()
7863
+ browseAvailable: browseAgentEnabled(),
7864
+ implementerModel: implementerSubagentModel()
6264
7865
  });
6265
7866
  state.peerMcpNonce = runtime.nonce;
6266
7867
  envVars.GH_ROUTER_HOOK_MCP_URL = serverUrl;
@@ -7411,9 +9012,10 @@ async function readPayload() {
7411
9012
  * live tree itself for anything beyond it, so a giant diff never blows the model
7412
9013
  * window. The Stop hook already caps the captured diff at 2 MiB. */
7413
9014
  const MAX_EMBEDDED_DIFF_BYTES = 200 * 1024;
7414
- /** Wall-clock the reviewer may take. Sized at the worker engine's own 30-min cap
7415
- * plus headroom this process is detached, so nothing waits on it; the bound
7416
- * only stops a hung request from lingering forever. */
9015
+ /** Wall-clock the stop-gate reviewer may take. This is INDEPENDENT of the
9016
+ * autonomous worker's wall-clock cap (`DEFAULT_MAX_WALLCLOCK_MS`, now 6h)
9017
+ * it bounds this one detached review request. Nothing waits on this process,
9018
+ * so the bound only stops a hung request from lingering forever. */
7417
9019
  const REVIEW_TIMEOUT_MS = 2100 * 1e3;
7418
9020
  function buildReviewBrief(payload) {
7419
9021
  const diff = payload.diff.length > MAX_EMBEDDED_DIFF_BYTES ? `${payload.diff.slice(0, MAX_EMBEDDED_DIFF_BYTES)}\n\n[diff truncated at ${MAX_EMBEDDED_DIFF_BYTES} bytes — read the files directly for the rest]` : payload.diff;