github-router 0.3.282 → 0.3.288
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -5
- package/dist/{attribution-settings-srL3gBR9.js → attribution-settings-CofaqSzw.js} +54 -15
- package/dist/attribution-settings-CofaqSzw.js.map +1 -0
- package/dist/{auth-VUL2Zxvw.js → auth-DG4vh8-F.js} +3 -3
- package/dist/{auth-VUL2Zxvw.js.map → auth-DG4vh8-F.js.map} +1 -1
- package/dist/browser-ext/manifest.json +1 -1
- package/dist/{check-usage-CcLdFPGr.js → check-usage-BqN7mBYv.js} +4 -4
- package/dist/{check-usage-CcLdFPGr.js.map → check-usage-BqN7mBYv.js.map} +1 -1
- package/dist/{claude-DJQjA5oQ.js → claude-CoZKRNB8.js} +25 -20
- package/dist/claude-CoZKRNB8.js.map +1 -0
- package/dist/{codex-BAXemWpJ.js → codex-y2OyLIdv.js} +5 -5
- package/dist/{codex-BAXemWpJ.js.map → codex-y2OyLIdv.js.map} +1 -1
- package/dist/{debug-CtzYWxpJ.js → debug-B5TjPTTH.js} +2 -2
- package/dist/{debug-CtzYWxpJ.js.map → debug-B5TjPTTH.js.map} +1 -1
- package/dist/engine-B5nVGH4b.js +2 -0
- package/dist/{gate-discovery-BwtiYKvW.js → gate-discovery-LQZ-enJa.js} +5 -5
- package/dist/{gate-discovery-BwtiYKvW.js.map → gate-discovery-LQZ-enJa.js.map} +1 -1
- package/dist/{get-copilot-usage-B-EDAqQb.js → get-copilot-usage-BjA0nyGR.js} +2 -2
- package/dist/{get-copilot-usage-B-EDAqQb.js.map → get-copilot-usage-BjA0nyGR.js.map} +1 -1
- package/dist/{internal-artifact-open-Dj5Nlq0L.js → internal-artifact-open-BskmUpnb.js} +2 -2
- package/dist/{internal-artifact-open-Dj5Nlq0L.js.map → internal-artifact-open-BskmUpnb.js.map} +1 -1
- package/dist/{internal-first-mate-guard-DMgHg2s6.js → internal-first-mate-guard-5XEHMaqy.js} +3 -3
- package/dist/{internal-first-mate-guard-DMgHg2s6.js.map → internal-first-mate-guard-5XEHMaqy.js.map} +1 -1
- package/dist/{internal-first-mate-guard-CQdXWjNz.js → internal-first-mate-guard-DHDQ6hFz.js} +1 -1
- package/dist/{internal-plan-review-DRFHET_B.js → internal-plan-review-BUuMx4ku.js} +3 -3
- package/dist/{internal-plan-review-DRFHET_B.js.map → internal-plan-review-BUuMx4ku.js.map} +1 -1
- package/dist/{internal-prompt-submit-f1LR2P2G.js → internal-prompt-submit-CQQ15xdO.js} +4 -4
- package/dist/{internal-prompt-submit-f1LR2P2G.js.map → internal-prompt-submit-CQQ15xdO.js.map} +1 -1
- package/dist/{internal-session-bind-DhZhxJ3T.js → internal-session-bind-D04W2yWI.js} +2 -2
- package/dist/{internal-session-bind-DhZhxJ3T.js.map → internal-session-bind-D04W2yWI.js.map} +1 -1
- package/dist/{internal-stop-hook-514RJeHO.js → internal-stop-hook-DSbaDb_m.js} +5 -5
- package/dist/{internal-stop-hook-514RJeHO.js.map → internal-stop-hook-DSbaDb_m.js.map} +1 -1
- package/dist/{internal-stop-review-BBsLcbPG.js → internal-stop-review-CdByyJLc.js} +2 -2
- package/dist/{internal-stop-review-BBsLcbPG.js.map → internal-stop-review-CdByyJLc.js.map} +1 -1
- package/dist/{internal-worker-guard-VucpClSi.js → internal-worker-guard-BIPN6Rv9.js} +2 -2
- package/dist/{internal-worker-guard-VucpClSi.js.map → internal-worker-guard-BIPN6Rv9.js.map} +1 -1
- package/dist/{internal-workspace-header-OYgHEnFt.js → internal-workspace-header-BKqejstG.js} +2 -2
- package/dist/{internal-workspace-header-OYgHEnFt.js.map → internal-workspace-header-BKqejstG.js.map} +1 -1
- package/dist/lifecycle-BTodQvn4.js +2 -0
- package/dist/lifecycle-C7JYNz-F.js +2 -0
- package/dist/{lifecycle-B7CHqKlF.js → lifecycle-DbM29FLK.js} +2 -2
- package/dist/{lifecycle-B7CHqKlF.js.map → lifecycle-DbM29FLK.js.map} +1 -1
- package/dist/{lifecycle-D-rL81tT.js → lifecycle-SXaWssN9.js} +2 -2
- package/dist/{lifecycle-D-rL81tT.js.map → lifecycle-SXaWssN9.js.map} +1 -1
- package/dist/main.js +17 -17
- package/dist/{mcp-workspace-header-ucs2SDST.js → mcp-workspace-header-DRCCWlOi.js} +2 -2
- package/dist/{mcp-workspace-header-ucs2SDST.js.map → mcp-workspace-header-DRCCWlOi.js.map} +1 -1
- package/dist/{models-C16mBK2M.js → models-Dz8d_SnI.js} +3 -3
- package/dist/{models-C16mBK2M.js.map → models-Dz8d_SnI.js.map} +1 -1
- package/dist/{orchestration-CtM6FYNx.js → orchestration-BrJwZxMN.js} +2 -2
- package/dist/{orchestration-CtM6FYNx.js.map → orchestration-BrJwZxMN.js.map} +1 -1
- package/dist/paths-CTr59UC6.js +2 -0
- package/dist/{paths-wLC0InjX.js → paths-D7_SAaIQ.js} +4 -4
- package/dist/{paths-wLC0InjX.js.map → paths-D7_SAaIQ.js.map} +1 -1
- package/dist/{peer-mcp-personas-l4P4s1fK.js → peer-mcp-personas-B5Wp6wIn.js} +837 -147
- package/dist/peer-mcp-personas-B5Wp6wIn.js.map +1 -0
- package/dist/{plan-review-hook-Lf9ISdF6.js → plan-review-hook-CVZsG9MZ.js} +3 -3
- package/dist/{plan-review-hook-Lf9ISdF6.js.map → plan-review-hook-CVZsG9MZ.js.map} +1 -1
- package/dist/{prompt-submit-hook-BlijaOn7.js → prompt-submit-hook-BW92FX2D.js} +3 -3
- package/dist/{prompt-submit-hook-BlijaOn7.js.map → prompt-submit-hook-BW92FX2D.js.map} +1 -1
- package/dist/{provision-DlT34zcf.js → provision-CUqPki1z.js} +4 -4
- package/dist/{provision-DlT34zcf.js.map → provision-CUqPki1z.js.map} +1 -1
- package/dist/{self-invocation-CKMjcA5F.js → self-invocation-CP_SOkrr.js} +2 -2
- package/dist/{self-invocation-CKMjcA5F.js.map → self-invocation-CP_SOkrr.js.map} +1 -1
- package/dist/{serve-orGUartp.js → serve-BASqoXb3.js} +42 -29
- package/dist/serve-BASqoXb3.js.map +1 -0
- package/dist/{server-setup-BRBe8gW4.js → server-setup-D5hilphf.js} +33 -213
- package/dist/server-setup-D5hilphf.js.map +1 -0
- package/dist/{start-sutbbkwc.js → start-Rfim4TeF.js} +3 -3
- package/dist/{start-sutbbkwc.js.map → start-Rfim4TeF.js.map} +1 -1
- package/dist/{stop-gate-hook-DgJ6sW8N.js → stop-gate-hook-DriRc9xN.js} +3 -3
- package/dist/{stop-gate-hook-DgJ6sW8N.js.map → stop-gate-hook-DriRc9xN.js.map} +1 -1
- package/dist/{stop-gate-policy-DG5yYWGn.js → stop-gate-policy-DMPanpoR.js} +2 -2
- package/dist/{stop-gate-policy-DG5yYWGn.js.map → stop-gate-policy-DMPanpoR.js.map} +1 -1
- package/dist/{token-CnlB0884.js → token-BGCjZwtj.js} +2 -2
- package/dist/{token-CnlB0884.js.map → token-BGCjZwtj.js.map} +1 -1
- package/dist/{worker-dispatch-4O3IWtP0.js → worker-dispatch-BCTMyNE-.js} +9 -5
- package/dist/{worker-dispatch-4O3IWtP0.js.map → worker-dispatch-BCTMyNE-.js.map} +1 -1
- package/package.json +1 -1
- package/dist/attribution-settings-srL3gBR9.js.map +0 -1
- package/dist/claude-DJQjA5oQ.js.map +0 -1
- package/dist/engine-BvLgh1QI.js +0 -2
- package/dist/lifecycle-CbmHMMSD.js +0 -2
- package/dist/lifecycle-DTcZwa7U.js +0 -2
- package/dist/paths-CV9K7Xqm.js +0 -2
- package/dist/peer-mcp-personas-l4P4s1fK.js.map +0 -1
- package/dist/serve-orGUartp.js.map +0 -1
- package/dist/server-setup-BRBe8gW4.js.map +0 -1
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
|
|
2
2
|
import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
|
|
3
|
-
import { t as PATHS } from "./paths-
|
|
4
|
-
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-
|
|
3
|
+
import { t as PATHS } from "./paths-D7_SAaIQ.js";
|
|
4
|
+
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-BGCjZwtj.js";
|
|
5
5
|
import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
|
|
6
|
-
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-
|
|
7
|
-
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-
|
|
6
|
+
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-DbM29FLK.js";
|
|
7
|
+
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-DRCCWlOi.js";
|
|
8
8
|
import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
|
|
9
|
-
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-
|
|
10
|
-
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-
|
|
11
|
-
import { t as liveExec } from "./orchestration-
|
|
9
|
+
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-SXaWssN9.js";
|
|
10
|
+
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DriRc9xN.js";
|
|
11
|
+
import { t as liveExec } from "./orchestration-BrJwZxMN.js";
|
|
12
12
|
import { createRequire } from "node:module";
|
|
13
13
|
import consola from "consola";
|
|
14
14
|
import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
|
|
@@ -163,6 +163,221 @@ function withOneMSuffix(id) {
|
|
|
163
163
|
if (oneMContextDisabled()) return id;
|
|
164
164
|
return catalogAdvertises1M(id) ? `${id}[1m]` : id;
|
|
165
165
|
}
|
|
166
|
+
/**
|
|
167
|
+
* Decorate a slug the USER named — a `-m` argument or a launcher default — with
|
|
168
|
+
* `[1m]` iff the model it actually RESOLVES to serves >=1M context.
|
|
169
|
+
*
|
|
170
|
+
* The difference from `withOneMSuffix` is the resolution step, and it exists
|
|
171
|
+
* because the two functions are handed different kinds of string. Every
|
|
172
|
+
* `withOneMSuffix` caller already holds a concrete catalog id from a catalog
|
|
173
|
+
* walk, so an exact-id match is both sufficient and the safer rule: inferring a
|
|
174
|
+
* match there could attach `[1m]` to a sibling that does not serve 1M. A lead
|
|
175
|
+
* slug is the opposite case — it is whatever the user typed, or an
|
|
176
|
+
* Anthropic-published dashed slug like `claude-opus-4-8` that the catalog
|
|
177
|
+
* carries in dotted form. Exact-id matching answers "no 1M" for those purely
|
|
178
|
+
* because it never found the entry, which is the silent under-accounting this
|
|
179
|
+
* function exists to stop.
|
|
180
|
+
*
|
|
181
|
+
* Resolving first also picks up the `-1m` SIBLING shape for free:
|
|
182
|
+
* `resolveModel`'s opus family preference maps `claude-opus-4-7` onto
|
|
183
|
+
* `claude-opus-4.7-1m-internal` when that is what the tier carries, and the
|
|
184
|
+
* sibling's own advertised window then answers the question. That is the same
|
|
185
|
+
* dual-signal conclusion `pickClaudeDefault` reaches for the family shorthand,
|
|
186
|
+
* so the two paths cannot disagree about a family both can be asked about.
|
|
187
|
+
*
|
|
188
|
+
* Idempotent: a slug that already carries the bracket is returned unchanged, so
|
|
189
|
+
* a user who pins `-m claude-opus-5[1m]` by hand does not get `[1m][1m]`. That
|
|
190
|
+
* early return deliberately does NOT re-validate the pin against the catalog.
|
|
191
|
+
* `-m claude-haiku-4-5[1m]` therefore survives even though Haiku 4.5 is a 200K
|
|
192
|
+
* model — the same as before this function existed, and `resolveModel` already
|
|
193
|
+
* warns loudly about exactly that case. Stripping a bracket the user typed
|
|
194
|
+
* would be the surprising behaviour, and it would be the only place in the
|
|
195
|
+
* launcher that overrides an explicit `-m`.
|
|
196
|
+
*
|
|
197
|
+
* A repeat can still arrive from the CLIENT side rather than from here: the
|
|
198
|
+
* `/model` picker rows are seeded already decorated, and Claude Code's alias
|
|
199
|
+
* path appends its own bracket (`getDefaultSonnetModel() + '[1m]'`), so
|
|
200
|
+
* selecting `sonnet[1m]` puts `claude-sonnet-5[1m][1m]` on the wire. That
|
|
201
|
+
* resolves to the same bare id — `resolveModel`'s strip recurses — and Claude
|
|
202
|
+
* Code's own detector is unanchored, so local accounting is right too. Pinned
|
|
203
|
+
* by a regression test in `tests/lib-utils.test.ts`.
|
|
204
|
+
*
|
|
205
|
+
* Degrades the same safe direction as everything else here. An unpopulated
|
|
206
|
+
* catalog makes `resolveModel` a pass-through and `catalogAdvertises1M` false,
|
|
207
|
+
* so the slug stays bare and Claude Code accounts at its conservative 200K
|
|
208
|
+
* default — under-accounting, never overflow.
|
|
209
|
+
*/
|
|
210
|
+
function withOneMSuffixForLead(slug) {
|
|
211
|
+
if (oneMContextDisabled()) return slug;
|
|
212
|
+
if (/\[1m\]$/i.test(slug)) return slug;
|
|
213
|
+
return catalogAdvertises1M(resolveModel(slug)) ? `${slug}[1m]` : slug;
|
|
214
|
+
}
|
|
215
|
+
//#endregion
|
|
216
|
+
//#region src/services/copilot/endpoint.ts
|
|
217
|
+
/**
|
|
218
|
+
* Catalog spellings that mean each of our two clients. Copilot is not
|
|
219
|
+
* self-consistent about the `/v1` prefix — the live catalog advertises
|
|
220
|
+
* `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
|
|
221
|
+
* fixtures carry both forms — so an exact-match on the bare spelling alone
|
|
222
|
+
* silently misses a real shape. `src/lib/model-validation.ts` already
|
|
223
|
+
* normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
|
|
224
|
+
*
|
|
225
|
+
* Matching is EXACT against this set, never a suffix/`includes` test: a
|
|
226
|
+
* `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
|
|
227
|
+
* client and must keep resolving to "serves neither".
|
|
228
|
+
*/
|
|
229
|
+
const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
|
|
230
|
+
const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
231
|
+
/**
|
|
232
|
+
* Decide which endpoint to call for a model from its catalog
|
|
233
|
+
* `supported_endpoints`. Prefers `/chat/completions` when available (the
|
|
234
|
+
* simpler, more widely-supported shape) and falls back to `/responses` for
|
|
235
|
+
* models that ONLY serve the Responses API — the gpt-5.x family except
|
|
236
|
+
* `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
|
|
237
|
+
* `*-codex` models). Returns undefined when the model serves neither, so a
|
|
238
|
+
* caller can skip it rather than 400 on `unsupported_api_for_model`.
|
|
239
|
+
*
|
|
240
|
+
* A model that OMITS `supported_endpoints` is treated as chat-eligible: the
|
|
241
|
+
* catalog historically omits the field for chat-default models, and
|
|
242
|
+
* excluding those would be a worse regression than the gap this guards.
|
|
243
|
+
*/
|
|
244
|
+
function pickEndpoint(model) {
|
|
245
|
+
const eps = model.supported_endpoints;
|
|
246
|
+
if (!eps || eps.length === 0) return "chat";
|
|
247
|
+
if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
|
|
248
|
+
if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
|
|
252
|
+
* "absent from the catalog" into "serves neither of our endpoints".
|
|
253
|
+
*
|
|
254
|
+
* This function deliberately has no default. The predecessor
|
|
255
|
+
* (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
|
|
256
|
+
* coerced both cases to "chat" — defensible for an unknown id, silently wrong
|
|
257
|
+
* for a catalog model serving only, say, `/v1/messages`: the caller would drive
|
|
258
|
+
* it through the chat client and get an opaque upstream 400 with no local
|
|
259
|
+
* signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
|
|
260
|
+
* treats that case correctly (`if (!endpoint) continue`); this makes the same
|
|
261
|
+
* distinction available to callers that resolve by id.
|
|
262
|
+
*
|
|
263
|
+
* Callers that legitimately want the chat default for an unknown id can still
|
|
264
|
+
* have it — they just have to write it, per case, on purpose.
|
|
265
|
+
*/
|
|
266
|
+
function resolveEndpointForModelId(id) {
|
|
267
|
+
const found = state.models?.data?.find((m) => m.id === id);
|
|
268
|
+
if (!found) return { kind: "unknown-model" };
|
|
269
|
+
const endpoint = pickEndpoint(found);
|
|
270
|
+
if (endpoint) return {
|
|
271
|
+
kind: "endpoint",
|
|
272
|
+
endpoint
|
|
273
|
+
};
|
|
274
|
+
return {
|
|
275
|
+
kind: "unreachable",
|
|
276
|
+
endpoints: found.supported_endpoints ?? []
|
|
277
|
+
};
|
|
278
|
+
}
|
|
279
|
+
//#endregion
|
|
280
|
+
//#region src/lib/anthropic-translate/classifier.ts
|
|
281
|
+
/**
|
|
282
|
+
* Routing classifier for `POST /v1/messages`.
|
|
283
|
+
*
|
|
284
|
+
* Claude Code speaks the Anthropic Messages wire format. Copilot only serves
|
|
285
|
+
* Claude models on its native `/v1/messages` endpoint; a gpt/gemini request
|
|
286
|
+
* sent there 400s. This classifier decides, from the RESOLVED model id and its
|
|
287
|
+
* catalog metadata, whether a request stays on the native passthrough
|
|
288
|
+
* (`createMessages`) or is diverted to the translation shim.
|
|
289
|
+
*
|
|
290
|
+
* Non-regression is the whole point — every Claude model (opus / sonnet / haiku /
|
|
291
|
+
* any `claude-*` or Anthropic-vendored id) MUST return "claude-passthrough" so
|
|
292
|
+
* its bytes reach `createMessages` unchanged — even if future catalog metadata
|
|
293
|
+
* were to (wrongly) advertise a `/responses` endpoint for it. The Claude check
|
|
294
|
+
* is therefore keyed off identity (id / vendor / family), NOT the endpoint.
|
|
295
|
+
*
|
|
296
|
+
* Non-Claude models are diverted to the translation shim by the endpoint the
|
|
297
|
+
* catalog says they serve: `/responses` models (gpt-5.5, gpt-5.3-codex, …) take
|
|
298
|
+
* the Responses path (`responses-shim`), `/chat/completions` models (gemini,
|
|
299
|
+
* and any chat-default model) take the chat path (`chat-shim`). The decision is
|
|
300
|
+
* derived from `pickEndpoint` (catalog `supported_endpoints`), never a
|
|
301
|
+
* hardcoded slug list, so it generalizes. Copilot only serves Claude models on
|
|
302
|
+
* its native `/v1/messages`, so diverting every non-Claude model to a shim is
|
|
303
|
+
* correct — a non-Claude request sent to `/v1/messages` would 400.
|
|
304
|
+
*/
|
|
305
|
+
/**
|
|
306
|
+
* Match "claude"/"anthropic" as a delimiter-bounded segment anywhere in a model
|
|
307
|
+
* id: at the start, or after a `/ _ . : -` path/version delimiter, and followed
|
|
308
|
+
* by end-of-string, another such delimiter, or a digit. This catches catalog
|
|
309
|
+
* aliases like `github/claude-3-7-sonnet` or `anthropic/…` whose vendor is
|
|
310
|
+
* "github" and whose family is empty — where the token only surfaces mid-id —
|
|
311
|
+
* while NOT firing on incidental substrings like `notclaude`. Deliberately
|
|
312
|
+
* over-inclusive at the boundaries: we fail CLOSED toward Claude (→ passthrough)
|
|
313
|
+
* so a real Claude model can never be diverted to the non-Claude shim.
|
|
314
|
+
*/
|
|
315
|
+
const CLAUDE_ID_RE = /(^|[/_.:-])(claude|anthropic)(?=$|[/_.:-]|\d)/i;
|
|
316
|
+
/**
|
|
317
|
+
* True when the target is a Claude / Anthropic model. Matches on any of:
|
|
318
|
+
* catalog vendor containing "anthropic", capability family containing "claude",
|
|
319
|
+
* or a Claude/anthropic path-segment (see `CLAUDE_ID_RE`) in ANY id we hold —
|
|
320
|
+
* the resolved id (`modelId`), the pre-resolution request id (`originalModelId`),
|
|
321
|
+
* or the catalog entry's own id (`model.id`). Conservative by design: when in
|
|
322
|
+
* doubt it returns true so a Claude request can never be diverted to the shim.
|
|
323
|
+
*/
|
|
324
|
+
function isClaudeModel(modelId, model, originalModelId) {
|
|
325
|
+
if (model) {
|
|
326
|
+
if ((model.vendor?.toLowerCase() ?? "").includes("anthropic")) return true;
|
|
327
|
+
if ((model.capabilities?.family?.toLowerCase() ?? "").includes("claude")) return true;
|
|
328
|
+
}
|
|
329
|
+
return [
|
|
330
|
+
modelId,
|
|
331
|
+
originalModelId,
|
|
332
|
+
model?.id
|
|
333
|
+
].some((id) => typeof id === "string" && CLAUDE_ID_RE.test(id));
|
|
334
|
+
}
|
|
335
|
+
/**
|
|
336
|
+
* Decide the route for a resolved model id + its catalog entry.
|
|
337
|
+
*
|
|
338
|
+
* - No model id, or a Claude model → "claude-passthrough" (existing behaviour).
|
|
339
|
+
* - A non-Claude model whose catalog endpoint is `/responses` → "responses-shim".
|
|
340
|
+
* - A non-Claude model whose catalog endpoint is `/chat/completions` (gemini and
|
|
341
|
+
* any chat-default model) → "chat-shim".
|
|
342
|
+
* - A non-Claude model absent from the catalog (so we can't confirm an endpoint)
|
|
343
|
+
* → "claude-passthrough" (unchanged; we don't divert what we can't classify).
|
|
344
|
+
* - A non-Claude model that IS in the catalog and serves NEITHER of the two shim
|
|
345
|
+
* endpoints (`pickEndpoint` → undefined) → "claude-passthrough" as well.
|
|
346
|
+
*
|
|
347
|
+
* Those last two land on the same route but are NOT the same answer, and the
|
|
348
|
+
* coincidence is deliberate rather than a collapsed default (contrast
|
|
349
|
+
* `resolveEndpointForModelId`, whose callers must tell them apart because
|
|
350
|
+
* guessing there produces an opaque upstream 400). Here neither shim is even a
|
|
351
|
+
* candidate: a shim can only speak `/responses` or `/chat/completions`, so
|
|
352
|
+
* diverting a model that serves neither would 400 just as surely. Passthrough
|
|
353
|
+
* is the better default because it is sometimes RIGHT — a non-Claude catalog
|
|
354
|
+
* model advertising `/v1/messages` is served by exactly the endpoint
|
|
355
|
+
* passthrough uses. It also preserves this module's fail-CLOSED-toward-Claude
|
|
356
|
+
* invariant: an unclassifiable model is never diverted.
|
|
357
|
+
*
|
|
358
|
+
* KNOWN GAP (audited, deliberately not fixed here): a non-Claude model serving
|
|
359
|
+
* only something we cannot speak at all (say `/embeddings`) also lands on
|
|
360
|
+
* passthrough and will still 400 upstream — `logEndpointMismatch(modelId,
|
|
361
|
+
* "/v1/messages")` logs it at the passthrough seam, but no local error is
|
|
362
|
+
* raised. Closing that needs a change in `src/routes/messages/handler.ts`,
|
|
363
|
+
* which this seam does not own. It is strictly narrower than the defect fixed
|
|
364
|
+
* in `resolveEndpointForModelId`: no such model is reachable as a Claude Code
|
|
365
|
+
* `/v1/messages` target today, whereas the plan worker's `claude-opus-5`
|
|
366
|
+
* default is.
|
|
367
|
+
*
|
|
368
|
+
* `originalModelId` is the optional pre-resolution request id; when supplied it
|
|
369
|
+
* is checked for Claude-likeness alongside the resolved id so an alias that
|
|
370
|
+
* resolves to a non-Claude-looking id can't slip past.
|
|
371
|
+
*/
|
|
372
|
+
function classifyMessagesRoute(modelId, model, originalModelId) {
|
|
373
|
+
if (!modelId) return "claude-passthrough";
|
|
374
|
+
if (isClaudeModel(modelId, model, originalModelId)) return "claude-passthrough";
|
|
375
|
+
if (!model) return "claude-passthrough";
|
|
376
|
+
const endpoint = pickEndpoint(model);
|
|
377
|
+
if (endpoint === "responses") return "responses-shim";
|
|
378
|
+
if (endpoint === "chat") return "chat-shim";
|
|
379
|
+
return "claude-passthrough";
|
|
380
|
+
}
|
|
166
381
|
//#endregion
|
|
167
382
|
//#region src/lib/port.ts
|
|
168
383
|
const DEFAULT_PORT = 8787;
|
|
@@ -209,11 +424,22 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
|
|
|
209
424
|
* This helper detects the catalog state at launch and only opts in
|
|
210
425
|
* when the backend can actually serve 1M.
|
|
211
426
|
*
|
|
212
|
-
*
|
|
213
|
-
*
|
|
214
|
-
*
|
|
215
|
-
* `
|
|
216
|
-
*
|
|
427
|
+
* This helper answers the question only for the OPUS families, because a
|
|
428
|
+
* family is what it is asked about (`-m 4.7` names no slug). Every other lead
|
|
429
|
+
* slug — `-m fast`, a full slug a power user pins, the implicit budget lead —
|
|
430
|
+
* goes through `withOneMSuffixForLead` (`./one-m-context`) instead, which
|
|
431
|
+
* resolves the slug first and then reads the resolved entry's advertised
|
|
432
|
+
* window. The two agree wherever both can be asked: a family that resolves to a
|
|
433
|
+
* 1M backend is 1M by either route.
|
|
434
|
+
*
|
|
435
|
+
* A previous revision of this comment claimed Sonnet and Haiku were left bare
|
|
436
|
+
* because "Copilot has no 1M backend for them". That was true when it was
|
|
437
|
+
* written and is now false for Sonnet: the live catalog advertises
|
|
438
|
+
* `max_context_window_tokens: 1_000_000` on both `claude-sonnet-5` and
|
|
439
|
+
* `claude-sonnet-4.6` (Haiku 4.5 really is 200K, and is left bare by the same
|
|
440
|
+
* catalog check rather than by a hardcoded family rule). Nothing here is
|
|
441
|
+
* family-gated any more — the catalog decides per model, so the next family
|
|
442
|
+
* that ships 1M is picked up without an edit.
|
|
217
443
|
*
|
|
218
444
|
* Must be called AFTER `cacheModels()` has populated `state.models`.
|
|
219
445
|
* Returns the bare slug if the catalog isn't populated (resolveModel
|
|
@@ -221,6 +447,88 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
|
|
|
221
447
|
* variant" — defaulting safe-side preserves the pre-change behavior).
|
|
222
448
|
*/
|
|
223
449
|
const DEFAULT_OPUS_FAMILY = "5";
|
|
450
|
+
/** The lead `-m fast` selects. Anthropic-published dashed slug that is also the
|
|
451
|
+
* Copilot catalog id verbatim, so `resolveModel` matches it exactly. */
|
|
452
|
+
const BUDGET_LEAD_MODEL = "claude-sonnet-5";
|
|
453
|
+
/** Small/fast tier for a budget lead, in the two forms this codebase needs.
|
|
454
|
+
*
|
|
455
|
+
* `SLUG` is the Anthropic-published DASHED form and is what goes into
|
|
456
|
+
* `ANTHROPIC_SMALL_FAST_MODEL` / `ANTHROPIC_DEFAULT_HAIKU_MODEL`: Claude Code's
|
|
457
|
+
* `/model` registry is keyed on Anthropic slugs, and seeding Copilot's dotted
|
|
458
|
+
* id there reproduces the documented `claude-opus-5` failure where the picker
|
|
459
|
+
* silently falls back to an older model. `CATALOG_ID` is Copilot's DOTTED id
|
|
460
|
+
* and is what the presence probe must test, because that is the id the catalog
|
|
461
|
+
* actually carries. `resolveModel` bridges the two at request time. */
|
|
462
|
+
const BUDGET_SMALL_FAST_SLUG = "claude-haiku-4-5";
|
|
463
|
+
const BUDGET_SMALL_FAST_CATALOG_ID = "claude-haiku-4.5";
|
|
464
|
+
/**
|
|
465
|
+
* Resolve the `-m` argument to the lead slug to launch with.
|
|
466
|
+
*
|
|
467
|
+
* - `fast` → `BUDGET_LEAD_MODEL` (budget mode)
|
|
468
|
+
* - `N.M` → the best variant of that Opus family, via `pickClaudeDefault`
|
|
469
|
+
* - a full slug → unchanged, including Copilot slugs a power user pins
|
|
470
|
+
* - absent → the ordinary default
|
|
471
|
+
*
|
|
472
|
+
* Every branch is `[1m]`-decorated against the live catalog, by
|
|
473
|
+
* `pickClaudeDefault` on the two Opus-family branches and by
|
|
474
|
+
* `withOneMSuffixForLead` on the other two. Pinning a MODEL is not a request to
|
|
475
|
+
* give up four fifths of its context window, which is what leaving the other
|
|
476
|
+
* two bare amounted to: `claude-sonnet-5` advertises a 1M window, so `-m fast`
|
|
477
|
+
* and `-m claude-sonnet-5` were both budgeted locally at Claude Code's 200K
|
|
478
|
+
* default and auto-compacted at roughly a fifth of the real window. The
|
|
479
|
+
* decoration is catalog-gated per model, so a genuinely 200K model
|
|
480
|
+
* (`claude-haiku-4.5`) still comes back bare.
|
|
481
|
+
*
|
|
482
|
+
* `fast` resolves to an ordinary slug rather than setting a mode flag, because
|
|
483
|
+
* budget mode is keyed off the RESOLVED lead everywhere it matters (the advisor
|
|
484
|
+
* escalation, the delegation prose, the small/fast tier). `-m fast` and
|
|
485
|
+
* `-m claude-sonnet-5` must therefore produce identical sessions, which a flag
|
|
486
|
+
* only one of the two set would break. The shared decoration is part of that
|
|
487
|
+
* identity: decorating one branch and not the other would reintroduce the
|
|
488
|
+
* divergence through the context budget instead of through a flag.
|
|
489
|
+
*
|
|
490
|
+
* Callers must keep treating any explicit `-m` as explicit: the
|
|
491
|
+
* `DEFAULT_CLAUDE_MODEL_FALLBACKS` walk applies to the implicit-default path
|
|
492
|
+
* only, so it cannot override a requested family (or `fast`) with an older Opus.
|
|
493
|
+
*/
|
|
494
|
+
function resolveLeadSlugArg(modelArg) {
|
|
495
|
+
const arg = modelArg?.trim();
|
|
496
|
+
if (!arg) return pickClaudeDefault();
|
|
497
|
+
if (arg.toLowerCase() === "fast") return withOneMSuffixForLead(BUDGET_LEAD_MODEL);
|
|
498
|
+
const opusFamilyShorthand = arg.match(/^(\d+\.\d+)$/)?.[1];
|
|
499
|
+
if (opusFamilyShorthand) return pickClaudeDefault(opusFamilyShorthand);
|
|
500
|
+
return withOneMSuffixForLead(arg);
|
|
501
|
+
}
|
|
502
|
+
/**
|
|
503
|
+
* True when `slug` names a Claude model that is NOT an Opus tier — the
|
|
504
|
+
* "budget lead" condition.
|
|
505
|
+
*
|
|
506
|
+
* Selecting sonnet or haiku as the lead is a decision to spend less while
|
|
507
|
+
* holding quality as far as possible, and three surfaces key off it: the
|
|
508
|
+
* advisor escalates to the Anthropic frontier (`resolveAdvisorModel`), the
|
|
509
|
+
* injected delegation prose puts the cheap agent tiers first
|
|
510
|
+
* (`buildNativeReachClauses`), and the small/fast tier drops to Haiku
|
|
511
|
+
* (`getClaudeCodeEnvVars`). One definition here so those three cannot disagree
|
|
512
|
+
* about what counts as a budget lead.
|
|
513
|
+
*
|
|
514
|
+
* Resolves before the family test so the Anthropic dashed form, Copilot's
|
|
515
|
+
* dotted form, and `pickClaudeDefault`'s literal `[1m]` suffix all classify
|
|
516
|
+
* alike. A non-Claude lead is not a budget lead: the concept is about picking a
|
|
517
|
+
* lighter tier WITHIN the Claude family, and the gpt/gemini shim models have
|
|
518
|
+
* their own cost profile that this switch says nothing about.
|
|
519
|
+
*
|
|
520
|
+
* CONTRACT: `slug` is an already-resolved LEAD SLUG, never a raw `-m` argument.
|
|
521
|
+
* `"fast"` and the `N.M` shorthand are not Claude slugs and would classify
|
|
522
|
+
* false here; run them through `resolveLeadSlugArg` first, which is what every
|
|
523
|
+
* caller does. Resolving internally instead would drag `pickClaudeDefault`'s
|
|
524
|
+
* catalog dependency into a pure predicate and make the same input answer
|
|
525
|
+
* differently before and after the catalog loads.
|
|
526
|
+
*/
|
|
527
|
+
function isBudgetClaudeLead(slug) {
|
|
528
|
+
if (!slug) return false;
|
|
529
|
+
if (!isClaudeModel(slug)) return false;
|
|
530
|
+
return !/opus/i.test(resolveModel(slug));
|
|
531
|
+
}
|
|
224
532
|
function pickClaudeDefault(opusFamily = DEFAULT_OPUS_FAMILY) {
|
|
225
533
|
const dotted = opusFamily.replace(/-/g, ".");
|
|
226
534
|
const bareSlug = `claude-opus-${dotted.replace(/\./g, "-")}`;
|
|
@@ -687,6 +995,9 @@ function enumProp$1(values, description) {
|
|
|
687
995
|
};
|
|
688
996
|
}
|
|
689
997
|
//#endregion
|
|
998
|
+
//#region src/lib/gemini-review-model.ts
|
|
999
|
+
const GEMINI_REVIEW_DEFAULT_MODEL = "gemini-3.1-pro-preview";
|
|
1000
|
+
//#endregion
|
|
690
1001
|
//#region src/lib/fleet/mesh-egress-agent.ts
|
|
691
1002
|
/**
|
|
692
1003
|
* Runtime-aware "route this fetch through the mesh loopback egress proxy" for a
|
|
@@ -18606,7 +18917,7 @@ function logAudit$1(record) {
|
|
|
18606
18917
|
try {
|
|
18607
18918
|
const fs = await import("node:fs/promises");
|
|
18608
18919
|
const path = await import("node:path");
|
|
18609
|
-
const { PATHS } = await import("./paths-
|
|
18920
|
+
const { PATHS } = await import("./paths-CTr59UC6.js");
|
|
18610
18921
|
const dir = path.join(PATHS.APP_DIR, "browser-mcp");
|
|
18611
18922
|
await fs.mkdir(dir, { recursive: true });
|
|
18612
18923
|
const line = JSON.stringify({
|
|
@@ -19985,70 +20296,6 @@ function detectAgentCall(input) {
|
|
|
19985
20296
|
});
|
|
19986
20297
|
}
|
|
19987
20298
|
//#endregion
|
|
19988
|
-
//#region src/services/copilot/endpoint.ts
|
|
19989
|
-
/**
|
|
19990
|
-
* Catalog spellings that mean each of our two clients. Copilot is not
|
|
19991
|
-
* self-consistent about the `/v1` prefix — the live catalog advertises
|
|
19992
|
-
* `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
|
|
19993
|
-
* fixtures carry both forms — so an exact-match on the bare spelling alone
|
|
19994
|
-
* silently misses a real shape. `src/lib/model-validation.ts` already
|
|
19995
|
-
* normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
|
|
19996
|
-
*
|
|
19997
|
-
* Matching is EXACT against this set, never a suffix/`includes` test: a
|
|
19998
|
-
* `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
|
|
19999
|
-
* client and must keep resolving to "serves neither".
|
|
20000
|
-
*/
|
|
20001
|
-
const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
|
|
20002
|
-
const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
20003
|
-
/**
|
|
20004
|
-
* Decide which endpoint to call for a model from its catalog
|
|
20005
|
-
* `supported_endpoints`. Prefers `/chat/completions` when available (the
|
|
20006
|
-
* simpler, more widely-supported shape) and falls back to `/responses` for
|
|
20007
|
-
* models that ONLY serve the Responses API — the gpt-5.x family except
|
|
20008
|
-
* `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
|
|
20009
|
-
* `*-codex` models). Returns undefined when the model serves neither, so a
|
|
20010
|
-
* caller can skip it rather than 400 on `unsupported_api_for_model`.
|
|
20011
|
-
*
|
|
20012
|
-
* A model that OMITS `supported_endpoints` is treated as chat-eligible: the
|
|
20013
|
-
* catalog historically omits the field for chat-default models, and
|
|
20014
|
-
* excluding those would be a worse regression than the gap this guards.
|
|
20015
|
-
*/
|
|
20016
|
-
function pickEndpoint(model) {
|
|
20017
|
-
const eps = model.supported_endpoints;
|
|
20018
|
-
if (!eps || eps.length === 0) return "chat";
|
|
20019
|
-
if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
|
|
20020
|
-
if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
|
|
20021
|
-
}
|
|
20022
|
-
/**
|
|
20023
|
-
* `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
|
|
20024
|
-
* "absent from the catalog" into "serves neither of our endpoints".
|
|
20025
|
-
*
|
|
20026
|
-
* This function deliberately has no default. The predecessor
|
|
20027
|
-
* (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
|
|
20028
|
-
* coerced both cases to "chat" — defensible for an unknown id, silently wrong
|
|
20029
|
-
* for a catalog model serving only, say, `/v1/messages`: the caller would drive
|
|
20030
|
-
* it through the chat client and get an opaque upstream 400 with no local
|
|
20031
|
-
* signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
|
|
20032
|
-
* treats that case correctly (`if (!endpoint) continue`); this makes the same
|
|
20033
|
-
* distinction available to callers that resolve by id.
|
|
20034
|
-
*
|
|
20035
|
-
* Callers that legitimately want the chat default for an unknown id can still
|
|
20036
|
-
* have it — they just have to write it, per case, on purpose.
|
|
20037
|
-
*/
|
|
20038
|
-
function resolveEndpointForModelId(id) {
|
|
20039
|
-
const found = state.models?.data?.find((m) => m.id === id);
|
|
20040
|
-
if (!found) return { kind: "unknown-model" };
|
|
20041
|
-
const endpoint = pickEndpoint(found);
|
|
20042
|
-
if (endpoint) return {
|
|
20043
|
-
kind: "endpoint",
|
|
20044
|
-
endpoint
|
|
20045
|
-
};
|
|
20046
|
-
return {
|
|
20047
|
-
kind: "unreachable",
|
|
20048
|
-
endpoints: found.supported_endpoints ?? []
|
|
20049
|
-
};
|
|
20050
|
-
}
|
|
20051
|
-
//#endregion
|
|
20052
20299
|
//#region src/lib/browser-mcp/compressor.ts
|
|
20053
20300
|
/**
|
|
20054
20301
|
* Static fallback chain for the inner compressor. Order is preference:
|
|
@@ -23672,6 +23919,10 @@ const FALLBACK_TOKEN_PRICES = Object.freeze({
|
|
|
23672
23919
|
"gemini-3.1-pro-preview": {
|
|
23673
23920
|
in: 200,
|
|
23674
23921
|
out: 1200
|
|
23922
|
+
},
|
|
23923
|
+
"gemini-3.7-flash": {
|
|
23924
|
+
in: 75,
|
|
23925
|
+
out: 375
|
|
23675
23926
|
}
|
|
23676
23927
|
});
|
|
23677
23928
|
/**
|
|
@@ -23732,6 +23983,22 @@ function catalogTokenPrices(modelId) {
|
|
|
23732
23983
|
* takes ~0.9s. That figure is deliberately NOT surfaced to the model, because a
|
|
23733
23984
|
* second speed axis invites optimising a routing choice that policy already
|
|
23734
23985
|
* settles (see the decorrelation note below).
|
|
23986
|
+
*
|
|
23987
|
+
* `gemini-3.7-flash` and the re-measured `gemini-3.1-pro-preview` (2026-08-13,
|
|
23988
|
+
* n=5, one untimed warmup rep) used the script's newer `GH_ROUTER_BENCH_STREAM=1`
|
|
23989
|
+
* mode, which splits wall-clock into TTFT and a decode-phase rate excluding the
|
|
23990
|
+
* first delta's own time+tokens. This row still records the SAME metric family
|
|
23991
|
+
* as every other row here (total tokens / total wall-clock, i.e. the script's
|
|
23992
|
+
* "total tok/s" column) — NOT the new decode-phase figure — so the table stays
|
|
23993
|
+
* internally comparable across rows measured at different times. The decode
|
|
23994
|
+
* figure is materially different and worth knowing for routing decisions that
|
|
23995
|
+
* care about steady-state throughput specifically: at matched effort,
|
|
23996
|
+
* `gemini-3.7-flash` decodes at ~275-315 tok/s (Google's own ~340 tok/s
|
|
23997
|
+
* Artificial Analysis figure, roughly confirmed once TTFT is excluded) against
|
|
23998
|
+
* `gpt-5.6-terra`'s ~135-190 tok/s decode — gemini-3.7-flash is the faster
|
|
23999
|
+
* decoder, but its ~1.1-1.7s TTFT (vs terra's ~0.9-1.0s) drags its TOTAL rate
|
|
24000
|
+
* below terra's on short responses, which is exactly why this row uses total,
|
|
24001
|
+
* not decode, to stay consistent with its neighbors.
|
|
23735
24002
|
*/
|
|
23736
24003
|
const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
|
|
23737
24004
|
"gpt-5.6-luna": 120,
|
|
@@ -23744,9 +24011,10 @@ const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
|
|
|
23744
24011
|
"gpt-5.6-sol": 75,
|
|
23745
24012
|
"grok-4.5": 70,
|
|
23746
24013
|
"gpt-5.5": 65,
|
|
24014
|
+
"gemini-3.7-flash": 91,
|
|
23747
24015
|
"gemini-3.6-flash": 45,
|
|
23748
24016
|
"gemini-3.5-flash": 40,
|
|
23749
|
-
"gemini-3.1-pro-preview":
|
|
24017
|
+
"gemini-3.1-pro-preview": 33
|
|
23750
24018
|
});
|
|
23751
24019
|
/** Returns the approximate, indicative output speed when it was measured. */
|
|
23752
24020
|
function indicativeTokensPerSecond(modelId) {
|
|
@@ -26769,6 +27037,7 @@ function shimDefaultsToXhigh(id) {
|
|
|
26769
27037
|
* live tool list would be a silent regression (the snippet would name
|
|
26770
27038
|
* a tool the live catalog doesn't expose).
|
|
26771
27039
|
*/
|
|
27040
|
+
const REVIEW_FAST_DEFAULT_MODEL = "gemini-3.7-flash";
|
|
26772
27041
|
/**
|
|
26773
27042
|
* Gate for the `stand_in` tool.
|
|
26774
27043
|
*
|
|
@@ -26777,9 +27046,8 @@ function shimDefaultsToXhigh(id) {
|
|
|
26777
27046
|
* - an OpenAI frontier model (`gpt-5.6-sol`, else `gpt-5.5` — see
|
|
26778
27047
|
* `resolveOpenAiFrontier`)
|
|
26779
27048
|
* - `claude-opus-5` (stand_in's Anthropic slot)
|
|
26780
|
-
* -
|
|
26781
|
-
*
|
|
26782
|
-
* the GA slug renames `gemini-3.1-pro-preview` → `gemini-3.1-pro`)
|
|
27049
|
+
* - the preferred Gemini reviewer model (`gemini-3.1-pro-preview`, falling
|
|
27050
|
+
* back to `gemini-3.7-flash` after the preview's 2026-09-01 removal)
|
|
26783
27051
|
*
|
|
26784
27052
|
* If any one is missing, `stand_in` is dropped from `tools/list` AND
|
|
26785
27053
|
* fails `tools/call` with -32601 (mirroring the `worker` capability's
|
|
@@ -26788,10 +27056,47 @@ function shimDefaultsToXhigh(id) {
|
|
|
26788
27056
|
* `claude-opus-5` is a single-segment slug (dotted == dashed), so the
|
|
26789
27057
|
* catalog probe matches Copilot's actual id shape directly.
|
|
26790
27058
|
*/
|
|
26791
|
-
|
|
27059
|
+
/**
|
|
27060
|
+
* Any live-catalog model matching Google's `gemini-3.x-pro` family, excluding
|
|
27061
|
+
* the two known literals — catches a GA rename of the preview slug (e.g.
|
|
27062
|
+
* `gemini-3.1-pro-preview` -> `gemini-3.1-pro`) so a vendor rename doesn't
|
|
27063
|
+
* silently downgrade every Gemini-gated resolver to the flash fallback while a
|
|
27064
|
+
* real pro-tier successor is actually present in the catalog. This is the
|
|
27065
|
+
* same regex the removed `geminiAvailable()` used, for the same reason —
|
|
27066
|
+
* losing it here was a real regression caught in review, not a deliberate
|
|
27067
|
+
* simplification.
|
|
27068
|
+
*/
|
|
27069
|
+
function findGeminiProGaRename(models) {
|
|
27070
|
+
return models.find((m) => /^gemini-3\..*pro/i.test(m.id) && m.id !== "gemini-3.1-pro-preview" && m.id !== "gemini-3.7-flash")?.id;
|
|
27071
|
+
}
|
|
27072
|
+
function resolveGeminiReviewModel(source = state) {
|
|
26792
27073
|
const models = source.models?.data;
|
|
26793
|
-
if (!models) return
|
|
26794
|
-
|
|
27074
|
+
if (!models) return void 0;
|
|
27075
|
+
if (models.some((m) => m.id === "gemini-3.1-pro-preview")) return GEMINI_REVIEW_DEFAULT_MODEL;
|
|
27076
|
+
const gaRename = findGeminiProGaRename(models);
|
|
27077
|
+
if (gaRename) return gaRename;
|
|
27078
|
+
if (models.some((m) => m.id === "gemini-3.7-flash")) return REVIEW_FAST_DEFAULT_MODEL;
|
|
27079
|
+
}
|
|
27080
|
+
/**
|
|
27081
|
+
* Gemini review candidates in preference order, for resolvers that ALSO need
|
|
27082
|
+
* `firstPresentInCatalog`'s `requireToolCalls`/`minContextTokens` enforcement
|
|
27083
|
+
* (`resolveGeminiReviewModel()` only checks id presence, not those capability
|
|
27084
|
+
* flags). Mirrors `resolveGeminiReviewModel()`'s own preference order: the
|
|
27085
|
+
* known preview id, then a GA rename of it, then the flash fallback — kept as
|
|
27086
|
+
* a shared helper so `reviewerModel()`/`brainstormModel()` can't drift from
|
|
27087
|
+
* `resolveGeminiReviewModel()`'s GA-rename handling the way the hardcoded
|
|
27088
|
+
* per-resolver chains did before this was extracted.
|
|
27089
|
+
*/
|
|
27090
|
+
function geminiReviewChainCandidates() {
|
|
27091
|
+
const gaRename = findGeminiProGaRename(state.models?.data ?? []);
|
|
27092
|
+
return [
|
|
27093
|
+
GEMINI_REVIEW_DEFAULT_MODEL,
|
|
27094
|
+
...gaRename ? [gaRename] : [],
|
|
27095
|
+
REVIEW_FAST_DEFAULT_MODEL
|
|
27096
|
+
];
|
|
27097
|
+
}
|
|
27098
|
+
function geminiAvailable(source = state) {
|
|
27099
|
+
return resolveGeminiReviewModel(source) != null;
|
|
26795
27100
|
}
|
|
26796
27101
|
/**
|
|
26797
27102
|
* First id in `chain` that is present in the live catalog. With
|
|
@@ -26863,7 +27168,7 @@ function nativeSubagentModel() {
|
|
|
26863
27168
|
* was one model checking its own output. Not merely the same lab: the same
|
|
26864
27169
|
* model. Two independent blind audits flagged it, and the repo already applies
|
|
26865
27170
|
* the opposite rule one layer down, where `worker-review` runs
|
|
26866
|
-
* `
|
|
27171
|
+
* `GEMINI_REVIEW_DEFAULT_MODEL` precisely so the reviewer's lab is decorrelated from the
|
|
26867
27172
|
* producer's.
|
|
26868
27173
|
*
|
|
26869
27174
|
* The Anthropic lead and the OpenAI-frontier `implementer` are the two producers
|
|
@@ -26871,7 +27176,7 @@ function nativeSubagentModel() {
|
|
|
26871
27176
|
* frontier remains the fallback: a same-lab reviewer still beats no reviewer.
|
|
26872
27177
|
*/
|
|
26873
27178
|
function reviewerModel() {
|
|
26874
|
-
return firstPresentInCatalog([
|
|
27179
|
+
return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
|
|
26875
27180
|
}
|
|
26876
27181
|
/** Model for `brainstorm`. Absent → inherits the lead's model.
|
|
26877
27182
|
*
|
|
@@ -26880,7 +27185,7 @@ function reviewerModel() {
|
|
|
26880
27185
|
* `implementer`/`reviewer`, so a same-lab brainstormer would mostly restate
|
|
26881
27186
|
* what the lead already thought of. */
|
|
26882
27187
|
function brainstormModel() {
|
|
26883
|
-
return firstPresentInCatalog([
|
|
27188
|
+
return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
|
|
26884
27189
|
}
|
|
26885
27190
|
/** Model for `scribe`. Absent → inherits the lead's model.
|
|
26886
27191
|
*
|
|
@@ -26902,18 +27207,24 @@ function scribeModel() {
|
|
|
26902
27207
|
* impostor wearing the cheap agent's name.
|
|
26903
27208
|
*
|
|
26904
27209
|
* `gpt-5.6-luna` leads because it is the cheapest 1M-context model in the
|
|
26905
|
-
* catalog; `gemini-3.
|
|
27210
|
+
* catalog; `gemini-3.7-flash` remains the cross-vendor fallback so an OpenAI-side
|
|
26906
27211
|
* outage does not remove the scout. Both entries must continue advertising at
|
|
26907
27212
|
* least 1M context so Claude Code's `[1m]` accounting remains honest if an
|
|
26908
27213
|
* upstream catalog entry shrinks.
|
|
26909
27214
|
*
|
|
27215
|
+
* The fallback moved off `gemini-3.6-flash` on 2026-08-13: `gemini-3.7-flash`
|
|
27216
|
+
* is strictly better on every axis this chain cares about — half the price
|
|
27217
|
+
* (75/375 vs 150/750 per 1M), materially faster (measured tool-call p50 ~1.2s
|
|
27218
|
+
* against 3.6's ~2.6s), same 1M window, same vendor, so the cross-vendor
|
|
27219
|
+
* property the fallback exists for is preserved.
|
|
27220
|
+
*
|
|
26910
27221
|
* This chain deliberately uses literal ids rather than `EXPLORE_DEFAULT_MODEL`:
|
|
26911
27222
|
* the explore worker default and scout's cross-vendor fallback are independent
|
|
26912
27223
|
* policies, so retuning one must not silently collapse the other. There is no
|
|
26913
27224
|
* 400K last resort. On a catalog carrying neither chain member, `scout` is
|
|
26914
27225
|
* dropped rather than inheriting the lead or presenting a narrower-context agent.
|
|
26915
27226
|
*/
|
|
26916
|
-
const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.
|
|
27227
|
+
const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.7-flash"]);
|
|
26917
27228
|
function scoutModel() {
|
|
26918
27229
|
return firstPresentInCatalog(SCOUT_MODEL_CHAIN, {
|
|
26919
27230
|
requireToolCalls: true,
|
|
@@ -26928,7 +27239,17 @@ function scoutModel() {
|
|
|
26928
27239
|
* well-specified, mechanical changes at a lower tier. Both entries are 1M+;
|
|
26929
27240
|
* their different speed and effort properties stay out of shared claims. */
|
|
26930
27241
|
function implementerFastModel() {
|
|
26931
|
-
return firstPresentInCatalog(["gpt-5.6-terra",
|
|
27242
|
+
return firstPresentInCatalog(["gpt-5.6-terra", GEMINI_REVIEW_DEFAULT_MODEL], {
|
|
27243
|
+
requireToolCalls: true,
|
|
27244
|
+
minContextTokens: ONE_M_TOKENS
|
|
27245
|
+
});
|
|
27246
|
+
}
|
|
27247
|
+
/** Model for `reviewer-fast` — the cheaper Google review tier. Absent → the
|
|
27248
|
+
* agent is dropped. Single-entry by design: inheriting the lead or falling
|
|
27249
|
+
* across labs would defeat both its cost purpose and its decorrelation from
|
|
27250
|
+
* the OpenAI-backed implementer. */
|
|
27251
|
+
function reviewerFastModel() {
|
|
27252
|
+
return firstPresentInCatalog([REVIEW_FAST_DEFAULT_MODEL], {
|
|
26932
27253
|
requireToolCalls: true,
|
|
26933
27254
|
minContextTokens: ONE_M_TOKENS
|
|
26934
27255
|
});
|
|
@@ -27229,6 +27550,7 @@ function resolveOpusCriticModel() {
|
|
|
27229
27550
|
}
|
|
27230
27551
|
function activePersonas() {
|
|
27231
27552
|
return PERSONAS_READ.filter((p) => !p.requiresGeminiCatalog || geminiAvailable()).map((p) => {
|
|
27553
|
+
if (p.requiresGeminiCatalog) return resolveGeminiPersona(p, resolveGeminiReviewModel());
|
|
27232
27554
|
if (p.toolNameHttp !== "opus_critic") return p;
|
|
27233
27555
|
const model = resolveOpusCriticModel();
|
|
27234
27556
|
const allowedEfforts = model === "claude-opus-5" ? [
|
|
@@ -27627,8 +27949,26 @@ function logTelemetry(t) {
|
|
|
27627
27949
|
function toolAcceptsWorkspace(tool) {
|
|
27628
27950
|
return tool.capability === "worker" || tool.toolNameHttp === "code" || tool.toolNameHttp === "run_workflow";
|
|
27629
27951
|
}
|
|
27952
|
+
/**
|
|
27953
|
+
* Fold the per-session `X-GH-Workspace` header into `args.workspace` when the
|
|
27954
|
+
* caller left it empty, and REPORT which of the two the tool ended up with.
|
|
27955
|
+
*
|
|
27956
|
+
* The return value is the load-bearing part. This function mutates `args`, so
|
|
27957
|
+
* once it has run a header-derived workspace is byte-indistinguishable from one
|
|
27958
|
+
* the caller chose — and those two cases warrant very different treatment. A
|
|
27959
|
+
* caller that named a directory has told us where it is; a header is a
|
|
27960
|
+
* connection-level default that may be stale (it is computed by a helper Claude
|
|
27961
|
+
* Code runs, and the calling agent may since have moved into a git worktree).
|
|
27962
|
+
* `runWorkerToolCall` uses the distinction to decide what to tell the caller
|
|
27963
|
+
* about the tree it actually ran in, so the provenance must survive the merge.
|
|
27964
|
+
*/
|
|
27630
27965
|
function applySessionWorkspace(args, sessionWorkspace, tool) {
|
|
27631
|
-
if (
|
|
27966
|
+
if (args.workspace !== void 0 && args.workspace !== "") return "argument";
|
|
27967
|
+
if ((!tool || toolAcceptsWorkspace(tool)) && typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace)) {
|
|
27968
|
+
args.workspace = sessionWorkspace;
|
|
27969
|
+
return "session";
|
|
27970
|
+
}
|
|
27971
|
+
return "absent";
|
|
27632
27972
|
}
|
|
27633
27973
|
async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
27634
27974
|
const params = body.params ?? {};
|
|
@@ -27662,7 +28002,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
|
27662
28002
|
personaContext = typeof args.context === "string" ? args.context : void 0;
|
|
27663
28003
|
if (args.imagePaths !== void 0) {
|
|
27664
28004
|
if (!Array.isArray(args.imagePaths) || args.imagePaths.some((v) => typeof v !== "string")) return rpcError(body.id, RPC_INVALID_PARAMS, "tools/call: arguments.imagePaths must be an array of strings");
|
|
27665
|
-
const
|
|
28005
|
+
const imageRoot = typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace) ? sessionWorkspace : process.cwd();
|
|
28006
|
+
const loaded = await loadPeerImages(args.imagePaths, imageRoot);
|
|
27666
28007
|
if (!loaded.ok) return rpcError(body.id, RPC_INVALID_PARAMS, `tools/call: ${loaded.error}`);
|
|
27667
28008
|
personaImages = loaded.images;
|
|
27668
28009
|
}
|
|
@@ -27700,8 +28041,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
|
27700
28041
|
const telemetryName = persona ? persona.agentName : nonPersonaTool.toolNameHttp;
|
|
27701
28042
|
const telemetryModel = persona ? persona.model : "(non-persona)";
|
|
27702
28043
|
try {
|
|
27703
|
-
|
|
27704
|
-
const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal);
|
|
28044
|
+
const workspaceSource = nonPersonaTool ? applySessionWorkspace(args, sessionWorkspace, nonPersonaTool) : "absent";
|
|
28045
|
+
const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal, { workspaceSource });
|
|
27705
28046
|
logTelemetry({
|
|
27706
28047
|
name: telemetryName,
|
|
27707
28048
|
model: telemetryModel,
|
|
@@ -28027,6 +28368,98 @@ function handleMcpDelete(c) {
|
|
|
28027
28368
|
return c.body(null, 200);
|
|
28028
28369
|
}
|
|
28029
28370
|
//#endregion
|
|
28371
|
+
//#region src/lib/reasoning-effort.ts
|
|
28372
|
+
/**
|
|
28373
|
+
* Reasoning-effort bucketing + clamping shared by the `/v1/messages` handler
|
|
28374
|
+
* (adaptive-thinking translation) and the Anthropic-translation shim
|
|
28375
|
+
* (thinking-budget → Responses `reasoning.effort`).
|
|
28376
|
+
*
|
|
28377
|
+
* Extracted from `src/routes/messages/handler.ts` so a `~/lib/*` module can
|
|
28378
|
+
* depend on it without importing route code (and without forming a
|
|
28379
|
+
* handler → shim → handler import cycle). `handler.ts` re-exports these for
|
|
28380
|
+
* backward compatibility with existing imports/tests.
|
|
28381
|
+
*/
|
|
28382
|
+
/**
|
|
28383
|
+
* Copilot's reasoning-effort tiers, lowest to highest.
|
|
28384
|
+
*
|
|
28385
|
+
* Both ends were added after the fact and both are load-bearing:
|
|
28386
|
+
*
|
|
28387
|
+
* `none` is advertised by every gpt-5.x entry in the live catalog. While it was
|
|
28388
|
+
* missing here it was treated as an UNRECOGNIZED value, so a client asking for
|
|
28389
|
+
* the MINIMUM on a model that does not offer it (gemini advertises only
|
|
28390
|
+
* low/medium/high) was anchored at the unknown-value tier and clamped to
|
|
28391
|
+
* `high` — the maximum. Listing it makes the clamp resolve to `low` instead,
|
|
28392
|
+
* which is what "nearest supported tier" should always have meant.
|
|
28393
|
+
*
|
|
28394
|
+
* `max` is advertised by `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`,
|
|
28395
|
+
* `claude-opus-5` and the 4.7/4.8 Opus lines, and Claude Code's effort picker
|
|
28396
|
+
* offers it for any model whose entry allows it. Listing it lets an explicit
|
|
28397
|
+
* selection pass through, and lets `clampEffort` land on it for a model that
|
|
28398
|
+
* advertises nothing lower.
|
|
28399
|
+
*
|
|
28400
|
+
* `bucketEffort` deliberately reaches neither end — see below.
|
|
28401
|
+
*/
|
|
28402
|
+
const EFFORT_ORDER = [
|
|
28403
|
+
"none",
|
|
28404
|
+
"low",
|
|
28405
|
+
"medium",
|
|
28406
|
+
"high",
|
|
28407
|
+
"xhigh",
|
|
28408
|
+
"max"
|
|
28409
|
+
];
|
|
28410
|
+
/** Anchor for an effort value that is not a recognized tier at all.
|
|
28411
|
+
*
|
|
28412
|
+
* Deliberately NOT the top of `EFFORT_ORDER`. Callers reach this only when the
|
|
28413
|
+
* incoming value is unrecognized, which is a guess — and resolving a guess to
|
|
28414
|
+
* the most expensive tier a model advertises would silently spend more than the
|
|
28415
|
+
* caller could have meant. Anchoring here and clamping DOWN keeps the behavior
|
|
28416
|
+
* identical to before `max` joined the ladder, while `max` stays reachable by
|
|
28417
|
+
* explicit, valid selection. */
|
|
28418
|
+
const UNKNOWN_EFFORT_ANCHOR = "xhigh";
|
|
28419
|
+
/**
|
|
28420
|
+
* Bucket a thinking budget into a Copilot reasoning-effort string.
|
|
28421
|
+
* `<2000`→low, `<8000`→medium, `<24000`→high, else→xhigh.
|
|
28422
|
+
* Defaults missing/non-numeric budgets to 8000 ("high").
|
|
28423
|
+
*
|
|
28424
|
+
* The ceiling stays at `xhigh` even though `max` exists: Anthropic's
|
|
28425
|
+
* `budget_tokens` is unbounded above, so any threshold chosen for a `max`
|
|
28426
|
+
* bucket would silently re-tier existing callers whose budgets already map to
|
|
28427
|
+
* `xhigh`. `max` is reachable only by explicit selection
|
|
28428
|
+
* (`output_config.effort`), which is an unambiguous request rather than an
|
|
28429
|
+
* inference from a token count.
|
|
28430
|
+
*/
|
|
28431
|
+
function bucketEffort(budget) {
|
|
28432
|
+
const n = typeof budget === "number" && Number.isFinite(budget) ? budget : 8e3;
|
|
28433
|
+
if (n < 2e3) return "low";
|
|
28434
|
+
if (n < 8e3) return "medium";
|
|
28435
|
+
if (n < 24e3) return "high";
|
|
28436
|
+
return "xhigh";
|
|
28437
|
+
}
|
|
28438
|
+
/**
|
|
28439
|
+
* Clamp a bucketed effort to the closest value in `supported`. Ties resolve to
|
|
28440
|
+
* the lower-tier option (per EFFORT_ORDER).
|
|
28441
|
+
*
|
|
28442
|
+
* Iterates EFFORT_ORDER (canonical low→xhigh) so the first match on a given
|
|
28443
|
+
* distance is always the lower-tier value, regardless of input order in
|
|
28444
|
+
* `supported`.
|
|
28445
|
+
*/
|
|
28446
|
+
function clampEffort(bucketed, supported) {
|
|
28447
|
+
if (supported.includes(bucketed)) return bucketed;
|
|
28448
|
+
const targetIdx = EFFORT_ORDER.indexOf(bucketed);
|
|
28449
|
+
let best;
|
|
28450
|
+
let bestDist = Infinity;
|
|
28451
|
+
for (let i = 0; i < EFFORT_ORDER.length; i++) {
|
|
28452
|
+
const value = EFFORT_ORDER[i];
|
|
28453
|
+
if (!supported.includes(value)) continue;
|
|
28454
|
+
const dist = Math.abs(i - targetIdx);
|
|
28455
|
+
if (dist < bestDist) {
|
|
28456
|
+
bestDist = dist;
|
|
28457
|
+
best = value;
|
|
28458
|
+
}
|
|
28459
|
+
}
|
|
28460
|
+
return best ?? bucketed;
|
|
28461
|
+
}
|
|
28462
|
+
//#endregion
|
|
28030
28463
|
//#region src/lib/stream-relay.ts
|
|
28031
28464
|
const ENCODER$1 = new TextEncoder();
|
|
28032
28465
|
/**
|
|
@@ -28514,10 +28947,15 @@ function rememberThinkingHistoryRepair(fingerprint) {
|
|
|
28514
28947
|
* re-call Copilot for the next turn — stream onto the SAME
|
|
28515
28948
|
* SSE connection (no new message_start; the original one is
|
|
28516
28949
|
* still open). Loop up to ADVISOR_MAX_TURNS times.
|
|
28517
|
-
* 4.
|
|
28518
|
-
* family than the main loop (gpt-5.6-sol
|
|
28519
|
-
*
|
|
28520
|
-
*
|
|
28950
|
+
* 4. Lead-aware model choice: route the advisor call to a different model
|
|
28951
|
+
* family than the main loop (gpt-5.6-sol) so the user gets a true "second
|
|
28952
|
+
* set of eyes" instead of Opus reviewing Opus (gemini-critic finding). When
|
|
28953
|
+
* the LEAD is a lighter Claude tier the choice inverts and the advisor
|
|
28954
|
+
* escalates to `ADVISOR_ESCALATION_MODEL` instead — see that constant for
|
|
28955
|
+
* why trading the cross-lab property is the right call on that path.
|
|
28956
|
+
* 5. Effort follows the Claude Code effort picker (`resolveAdvisorEffort`)
|
|
28957
|
+
* rather than a hardcoded constant, floored so a low picker cannot render
|
|
28958
|
+
* the consultation useless.
|
|
28521
28959
|
*
|
|
28522
28960
|
* The translate-loop is bounded to a single user request — no
|
|
28523
28961
|
* persistent state across requests is needed (unlike Phase G's
|
|
@@ -28541,6 +28979,185 @@ const ADVISOR_CLIENT_TOOL_NAME = "advisor";
|
|
|
28541
28979
|
* model — Opus 4.6/Sonnet 4.6 typically). */
|
|
28542
28980
|
const ADVISOR_DEFAULT_MODEL = "gpt-5.6-sol";
|
|
28543
28981
|
const ADVISOR_DEFAULT_EFFORT = "xhigh";
|
|
28982
|
+
/** The Anthropic frontier model the advisor escalates to when the LEAD is a
|
|
28983
|
+
* lighter Claude tier (sonnet, haiku).
|
|
28984
|
+
*
|
|
28985
|
+
* Selecting a lighter lead is a decision to work on a budget while holding
|
|
28986
|
+
* quality: the lead does the legwork and escalates for direction. Without this,
|
|
28987
|
+
* a budget lead has no transcript-aware path to the strongest Anthropic
|
|
28988
|
+
* reasoner at all — `opus_critic` is stateless and sees one artifact, and the
|
|
28989
|
+
* `plan` worker is read-only and never sees the transcript.
|
|
28990
|
+
*
|
|
28991
|
+
* This deliberately trades the advisor's cross-lab property on that path. The
|
|
28992
|
+
* advisor is not this repo's review instrument: it catches drift and momentum
|
|
28993
|
+
* and inherits the lead's framing by design, while the fresh-context critics
|
|
28994
|
+
* (`codex_critic`, `gemini_critic`, `codex_reviewer`, `gemini_reviewer`) are
|
|
28995
|
+
* the decorrelation instrument and are untouched. `GH_ROUTER_ADVISOR_MODEL`
|
|
28996
|
+
* keeps a cross-lab advisor one env var away for anyone who wants it back. */
|
|
28997
|
+
const ADVISOR_ESCALATION_MODEL = "claude-opus-5";
|
|
28998
|
+
/** Floor for the advisor's reasoning effort.
|
|
28999
|
+
*
|
|
29000
|
+
* The advisor follows the Claude Code effort picker (see `resolveAdvisorEffort`)
|
|
29001
|
+
* so dialing the picker down makes it cheaper, but it does NOT follow it all the
|
|
29002
|
+
* way to the bottom of `EFFORT_ORDER`. The advisor fires a handful of times per
|
|
29003
|
+
* session (`ADVISOR_MAX_TURNS`, typically 1-3), so it is not the budget line —
|
|
29004
|
+
* the lead's own turns are — while an advisor reasoning at `none`/`low` cannot
|
|
29005
|
+
* do the job the consultation exists for. The picker therefore governs the
|
|
29006
|
+
* `high..max` range. */
|
|
29007
|
+
const ADVISOR_MIN_EFFORT = "high";
|
|
29008
|
+
/** Output cap for the Anthropic-branch advisor call when the catalog carries no
|
|
29009
|
+
* limits for the resolved model. The value the branch used unconditionally
|
|
29010
|
+
* before it became reachable, kept so a catalog-less path is no worse off. */
|
|
29011
|
+
const ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS = 4096;
|
|
29012
|
+
/** Catalog spellings that mean the Responses API. Copilot is inconsistent about
|
|
29013
|
+
* the `/v1` prefix, so both are matched — mirroring `CHAT_ENDPOINTS` /
|
|
29014
|
+
* `RESPONSES_ENDPOINTS` in `src/services/copilot/endpoint.ts`. */
|
|
29015
|
+
const ADVISOR_RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
29016
|
+
/**
|
|
29017
|
+
* Which transport the advisor dispatches on: `/responses` (with
|
|
29018
|
+
* `reasoning.effort`) or `/v1/messages`.
|
|
29019
|
+
*
|
|
29020
|
+
* Catalog-first, name-regex second, and BOTH tests run against the bare id as
|
|
29021
|
+
* well as the given one. `pickEndpoint` is deliberately not reused: it answers
|
|
29022
|
+
* "chat or responses" for the two tool-calling clients and would send
|
|
29023
|
+
* `claude-opus-5` — which advertises `/v1/messages` AND `/chat/completions` — to
|
|
29024
|
+
* chat. The advisor's question is narrower: does this model serve `/responses`?
|
|
29025
|
+
*
|
|
29026
|
+
* The bare-id fallback is what makes `GH_ROUTER_ADVISOR_MODEL` safe. That pin is
|
|
29027
|
+
* accepted verbatim, so an operator can write a vendor-namespaced value like
|
|
29028
|
+
* `openai/gpt-5.6-sol`. Such an id is in no catalog and fails the start-anchored
|
|
29029
|
+
* name regex, so a catalog-only fix still posted it to `/v1/messages` and 400'd
|
|
29030
|
+
* — exported and directly tested for that exact input, because an earlier
|
|
29031
|
+
* version of this function claimed to handle it and did not.
|
|
29032
|
+
*/
|
|
29033
|
+
function advisorUsesResponses(resolvedAdvisorModel) {
|
|
29034
|
+
const bare = resolvedAdvisorModel.slice(resolvedAdvisorModel.lastIndexOf("/") + 1);
|
|
29035
|
+
const endpoints = (state.models?.data?.find((m) => m.id === resolvedAdvisorModel || m.id === bare))?.supported_endpoints;
|
|
29036
|
+
if (endpoints && endpoints.length > 0) return endpoints.some((e) => ADVISOR_RESPONSES_ENDPOINTS.has(e));
|
|
29037
|
+
return /^(gpt-|o\d|.*codex)/i.test(bare);
|
|
29038
|
+
}
|
|
29039
|
+
/** True when the model advertises a usable reasoning-effort ladder. */
|
|
29040
|
+
function advertisedEffortLadder(resolvedAdvisorModel) {
|
|
29041
|
+
const supported = state.models?.data?.find((m) => m.id === resolvedAdvisorModel)?.capabilities?.supports?.reasoning_effort;
|
|
29042
|
+
return Array.isArray(supported) && supported.length > 0 ? supported : void 0;
|
|
29043
|
+
}
|
|
29044
|
+
/** True when the advisor should escalate to `ADVISOR_ESCALATION_MODEL` for this
|
|
29045
|
+
* lead: a Claude lead that is NOT already an Opus tier, on a catalog that
|
|
29046
|
+
* actually carries the escalation model.
|
|
29047
|
+
*
|
|
29048
|
+
* The catalog probe mirrors `standInToolEnabled`'s: never name a model the
|
|
29049
|
+
* account cannot reach. A non-Claude lead never gets here in practice (the
|
|
29050
|
+
* advisor tool is stripped for those before the request reaches this module),
|
|
29051
|
+
* but the check is explicit rather than assumed.
|
|
29052
|
+
*
|
|
29053
|
+
* The probe compares the BARE constant rather than `resolveModel`-ing it first,
|
|
29054
|
+
* which is deliberate and not an oversight: `claude-opus-5` is a single-segment
|
|
29055
|
+
* slug whose dashed and dotted spellings are identical, so resolution is a
|
|
29056
|
+
* no-op, and `resolveModel` WARNS on an id it cannot find — routing this probe
|
|
29057
|
+
* through it would emit that warning on every advisor request for anyone whose
|
|
29058
|
+
* catalog lacks opus-5, which is exactly the tier this returns false for.
|
|
29059
|
+
* `standInToolEnabled` compares the same id the same way. */
|
|
29060
|
+
function shouldEscalateAdvisor(leadModel) {
|
|
29061
|
+
if (!isBudgetClaudeLead(leadModel)) return false;
|
|
29062
|
+
return state.models?.data?.some((m) => m.id === "claude-opus-5") ?? false;
|
|
29063
|
+
}
|
|
29064
|
+
/**
|
|
29065
|
+
* Pick the advisor model for one request from the LEAD model that request is
|
|
29066
|
+
* running on.
|
|
29067
|
+
*
|
|
29068
|
+
* Resolved per request rather than at launch because the lead changes
|
|
29069
|
+
* mid-session via the `/model` picker; launch-time env plumbing would pin the
|
|
29070
|
+
* advisor to whatever was selected at spawn.
|
|
29071
|
+
*
|
|
29072
|
+
* Precedence:
|
|
29073
|
+
* 1. `GH_ROUTER_ADVISOR_MODEL` (trimmed) — the operator pin, checked first so
|
|
29074
|
+
* it works on every lead.
|
|
29075
|
+
* 2. A lighter Claude lead with the escalation model in the catalog.
|
|
29076
|
+
* 3. `ADVISOR_DEFAULT_MODEL`.
|
|
29077
|
+
*
|
|
29078
|
+
* Step 3 returns the LITERAL constant rather than walking the OpenAI frontier
|
|
29079
|
+
* chain. An Opus lead must resolve to exactly what it resolves to today, and a
|
|
29080
|
+
* frontier walk could yield `gpt-5.5` on a catalog missing `gpt-5.6-sol` —
|
|
29081
|
+
* a silent change to the one path that is required not to move.
|
|
29082
|
+
*/
|
|
29083
|
+
/**
|
|
29084
|
+
* Map an operator pin onto the id the catalog actually carries.
|
|
29085
|
+
*
|
|
29086
|
+
* `GH_ROUTER_ADVISOR_MODEL` is free-form, and the natural thing to write is a
|
|
29087
|
+
* vendor-namespaced id like `openai/gpt-5.6-sol`. Copilot's catalog carries the
|
|
29088
|
+
* bare `gpt-5.6-sol`, so forwarding the namespaced form verbatim gets a 400
|
|
29089
|
+
* `model_not_supported` and the advisor silently degrades to its
|
|
29090
|
+
* "[Advisor unavailable: ...]" fallback — measured, not theorised: choosing the
|
|
29091
|
+
* transport correctly was NOT sufficient, because the id itself was still
|
|
29092
|
+
* wrong on the wire.
|
|
29093
|
+
*
|
|
29094
|
+
* An exact catalog hit wins first, so a real id containing a slash could never
|
|
29095
|
+
* be mangled. Only when the pin is absent from the catalog do we try its last
|
|
29096
|
+
* path segment, and only when THAT is present do we rewrite. A pin that matches
|
|
29097
|
+
* nothing is passed through untouched: the catalog may simply not be loaded
|
|
29098
|
+
* yet, and inventing an id would be worse than letting upstream reject it.
|
|
29099
|
+
*/
|
|
29100
|
+
function normalizeAdvisorPin(pinned) {
|
|
29101
|
+
const models = state.models?.data;
|
|
29102
|
+
if (!models) return pinned;
|
|
29103
|
+
if (models.some((m) => m.id === pinned)) return pinned;
|
|
29104
|
+
const bare = pinned.slice(pinned.lastIndexOf("/") + 1);
|
|
29105
|
+
return bare !== pinned && models.some((m) => m.id === bare) ? bare : pinned;
|
|
29106
|
+
}
|
|
29107
|
+
function resolveAdvisorModel(leadModel) {
|
|
29108
|
+
const pinned = process.env.GH_ROUTER_ADVISOR_MODEL?.trim();
|
|
29109
|
+
if (pinned) return {
|
|
29110
|
+
model: normalizeAdvisorPin(pinned),
|
|
29111
|
+
escalated: false
|
|
29112
|
+
};
|
|
29113
|
+
if (leadModel && shouldEscalateAdvisor(leadModel)) return {
|
|
29114
|
+
model: ADVISOR_ESCALATION_MODEL,
|
|
29115
|
+
escalated: true
|
|
29116
|
+
};
|
|
29117
|
+
return {
|
|
29118
|
+
model: ADVISOR_DEFAULT_MODEL,
|
|
29119
|
+
escalated: false
|
|
29120
|
+
};
|
|
29121
|
+
}
|
|
29122
|
+
/**
|
|
29123
|
+
* Resolve the advisor's reasoning effort from the ORIGINAL request body, so the
|
|
29124
|
+
* advisor thinks at the level selected in the Claude Code effort picker instead
|
|
29125
|
+
* of a hardcoded constant.
|
|
29126
|
+
*
|
|
29127
|
+
* The source is the RAW pre-`resolveModelInBody` body, deliberately. By the time
|
|
29128
|
+
* the handler holds a parsed body, `translateThinking` has already bucketed
|
|
29129
|
+
* `thinking.budget_tokens` into `output_config.effort` AND clamped it to the
|
|
29130
|
+
* LEAD model's allowlist — so that value encodes "what the lead could do", not
|
|
29131
|
+
* "what the user picked". Re-clamping it against the advisor cannot recover the
|
|
29132
|
+
* difference: a `max` pick on a lead whose ceiling is `high` would reach an
|
|
29133
|
+
* xhigh-capable advisor as `high`.
|
|
29134
|
+
*
|
|
29135
|
+
* Precedence mirrors the repo-wide rule that an explicit client effort wins:
|
|
29136
|
+
* 1. `output_config.effort`
|
|
29137
|
+
* 2. `bucketEffort(thinking.budget_tokens)`
|
|
29138
|
+
* 3. `ADVISOR_DEFAULT_EFFORT` — a request expressing no preference behaves
|
|
29139
|
+
* exactly as it did before the picker was honored at all.
|
|
29140
|
+
*
|
|
29141
|
+
* Then floor, THEN clamp. The order is load-bearing: a model whose ceiling sits
|
|
29142
|
+
* below `ADVISOR_MIN_EFFORT` must still receive a value it accepts, so the clamp
|
|
29143
|
+
* is allowed to pull back under the floor. Flipping the two would forward an
|
|
29144
|
+
* effort upstream rejects.
|
|
29145
|
+
*/
|
|
29146
|
+
function resolveAdvisorEffort(rawRequestBody, advisorModel) {
|
|
29147
|
+
let requested = ADVISOR_DEFAULT_EFFORT;
|
|
29148
|
+
if (rawRequestBody) try {
|
|
29149
|
+
const body = JSON.parse(rawRequestBody);
|
|
29150
|
+
const oc = body.output_config;
|
|
29151
|
+
const explicit = oc && typeof oc === "object" ? oc.effort : void 0;
|
|
29152
|
+
const thinking = body.thinking;
|
|
29153
|
+
if (typeof explicit === "string" && EFFORT_ORDER.includes(explicit)) requested = explicit;
|
|
29154
|
+
else if (thinking && typeof thinking === "object" && thinking.type === "enabled") requested = bucketEffort(thinking.budget_tokens);
|
|
29155
|
+
} catch {}
|
|
29156
|
+
const floored = EFFORT_ORDER.indexOf(requested) < EFFORT_ORDER.indexOf(ADVISOR_MIN_EFFORT) ? ADVISOR_MIN_EFFORT : requested;
|
|
29157
|
+
const supported = state.models?.data?.find((m) => m.id === resolveModel(advisorModel))?.capabilities?.supports?.reasoning_effort;
|
|
29158
|
+
if (!Array.isArray(supported) || supported.length === 0) return floored;
|
|
29159
|
+
return clampEffort(floored, supported);
|
|
29160
|
+
}
|
|
28544
29161
|
/** ADVISOR_TOOL_INSTRUCTIONS verbatim from cc-backup
|
|
28545
29162
|
* src/utils/advisor.ts — describes when the model should invoke
|
|
28546
29163
|
* the advisor. Long-form prose; see source for justification. */
|
|
@@ -28636,8 +29253,12 @@ const ADVISOR_FALLBACK_MAX_TOKENS = 24e4;
|
|
|
28636
29253
|
* our o200k count and Copilot's full-payload count. The transcript token
|
|
28637
29254
|
* budget is `max_prompt_tokens - reserve`. Generous on purpose: a 400
|
|
28638
29255
|
* `model_max_prompt_tokens_exceeded` degrades to a silent advisor
|
|
28639
|
-
* fallback, and the
|
|
28640
|
-
* gpt-5.6-sol
|
|
29256
|
+
* fallback, and the window given up is marginal against either advisor
|
|
29257
|
+
* model's real prompt window (`claude-opus-5` 936k, `gpt-5.6-sol` ~1M off
|
|
29258
|
+
* the live catalog). Sized as a fraction of the smaller of the two, not as
|
|
29259
|
+
* "irrelevant next to ~1M" — that framing assumed the advisor was always
|
|
29260
|
+
* the cheap side of the pair, which stopped being true once a budget lead
|
|
29261
|
+
* escalates to Opus. */
|
|
28641
29262
|
const ADVISOR_PROMPT_TOKEN_RESERVE = 8e3;
|
|
28642
29263
|
/**
|
|
28643
29264
|
* Derive the TOKEN budget for the rendered transcript from the advisor
|
|
@@ -28748,9 +29369,9 @@ function truncateTailToUnits(text, maxUnits, measure) {
|
|
|
28748
29369
|
* Anthropic's own ADVISOR ("see the whole task + every tool call +
|
|
28749
29370
|
* every result").
|
|
28750
29371
|
*/
|
|
28751
|
-
async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
29372
|
+
async function runAdvisor(conversation, advisorModel, advisorEffort, signal, advisorEscalated = false) {
|
|
28752
29373
|
if (signal?.aborted) throw new Error("advisor call aborted before dispatch");
|
|
28753
|
-
const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance.";
|
|
29374
|
+
const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance." + (advisorEscalated ? " The requesting agent is running a lighter, faster model than you. Give a directive recommendation and commit to the decision rather than laying out options for it to weigh." : "");
|
|
28754
29375
|
const resolvedAdvisorModel = resolveModel(advisorModel);
|
|
28755
29376
|
let measure;
|
|
28756
29377
|
let maxUnits;
|
|
@@ -28765,7 +29386,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
|
28765
29386
|
maxUnits = ADVISOR_MAX_CONVERSATION_CHARS;
|
|
28766
29387
|
}
|
|
28767
29388
|
const conversationText = renderConversationAsText(conversation, maxUnits, measure);
|
|
28768
|
-
if (
|
|
29389
|
+
if (advisorUsesResponses(resolvedAdvisorModel)) {
|
|
28769
29390
|
const payload = {
|
|
28770
29391
|
model: resolvedAdvisorModel,
|
|
28771
29392
|
instructions: advisorSystem,
|
|
@@ -28800,15 +29421,22 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
|
28800
29421
|
if (!text) throw new Error(`Advisor model ${resolvedAdvisorModel} returned empty assistant output`);
|
|
28801
29422
|
return text;
|
|
28802
29423
|
}
|
|
29424
|
+
const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
|
|
29425
|
+
const limits = advisorEntry?.capabilities?.limits;
|
|
29426
|
+
const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
|
|
28803
29427
|
const advisorBody = JSON.stringify({
|
|
28804
29428
|
model: resolvedAdvisorModel,
|
|
28805
|
-
max_tokens:
|
|
29429
|
+
max_tokens: maxTokens,
|
|
28806
29430
|
system: advisorSystem,
|
|
28807
29431
|
messages: [{
|
|
28808
29432
|
role: "user",
|
|
28809
29433
|
content: conversationText
|
|
28810
29434
|
}],
|
|
28811
|
-
stream: false
|
|
29435
|
+
stream: false,
|
|
29436
|
+
...advisorEntry?.capabilities?.supports?.adaptive_thinking ? {
|
|
29437
|
+
thinking: { type: "adaptive" },
|
|
29438
|
+
...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
|
|
29439
|
+
} : {}
|
|
28812
29440
|
});
|
|
28813
29441
|
const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
|
|
28814
29442
|
signal,
|
|
@@ -28863,6 +29491,7 @@ function sseEvent(type, data) {
|
|
|
28863
29491
|
function buildAdvisorStream(opts) {
|
|
28864
29492
|
const advisorModel = opts.advisorModel ?? "gpt-5.6-sol";
|
|
28865
29493
|
const advisorEffort = opts.advisorEffort ?? "xhigh";
|
|
29494
|
+
const advisorEscalated = opts.advisorEscalated ?? false;
|
|
28866
29495
|
const aborter = opts.externalAborter ?? new AbortController();
|
|
28867
29496
|
let conversation = [...opts.initialConversation];
|
|
28868
29497
|
return new ReadableStream({
|
|
@@ -29112,7 +29741,7 @@ function buildAdvisorStream(opts) {
|
|
|
29112
29741
|
const advisorConversation = conversation;
|
|
29113
29742
|
const advisorTexts = await Promise.all(advisorToolUses.map(async () => {
|
|
29114
29743
|
try {
|
|
29115
|
-
return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal);
|
|
29744
|
+
return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal, advisorEscalated);
|
|
29116
29745
|
} catch (err) {
|
|
29117
29746
|
if (aborter.signal.aborted) throw err;
|
|
29118
29747
|
const msg = err instanceof Error ? err.message : String(err);
|
|
@@ -32376,7 +33005,7 @@ function appendPlanReminder(messages, planState) {
|
|
|
32376
33005
|
* gemini-3.1-pro-preview is pinned to `high` because the model rejects
|
|
32377
33006
|
* `xhigh` at the wire with a Copilot 400. `high` is the realistic ceiling.
|
|
32378
33007
|
*/
|
|
32379
|
-
const
|
|
33008
|
+
const STAND_IN_MODELS_BASE = Object.freeze([
|
|
32380
33009
|
{
|
|
32381
33010
|
key: "gpt-5.6-sol",
|
|
32382
33011
|
model: "gpt-5.6-sol",
|
|
@@ -32396,6 +33025,13 @@ const STAND_IN_MODELS = Object.freeze([
|
|
|
32396
33025
|
effort: "high"
|
|
32397
33026
|
}
|
|
32398
33027
|
]);
|
|
33028
|
+
function standInModels() {
|
|
33029
|
+
const geminiModel = resolveGeminiReviewModel();
|
|
33030
|
+
return STAND_IN_MODELS_BASE.map((config) => config.key === "gemini-3.1-pro-preview" ? {
|
|
33031
|
+
...config,
|
|
33032
|
+
model: geminiModel ?? "gemini-3.7-flash"
|
|
33033
|
+
} : config);
|
|
33034
|
+
}
|
|
32399
33035
|
const SYSTEM_PROMPT_R1 = `You are one of three frontier reasoning models the user has authorized to stand in for them on a bounded decision while they are unavailable. Your task: pick the best option from those provided.
|
|
32400
33036
|
|
|
32401
33037
|
Respond with ONLY a single JSON object — no prose, no markdown fences, no preamble. Schema:
|
|
@@ -32447,7 +33083,7 @@ const RETRY_PROMPT_SUFFIX = `\n\nYour previous response was not valid JSON match
|
|
|
32447
33083
|
async function runStandIn(input, signal) {
|
|
32448
33084
|
const validIds = new Set(input.options.map((o) => o.id));
|
|
32449
33085
|
const r1UserText = buildRound1UserText(input);
|
|
32450
|
-
const r1 = await Promise.all(
|
|
33086
|
+
const r1 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R1, r1UserText, validIds, signal)));
|
|
32451
33087
|
const successfulR1 = r1.filter((r) => isVote(r.vote));
|
|
32452
33088
|
const nmiR1 = gapAbstainVerdict(successfulR1, r1, null);
|
|
32453
33089
|
if (nmiR1) return nmiR1;
|
|
@@ -32467,7 +33103,7 @@ async function runStandIn(input, signal) {
|
|
|
32467
33103
|
notes: `Only ${successfulR1.length} of 3 models returned a parseable round-1 vote; insufficient signal to run round 2.`
|
|
32468
33104
|
}, r1, null);
|
|
32469
33105
|
const r2UserTextBase = buildRound2UserTextBase(input, r1);
|
|
32470
|
-
const r2 = await Promise.all(
|
|
33106
|
+
const r2 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R2, r2UserTextBase + `\n\nYou are ${cfg.key}. Reconsider and vote.`, validIds, signal)));
|
|
32471
33107
|
const successfulR2 = r2.filter((r) => isVote(r.vote));
|
|
32472
33108
|
if (successfulR2.length < 2) return withDerivedNotes({
|
|
32473
33109
|
verdict: "no_consensus",
|
|
@@ -32644,7 +33280,7 @@ function aggregateVotes(results) {
|
|
|
32644
33280
|
topCount = count;
|
|
32645
33281
|
topSumConfidence = sumConfidence;
|
|
32646
33282
|
}
|
|
32647
|
-
const total =
|
|
33283
|
+
const total = standInModels().length;
|
|
32648
33284
|
if (topChoice && topCount === total) return {
|
|
32649
33285
|
verdict: "consensus",
|
|
32650
33286
|
winner: topChoice,
|
|
@@ -32695,7 +33331,7 @@ function gapAbstainVerdict(successful, r1, r2) {
|
|
|
32695
33331
|
const gapVotes = successful.filter((r) => r.vote.choice === null && r.vote.needMoreInfo);
|
|
32696
33332
|
if (gapVotes.length < 2) return null;
|
|
32697
33333
|
const gaps = gapVotes.map((r) => `- ${r.key}: ${r.vote.needMoreInfo}`).join("\n");
|
|
32698
|
-
const header = gapVotes.length ===
|
|
33334
|
+
const header = gapVotes.length === standInModels().length ? "All three models reported they need more context to decide:" : `${gapVotes.length} of 3 models reported they need more context to decide:`;
|
|
32699
33335
|
return withDerivedNotes({
|
|
32700
33336
|
verdict: "need_more_info",
|
|
32701
33337
|
recommendation: null,
|
|
@@ -32711,7 +33347,7 @@ function gapAbstainVerdict(successful, r1, r2) {
|
|
|
32711
33347
|
*/
|
|
32712
33348
|
function freshestVotes(r1, r2) {
|
|
32713
33349
|
const out = [];
|
|
32714
|
-
for (const cfg of
|
|
33350
|
+
for (const cfg of standInModels()) {
|
|
32715
33351
|
const r2Entry = r2?.find((r) => r.key === cfg.key);
|
|
32716
33352
|
const r1Entry = r1.find((r) => r.key === cfg.key);
|
|
32717
33353
|
const vote = r2Entry && isVote(r2Entry.vote) ? r2Entry.vote : r1Entry && isVote(r1Entry.vote) ? r1Entry.vote : null;
|
|
@@ -32749,7 +33385,7 @@ function withDerivedNotes(result, r1, r2) {
|
|
|
32749
33385
|
}
|
|
32750
33386
|
function voteRecord(r1, r2) {
|
|
32751
33387
|
const record = {};
|
|
32752
|
-
for (const cfg of
|
|
33388
|
+
for (const cfg of standInModels()) {
|
|
32753
33389
|
const r1Entry = r1.find((r) => r.key === cfg.key);
|
|
32754
33390
|
const r2Entry = r2?.find((r) => r.key === cfg.key) ?? null;
|
|
32755
33391
|
record[cfg.key] = {
|
|
@@ -33978,7 +34614,7 @@ const PERSONAS_READ = Object.freeze([
|
|
|
33978
34614
|
{
|
|
33979
34615
|
agentName: "gemini-critic",
|
|
33980
34616
|
toolNameHttp: "gemini_critic",
|
|
33981
|
-
model:
|
|
34617
|
+
model: GEMINI_REVIEW_DEFAULT_MODEL,
|
|
33982
34618
|
endpoint: "/v1/chat/completions",
|
|
33983
34619
|
description: "Adversarial third-lab critic backed by gemini-3.1-pro-preview (Google), strong on formal reasoning, invariants, proofs, and cross-checking another critic's conclusion. It reviews plans, designs, mathematical arguments, and large artifacts for assumption gaps or invariant failures, then returns a focused critique or no-material-objection style verdict. Use when codex_critic's result needs an independent lab check or when the artifact hinges on formal correctness. Not for line-level diff review, use gemini_reviewer or codex_reviewer; pass the artifact and constraints verbatim.",
|
|
33984
34620
|
baseInstructions: GEMINI_CRITIC_BASE,
|
|
@@ -34014,7 +34650,7 @@ const PERSONAS_READ = Object.freeze([
|
|
|
34014
34650
|
{
|
|
34015
34651
|
agentName: "gemini-reviewer",
|
|
34016
34652
|
toolNameHttp: "gemini_reviewer",
|
|
34017
|
-
model:
|
|
34653
|
+
model: GEMINI_REVIEW_DEFAULT_MODEL,
|
|
34018
34654
|
endpoint: "/v1/chat/completions",
|
|
34019
34655
|
description: "Line-level code reviewer backed by gemini-3.1-pro-preview (Google), providing second-lab coverage that catches a different slice of concrete-code defects than codex_reviewer. It reviews diffs, files, or function bodies and returns severity-ranked findings with file:line citations and suggested fixes. Use alongside codex_reviewer when a non-trivial diff benefits from cross-lab code-review coverage, especially around invariants or edge cases. Not for architecture or product-design review, use codex_critic or gemini_critic; pass the code artifact verbatim.",
|
|
34020
34656
|
baseInstructions: GEMINI_REVIEWER_BASE,
|
|
@@ -34183,15 +34819,16 @@ function buildPeerAwarenessSnippet(opts) {
|
|
|
34183
34819
|
const powerBrowseAvailable = opts.browseAvailable && opts.powerBrowseAvailable === true;
|
|
34184
34820
|
const criticList = ["`codex_critic` (gpt-5.6-sol)", "`codex_reviewer` (gpt-5.3-codex)"];
|
|
34185
34821
|
if (opts.geminiAvailable) {
|
|
34186
|
-
|
|
34187
|
-
criticList.push(
|
|
34822
|
+
const geminiModel = opts.geminiModel ?? "gemini-3.1-pro-preview";
|
|
34823
|
+
criticList.push(`\`gemini_reviewer\` (${geminiModel}, line-level code review)`);
|
|
34824
|
+
criticList.push(`\`gemini_critic\` (${geminiModel})`);
|
|
34188
34825
|
}
|
|
34189
34826
|
criticList.push("`opus_critic` (Opus 5)");
|
|
34190
34827
|
const codexCliClause = opts.codexCli ? " `mcp__codex-cli__codex` dispatches to `codex-implementer` (gpt-5.3-codex with workspace-write) for end-to-end coding tasks." : "";
|
|
34191
34828
|
const para2Parts = [`\`mcp__${searchKey}__code\` is the one-stop code search (no extra model call). Its DEFAULT mode (or \`mode:"semantic"\`) ranks by MEANING via ColBERT over a per-workspace index, the first thing to reach for on intent/concept questions ("where is retry/backoff handled", "how does auth work"); when that index isn't ready it transparently falls back to lexical (the response \`source\` says which engine ran). Forced modes cover the rest: \`lexical\` (BM25F-ranked + tree-sitter, best for exact symbols), \`exact\`, \`regex\`, \`complete\` (exhaustive set), \`ast_pattern\`+\`ast_lang\` for multi-line AST shapes, \`scan\` for a whole-workspace symbol outline, \`multiline\` for cross-line regex. Multiple queries can run in a single turn. The index covers code-shaped files; for unstructured files (logs, \`.csv\`, \`.env*\`, config-only wiring), \`grep\`/\`glob\` still apply.`];
|
|
34192
34829
|
if (opts.workerToolsAvailable) para2Parts.push(`\`worker-*\` are background Agent subagents (subagent_type) that run the matching worker in its own context and deliver the result as a completion notification, so a long run never blocks the turn: \`worker-explore\` (read-only research), \`worker-review\` (reads the code to verify a change or claim), \`worker-plan\` (ordered implementation plan), \`worker-implement\` (edit/write/bash; ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file; for in-place edits use the \`implementer\` subagent), \`worker-test\` (independent test author; also always worktree-isolated)${opts.browseAvailable ? ", `worker-browse` (autonomous browser agent driving a real browser)" : ""}. The raw \`mcp__${workersKey}__*\` tools they call are guarded (a direct main-thread call is redirected to the matching agent); Workers themselves have \`code_search\`.`);
|
|
34193
34830
|
const catchAllClause = opts.generalPurposeFastAvailable === false ? "" : " Catch-all on a fast, economical non-lead model for work no specialist fits: `general-purpose-fast`.";
|
|
34194
|
-
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
34831
|
+
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure)${opts.reviewerFastAvailable === false ? "" : ", `reviewer-fast` (lower-stakes assessment on a cheaper cross-lab model)"}, \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
34195
34832
|
if (opts.workerToolsAvailable) para2Parts.push(`For a bounded, well-scoped implementation, prefer the \`implementer\` subagent over \`worker-implement\`; reach for \`worker-implement\` only when you specifically need git-worktree isolation, parallel variants, or a throwaway experiment.`);
|
|
34196
34833
|
if (opts.workerToolsAvailable) para2Parts.push(`\`mcp__${orchestrateKey}__decompose\` composes an open-ended ask into a typed, VERIFIED workflow IR (a strong driver decorrelated by a cross-lab critic, so the decompose step isn't a single point of failure), and \`mcp__${orchestrateKey}__run_workflow\` executes that IR through a frozen kernel delivering max(orchestrated, baseline) over a sealed executable gate, so it never ships worse than a plain single-model run. \`mcp__${orchestrateKey}__verify_workflow\` checks an IR's floor invariants before you run it, and \`mcp__${orchestrateKey}__attest_step\` audits that a finished run's producers were each checked by a different lab. They suit non-trivial, role-separated asks; a trivial ask does not need them.`);
|
|
34197
34834
|
else para2Parts.push(`\`mcp__${orchestrateKey}__verify_workflow\` statically checks a workflow IR's floor invariants and \`mcp__${orchestrateKey}__attest_step\` audits a run's cross-lab lineage (the \`decompose\`/\`run_workflow\` composer + kernel need the worker backend, unavailable here).`);
|
|
@@ -34236,6 +34873,7 @@ function buildPeerAwarenessSummary(opts) {
|
|
|
34236
34873
|
renderNative("implementer"),
|
|
34237
34874
|
opts.implementerFastAvailable === false ? void 0 : renderNative("implementer-fast"),
|
|
34238
34875
|
renderNative("reviewer"),
|
|
34876
|
+
opts.reviewerFastAvailable === false ? void 0 : renderNative("reviewer-fast"),
|
|
34239
34877
|
renderNative("brainstorm"),
|
|
34240
34878
|
opts.scoutAvailable === false ? void 0 : renderNative("scout"),
|
|
34241
34879
|
renderNative("scribe"),
|
|
@@ -34256,11 +34894,38 @@ function buildPeerAwarenessSummary(opts) {
|
|
|
34256
34894
|
lines.push(`Each tool's own description carries when to use it and when not. The full per-tool inventory (models, gating, workers, skills) is in the "Peer review and advisor" section of your CLAUDE.md project instructions.`);
|
|
34257
34895
|
return lines.join("\n");
|
|
34258
34896
|
}
|
|
34897
|
+
/**
|
|
34898
|
+
* Applies the resolved Gemini review model to a persona requiring the Gemini
|
|
34899
|
+
* catalog: swaps `.model` and rewrites every literal occurrence of the
|
|
34900
|
+
* default id in `.description` so the two never disagree about which model
|
|
34901
|
+
* actually backs the tool. A mismatch here is user-visible — `tools/list`
|
|
34902
|
+
* would advertise "backed by gemini-3.1-pro-preview" while dispatch actually
|
|
34903
|
+
* ran the flash fallback — so every call site that resolves a
|
|
34904
|
+
* `requiresGeminiCatalog` persona MUST go through this helper rather than
|
|
34905
|
+
* setting `.model` directly (a prior draft of this fix did exactly that in
|
|
34906
|
+
* `routes/mcp/handler.ts`'s `activePersonas()` and left the description
|
|
34907
|
+
* stale). Relies on every `requiresGeminiCatalog` persona's description
|
|
34908
|
+
* literally containing `GEMINI_REVIEW_DEFAULT_MODEL`'s exact string — true
|
|
34909
|
+
* for both current entries (gemini-critic, gemini-reviewer); keep it true for
|
|
34910
|
+
* any future one, or `replaceAll` silently no-ops.
|
|
34911
|
+
*/
|
|
34912
|
+
function resolveGeminiPersona(p, geminiModel) {
|
|
34913
|
+
const model = geminiModel ?? "gemini-3.1-pro-preview";
|
|
34914
|
+
return {
|
|
34915
|
+
...p,
|
|
34916
|
+
model,
|
|
34917
|
+
description: p.description.replaceAll(GEMINI_REVIEW_DEFAULT_MODEL, model)
|
|
34918
|
+
};
|
|
34919
|
+
}
|
|
34259
34920
|
/** Convenience: every persona that should be registered for the given mode. */
|
|
34260
34921
|
function personasFor(opts) {
|
|
34261
34922
|
const result = [];
|
|
34262
34923
|
for (const p of PERSONAS_READ) {
|
|
34263
|
-
if (p.requiresGeminiCatalog
|
|
34924
|
+
if (p.requiresGeminiCatalog) {
|
|
34925
|
+
if (!opts.geminiAvailable) continue;
|
|
34926
|
+
result.push(resolveGeminiPersona(p, opts.geminiModel));
|
|
34927
|
+
continue;
|
|
34928
|
+
}
|
|
34264
34929
|
result.push(p);
|
|
34265
34930
|
}
|
|
34266
34931
|
if (opts.codexCli) for (const p of PERSONAS_WRITE) result.push(p);
|
|
@@ -34568,7 +35233,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34568
35233
|
},
|
|
34569
35234
|
workspace: {
|
|
34570
35235
|
type: "string",
|
|
34571
|
-
description: "
|
|
35236
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34572
35237
|
},
|
|
34573
35238
|
maxWallClockMs: {
|
|
34574
35239
|
type: "integer",
|
|
@@ -34576,11 +35241,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34576
35241
|
}
|
|
34577
35242
|
}
|
|
34578
35243
|
},
|
|
34579
|
-
async handler(args, signal) {
|
|
35244
|
+
async handler(args, signal, ctx) {
|
|
34580
35245
|
return runWorkerToolCall({
|
|
34581
35246
|
mode: "explore",
|
|
34582
35247
|
args,
|
|
34583
|
-
signal
|
|
35248
|
+
signal,
|
|
35249
|
+
ctx
|
|
34584
35250
|
});
|
|
34585
35251
|
}
|
|
34586
35252
|
},
|
|
@@ -34613,7 +35279,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34613
35279
|
},
|
|
34614
35280
|
workspace: {
|
|
34615
35281
|
type: "string",
|
|
34616
|
-
description: "
|
|
35282
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (implement always runs in a worktree)."
|
|
34617
35283
|
},
|
|
34618
35284
|
maxWallClockMs: {
|
|
34619
35285
|
type: "integer",
|
|
@@ -34621,11 +35287,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34621
35287
|
}
|
|
34622
35288
|
}
|
|
34623
35289
|
},
|
|
34624
|
-
async handler(args, signal) {
|
|
35290
|
+
async handler(args, signal, ctx) {
|
|
34625
35291
|
return runWorkerToolCall({
|
|
34626
35292
|
mode: "implement",
|
|
34627
35293
|
args,
|
|
34628
|
-
signal
|
|
35294
|
+
signal,
|
|
35295
|
+
ctx
|
|
34629
35296
|
});
|
|
34630
35297
|
}
|
|
34631
35298
|
},
|
|
@@ -34633,7 +35300,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34633
35300
|
toolNameHttp: "review",
|
|
34634
35301
|
group: "workers",
|
|
34635
35302
|
capability: "worker",
|
|
34636
|
-
description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification.
|
|
35303
|
+
description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification. Code review by an autonomous worker (Pi runtime; default model `gemini-3.1-pro-preview`, default thinking xhigh clamped to high for that model, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has explore's read-and-search tools PLUS `bash`, so it verifies claims by running them — reproducing a failure or running the build or suite — rather than only reading, and returns severity-ranked findings with `file:line` citations. It gets no edit/write tools unless you pass `worktree: true`, but `bash` runs real commands in the workspace, so a build or test it invokes can touch the tree. Use for reviewing a change, diff, or correctness claim when the reviewer should read the code itself rather than trusting a pasted artifact. Not for architecture critique, implementation, or test authoring; use codex_critic or gemini_critic for design review, implement for edits, and test for independent test creation. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34637
35304
|
inputSchema: {
|
|
34638
35305
|
type: "object",
|
|
34639
35306
|
required: ["prompt"],
|
|
@@ -34652,9 +35319,13 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34652
35319
|
enum: WORKER_THINKING_LEVELS,
|
|
34653
35320
|
description: "Optional reasoning depth (defaults to xhigh, clamped to high for the default review model). Silently clamped to the model's allowed range; \"off\" drops the parameter entirely."
|
|
34654
35321
|
},
|
|
35322
|
+
worktree: {
|
|
35323
|
+
type: "boolean",
|
|
35324
|
+
description: "Optional. When true, the review runs in an isolated git worktree replaying the workspace's working tree (dirty tracked changes and untracked-not-ignored files), which additionally grants `edit`/`write` so the reviewer can author a throwaway probe test to prove a claim. Default false: the reviewer reads and runs commands in the workspace itself. Prefer the default when verifying needs the build to work — a fresh worktree does not carry IGNORED files, so installed dependencies are absent. Requires a git repository."
|
|
35325
|
+
},
|
|
34655
35326
|
workspace: {
|
|
34656
35327
|
type: "string",
|
|
34657
|
-
description: "
|
|
35328
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34658
35329
|
},
|
|
34659
35330
|
maxWallClockMs: {
|
|
34660
35331
|
type: "integer",
|
|
@@ -34662,11 +35333,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34662
35333
|
}
|
|
34663
35334
|
}
|
|
34664
35335
|
},
|
|
34665
|
-
async handler(args, signal) {
|
|
35336
|
+
async handler(args, signal, ctx) {
|
|
34666
35337
|
return runWorkerToolCall({
|
|
34667
35338
|
mode: "review",
|
|
34668
35339
|
args,
|
|
34669
|
-
signal
|
|
35340
|
+
signal,
|
|
35341
|
+
ctx
|
|
34670
35342
|
});
|
|
34671
35343
|
}
|
|
34672
35344
|
},
|
|
@@ -34695,7 +35367,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34695
35367
|
},
|
|
34696
35368
|
workspace: {
|
|
34697
35369
|
type: "string",
|
|
34698
|
-
description: "
|
|
35370
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34699
35371
|
},
|
|
34700
35372
|
maxWallClockMs: {
|
|
34701
35373
|
type: "integer",
|
|
@@ -34703,11 +35375,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34703
35375
|
}
|
|
34704
35376
|
}
|
|
34705
35377
|
},
|
|
34706
|
-
async handler(args, signal) {
|
|
35378
|
+
async handler(args, signal, ctx) {
|
|
34707
35379
|
return runWorkerToolCall({
|
|
34708
35380
|
mode: "plan",
|
|
34709
35381
|
args,
|
|
34710
|
-
signal
|
|
35382
|
+
signal,
|
|
35383
|
+
ctx
|
|
34711
35384
|
});
|
|
34712
35385
|
}
|
|
34713
35386
|
},
|
|
@@ -34740,7 +35413,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34740
35413
|
},
|
|
34741
35414
|
workspace: {
|
|
34742
35415
|
type: "string",
|
|
34743
|
-
description: "
|
|
35416
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (test always runs in a worktree)."
|
|
34744
35417
|
},
|
|
34745
35418
|
maxWallClockMs: {
|
|
34746
35419
|
type: "integer",
|
|
@@ -34748,11 +35421,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34748
35421
|
}
|
|
34749
35422
|
}
|
|
34750
35423
|
},
|
|
34751
|
-
async handler(args, signal) {
|
|
35424
|
+
async handler(args, signal, ctx) {
|
|
34752
35425
|
return runWorkerToolCall({
|
|
34753
35426
|
mode: "test",
|
|
34754
35427
|
args,
|
|
34755
|
-
signal
|
|
35428
|
+
signal,
|
|
35429
|
+
ctx
|
|
34756
35430
|
});
|
|
34757
35431
|
}
|
|
34758
35432
|
},
|
|
@@ -35075,10 +35749,12 @@ function assertMcpToolSurfaceConsistent() {
|
|
|
35075
35749
|
/**
|
|
35076
35750
|
* Shared closure body for the two worker MCP tools. Validates the
|
|
35077
35751
|
* minimal arg shape (prompt required + optional knobs typed), then
|
|
35078
|
-
* forwards to `runWorkerAgent`.
|
|
35079
|
-
*
|
|
35080
|
-
*
|
|
35081
|
-
*
|
|
35752
|
+
* forwards to `runWorkerAgent`. `workspace` comes from the caller's
|
|
35753
|
+
* argument or, failing that, the per-connection session header the
|
|
35754
|
+
* boundary folds in; with neither, the call is REFUSED rather than
|
|
35755
|
+
* defaulted to the proxy's launch cwd (see the resolution block below
|
|
35756
|
+
* for why that default was a bug, and `GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE`
|
|
35757
|
+
* for the raw-client escape hatch). The engine performs every
|
|
35082
35758
|
* deeper validation (model existence, thinking clamp, worktree
|
|
35083
35759
|
* provisioning, semaphore acquisition, workspace realpath +
|
|
35084
35760
|
* accessibility) and never throws — its `{text, isError?}` envelope
|
|
@@ -35091,7 +35767,7 @@ function assertMcpToolSurfaceConsistent() {
|
|
|
35091
35767
|
* client that ignores the schema.
|
|
35092
35768
|
*/
|
|
35093
35769
|
async function runWorkerToolCall(call) {
|
|
35094
|
-
const { mode, args, signal } = call;
|
|
35770
|
+
const { mode, args, signal, ctx } = call;
|
|
35095
35771
|
const prompt = typeof args.prompt === "string" ? args.prompt : "";
|
|
35096
35772
|
if (!prompt) return {
|
|
35097
35773
|
content: [{
|
|
@@ -35123,7 +35799,7 @@ async function runWorkerToolCall(call) {
|
|
|
35123
35799
|
}
|
|
35124
35800
|
let worktree;
|
|
35125
35801
|
let worktreeNote = "";
|
|
35126
|
-
if (mode === "implement" || mode === "test") {
|
|
35802
|
+
if (mode === "implement" || mode === "test" || mode === "review") {
|
|
35127
35803
|
if (args.worktree !== void 0 && typeof args.worktree !== "boolean") return {
|
|
35128
35804
|
content: [{
|
|
35129
35805
|
type: "text",
|
|
@@ -35131,11 +35807,16 @@ async function runWorkerToolCall(call) {
|
|
|
35131
35807
|
}],
|
|
35132
35808
|
isError: true
|
|
35133
35809
|
};
|
|
35134
|
-
worktree = true;
|
|
35135
|
-
|
|
35810
|
+
if (mode === "review") worktree = args.worktree === true ? true : void 0;
|
|
35811
|
+
else {
|
|
35812
|
+
worktree = true;
|
|
35813
|
+
if (args.worktree === false) worktreeNote = `[note: worker_${mode} always runs in an isolated git worktree; the requested worktree:false was overridden. For in-place edits, use the \`implementer\` subagent.]
|
|
35136
35814
|
|
|
35137
35815
|
`;
|
|
35816
|
+
}
|
|
35138
35817
|
}
|
|
35818
|
+
const allowProxyCwd = process.env.GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE === "1";
|
|
35819
|
+
const callerGaveWorkspace = args.workspace !== void 0;
|
|
35139
35820
|
let workspace;
|
|
35140
35821
|
if (args.workspace !== void 0) {
|
|
35141
35822
|
if (typeof args.workspace !== "string" || args.workspace.length === 0) return {
|
|
@@ -35160,7 +35841,16 @@ async function runWorkerToolCall(call) {
|
|
|
35160
35841
|
}],
|
|
35161
35842
|
isError: true
|
|
35162
35843
|
};
|
|
35163
|
-
else workspace = process.cwd();
|
|
35844
|
+
else if (allowProxyCwd) workspace = process.cwd();
|
|
35845
|
+
else return {
|
|
35846
|
+
content: [{
|
|
35847
|
+
type: "text",
|
|
35848
|
+
text: `worker_${mode}: a workspace is required. Nothing in this call said which directory to run in, and the proxy's own launch directory is not a safe guess — it is where the proxy was started, which may be a different checkout or git worktree than the one you are working in. Re-issue this call with \`workspace\` set to the absolute path of your current working directory. (Operators running a raw MCP client that cannot send one can restore the old launch-cwd default with GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE=1.)`
|
|
35849
|
+
}],
|
|
35850
|
+
isError: true
|
|
35851
|
+
};
|
|
35852
|
+
const effectiveSource = ctx?.workspaceSource ?? (callerGaveWorkspace ? "argument" : "absent");
|
|
35853
|
+
const workspaceNote = effectiveSource === "argument" ? "" : `[workspace: ${workspace} (${effectiveSource === "session" ? "from your session's working directory; pass `workspace` explicitly if you are running somewhere else, such as a git worktree" : "the proxy's launch directory"})]\n\n`;
|
|
35164
35854
|
let maxWallClockMs;
|
|
35165
35855
|
let clampNote = "";
|
|
35166
35856
|
if (args.maxWallClockMs !== void 0) {
|
|
@@ -35188,7 +35878,7 @@ async function runWorkerToolCall(call) {
|
|
|
35188
35878
|
maxWallClockMs,
|
|
35189
35879
|
signal
|
|
35190
35880
|
});
|
|
35191
|
-
const notePrefix = `${clampNote}${worktreeNote}`;
|
|
35881
|
+
const notePrefix = `${workspaceNote}${clampNote}${worktreeNote}`;
|
|
35192
35882
|
return {
|
|
35193
35883
|
content: [{
|
|
35194
35884
|
type: "text",
|
|
@@ -35421,6 +36111,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
|
|
|
35421
36111
|
return [...new Set(names)];
|
|
35422
36112
|
}
|
|
35423
36113
|
//#endregion
|
|
35424
|
-
export {
|
|
36114
|
+
export { handleMcpDelete as $, generateRandomPort as $t, satisfiesMinVersion as A, readResponseBodyCapped as At, rememberThinkingHistoryRepair as B, CONDENSED_OPERATING_SEQUENCE as Bt, availableToolCommands as C, getTokenCount as Ct, vscodeRipgrepPath as D, createResponses as Dt, toolbeltSkipSet as E, resolveMcpToolTimeoutMs as Et, injectAdvisorTool as F, provisionAndIndexColbert as Ft, isControllerClosedError as G, BUDGET_SMALL_FAST_CATALOG_ID as Gt, repairRejectedThinkingHistory as H, shouldUseInsecureTls as Ht, isAdvisorRequested as I, extractTarGzMember as It, relayAnthropicStream as J, DEFAULT_CODEX_MODEL as Jt, logStreamError as K, BUDGET_SMALL_FAST_SLUG as Kt, resolveAdvisorEffort as L, extractZipMember as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, provisionBrowserAssets as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, hasSupportedBrowserInstalled as Nt, TOOLBELT_TOOLS$1 as O, createChatCompletions as Ot, buildAdvisorStream as P, colbertDegradedWarning as Pt, clampEffort as Q, UPSTREAM_INACTIVITY_TIMEOUT_MS as Qt, resolveAdvisorModel as R, warmTreeSitterPool as Rt, buildEnv as S, createMessages as St, toolbeltEnabled as T, warnOnTokenPriceDrift as Tt, buildAnthropicErrorEvent as U, collapsePathKeys as Ut, repairKnownThinkingHistory as V, DEFINITION_OF_GREATNESS as Vt, buildOpenAIErrorEvent as W, toolbeltPathOverride as Wt, UNKNOWN_EFFORT_ANCHOR as X, DEFAULT_PORT as Xt, EFFORT_ORDER as Y, DEFAULT_CODEX_MODEL_FALLBACKS as Yt, bucketEffort as Z, UPSTREAM_FETCH_TIMEOUT_MS as Zt, appendPlanReminder as _, scribeModel as _t, buildPeerAwarenessSnippet as a, classifyMessagesRoute as an, browseAgentEnabled as at, resolveWorkerRunOpts as b, shimDefaultsToXhigh as bt, personasFor as c, withInstallLock as cn, fleetToolsEnabled as ct, EXPLORE_DEFAULT_MODEL as d, implementerFastModel as dt, isBudgetClaudeLead as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, nativeSubagentModel as ft, TEST_DEFAULT_MODEL as g, scoutModel as gt, REVIEW_DEFAULT_MODEL as h, reviewerModel as ht, buildAgentPrompt as i, upstreamMaxConnections as in, brainstormModel as it, searchWeb as j, parseJsonOrDiagnose as jt, assetFor as k, MAX_RESPONSE_BODY_BYTES as kt, BROWSE_DEFAULT_MODEL as l, geminiAvailable as lt, PLAN_DEFAULT_MODEL as m, reviewerFastModel as mt, MCP_GROUPS as n, resolveLeadSlugArg as nn, agentToolsEnabled as nt, buildPeerAwarenessSummary as o, withOneMSuffix as on, browserCompoundToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, resolveGeminiReviewModel as pt, readIteratorWithTimeout as q, DEFAULT_CLAUDE_MODEL_FALLBACKS as qt, assertMcpToolSurfaceConsistent as r, upstreamAllowH2 as rn, artifactToolsEnabled as rt, enumerateInjectedMcpToolNames as s, withOneMSuffixForLead as sn, browserToolsEnabled as st, GROUP_META as t, pickClaudeDefault as tn, REVIEW_FAST_DEFAULT_MODEL as tt, DEFAULT_MODEL_CHAIN as u, generalPurposeFastModel as ut, resolveDefaultModel as v, standInToolEnabled as vt, buildToolbeltAwareness as w, assembleResponsesPayload as wt, runWorkerAgent as x, countTokens as xt, resolveModeDefaults as y, workerToolsEnabled as yt, formatThinkingRepairDecline as z, provisionTreeSitterAssets as zt };
|
|
35425
36115
|
|
|
35426
|
-
//# sourceMappingURL=peer-mcp-personas-
|
|
36116
|
+
//# sourceMappingURL=peer-mcp-personas-B5Wp6wIn.js.map
|