github-router 0.3.276 → 0.3.285
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{attribution-settings-VOPvSn6v.js → attribution-settings-B8M2fvhz.js} +44 -45
- package/dist/attribution-settings-B8M2fvhz.js.map +1 -0
- package/dist/{auth-CYoRwhC9.js → auth-VUL2Zxvw.js} +3 -3
- package/dist/{auth-CYoRwhC9.js.map → auth-VUL2Zxvw.js.map} +1 -1
- package/dist/browser-ext/manifest.json +1 -1
- package/dist/{check-usage-DDUyg8ZB.js → check-usage-CcLdFPGr.js} +4 -4
- package/dist/{check-usage-DDUyg8ZB.js.map → check-usage-CcLdFPGr.js.map} +1 -1
- package/dist/{claude-OTHh7iKz.js → claude-e-JIQGhR.js} +55 -38
- package/dist/claude-e-JIQGhR.js.map +1 -0
- package/dist/{codex-tCxK30Su.js → codex-DSnVZ9oK.js} +5 -5
- package/dist/{codex-tCxK30Su.js.map → codex-DSnVZ9oK.js.map} +1 -1
- package/dist/{debug-CJgsxoVw.js → debug-CtzYWxpJ.js} +2 -2
- package/dist/{debug-CJgsxoVw.js.map → debug-CtzYWxpJ.js.map} +1 -1
- package/dist/engine-DMveEa9x.js +2 -0
- package/dist/{gate-discovery-nf7PGgnS.js → gate-discovery-BCFwLm0q.js} +5 -5
- package/dist/{gate-discovery-nf7PGgnS.js.map → gate-discovery-BCFwLm0q.js.map} +1 -1
- package/dist/{get-copilot-usage-D1OAH0EU.js → get-copilot-usage-B-EDAqQb.js} +2 -2
- package/dist/{get-copilot-usage-D1OAH0EU.js.map → get-copilot-usage-B-EDAqQb.js.map} +1 -1
- package/dist/hooks.mjs +37870 -0
- package/dist/hooks.sha256 +1 -0
- package/dist/{internal-artifact-open-BCz9lnq0.js → internal-artifact-open-Dj5Nlq0L.js} +2 -2
- package/dist/{internal-artifact-open-BCz9lnq0.js.map → internal-artifact-open-Dj5Nlq0L.js.map} +1 -1
- package/dist/{internal-first-mate-guard-7v2CtMA6.js → internal-first-mate-guard-B7p4NttK.js} +1 -1
- package/dist/{internal-first-mate-guard-La0tIjTU.js → internal-first-mate-guard-DOgFVki5.js} +5 -4
- package/dist/{internal-first-mate-guard-La0tIjTU.js.map → internal-first-mate-guard-DOgFVki5.js.map} +1 -1
- package/dist/{internal-plan-review-jRMpHu68.js → internal-plan-review-DRFHET_B.js} +11 -4
- package/dist/internal-plan-review-DRFHET_B.js.map +1 -0
- package/dist/{internal-prompt-submit-CqWRFCuv.js → internal-prompt-submit-f1LR2P2G.js} +4 -4
- package/dist/{internal-prompt-submit-CqWRFCuv.js.map → internal-prompt-submit-f1LR2P2G.js.map} +1 -1
- package/dist/{internal-session-bind-BiHWDxuC.js → internal-session-bind-DhZhxJ3T.js} +2 -2
- package/dist/{internal-session-bind-BiHWDxuC.js.map → internal-session-bind-DhZhxJ3T.js.map} +1 -1
- package/dist/{internal-stop-hook-Ds9LwmZA.js → internal-stop-hook-Cf-7w3RH.js} +68 -20
- package/dist/internal-stop-hook-Cf-7w3RH.js.map +1 -0
- package/dist/{internal-stop-review-BOFj7R3q.js → internal-stop-review-BBsLcbPG.js} +2 -2
- package/dist/{internal-stop-review-BOFj7R3q.js.map → internal-stop-review-BBsLcbPG.js.map} +1 -1
- package/dist/{internal-worker-guard-Lu8VHj5k.js → internal-worker-guard-CKgYFiFO.js} +2 -2
- package/dist/{internal-worker-guard-Lu8VHj5k.js.map → internal-worker-guard-CKgYFiFO.js.map} +1 -1
- package/dist/{internal-workspace-header-BRMz0Yql.js → internal-workspace-header-OYgHEnFt.js} +2 -2
- package/dist/{internal-workspace-header-BRMz0Yql.js.map → internal-workspace-header-OYgHEnFt.js.map} +1 -1
- package/dist/lib/tree-sitter-pool/lifecycle-C7W8D2lx.js +165 -0
- package/dist/lib/tree-sitter-pool/lifecycle-Dx0SnP7Z.js +157 -0
- package/dist/lib/tree-sitter-pool/worker.js +287 -14
- package/dist/{lifecycle-LA4rAuAL.js → lifecycle-B7CHqKlF.js} +2 -2
- package/dist/{lifecycle-LA4rAuAL.js.map → lifecycle-B7CHqKlF.js.map} +1 -1
- package/dist/lifecycle-CbmHMMSD.js +2 -0
- package/dist/{lifecycle-AGXnd-ZY.js → lifecycle-D-rL81tT.js} +2 -2
- package/dist/{lifecycle-AGXnd-ZY.js.map → lifecycle-D-rL81tT.js.map} +1 -1
- package/dist/lifecycle-DTcZwa7U.js +2 -0
- package/dist/lifecycle-K9oVtGde.mjs +16 -0
- package/dist/main.js +18 -18
- package/dist/{mcp-workspace-header-CGJbNeHb.js → mcp-workspace-header-ucs2SDST.js} +5 -6
- package/dist/mcp-workspace-header-ucs2SDST.js.map +1 -0
- package/dist/{models-4Q45Q2Dw.js → models-C16mBK2M.js} +3 -3
- package/dist/{models-4Q45Q2Dw.js.map → models-C16mBK2M.js.map} +1 -1
- package/dist/{orchestration-CpkG8ScN.js → orchestration-CtM6FYNx.js} +2 -2
- package/dist/{orchestration-CpkG8ScN.js.map → orchestration-CtM6FYNx.js.map} +1 -1
- package/dist/package-root-B-osctCk.js +29 -0
- package/dist/package-root-B-osctCk.js.map +1 -0
- package/dist/paths-CV9K7Xqm.js +2 -0
- package/dist/{paths-j2B7b0DZ.js → paths-wLC0InjX.js} +28 -4
- package/dist/paths-wLC0InjX.js.map +1 -0
- package/dist/{peer-mcp-personas-DRL_xT4h.js → peer-mcp-personas-V6stFvpq.js} +1158 -297
- package/dist/peer-mcp-personas-V6stFvpq.js.map +1 -0
- package/dist/{plan-review-hook-Dk5zTNke.js → plan-review-hook-Lf9ISdF6.js} +5 -6
- package/dist/plan-review-hook-Lf9ISdF6.js.map +1 -0
- package/dist/{prompt-submit-hook-lvTWwaTV.js → prompt-submit-hook-BlijaOn7.js} +6 -11
- package/dist/prompt-submit-hook-BlijaOn7.js.map +1 -0
- package/dist/{provision-HzZ547dW.js → provision-BpL6gZIt.js} +4 -4
- package/dist/{provision-HzZ547dW.js.map → provision-BpL6gZIt.js.map} +1 -1
- package/dist/self-invocation-CKMjcA5F.js +380 -0
- package/dist/self-invocation-CKMjcA5F.js.map +1 -0
- package/dist/{serve-CM3OmF4I.js → serve-CXUf7RtJ.js} +34 -27
- package/dist/serve-CXUf7RtJ.js.map +1 -0
- package/dist/{server-setup-C9r7jYnz.js → server-setup-Bppjt9xO.js} +81 -207
- package/dist/server-setup-Bppjt9xO.js.map +1 -0
- package/dist/{start-KAUCsDao.js → start-C1-jrHfU.js} +3 -3
- package/dist/{start-KAUCsDao.js.map → start-C1-jrHfU.js.map} +1 -1
- package/dist/{stop-gate-hook-CaUrldLh.js → stop-gate-hook-DgJ6sW8N.js} +100 -52
- package/dist/stop-gate-hook-DgJ6sW8N.js.map +1 -0
- package/dist/{stop-gate-policy-lAabj_pP.js → stop-gate-policy-DG5yYWGn.js} +2 -2
- package/dist/{stop-gate-policy-lAabj_pP.js.map → stop-gate-policy-DG5yYWGn.js.map} +1 -1
- package/dist/{token-CxPaBEUS.js → token-CnlB0884.js} +2 -2
- package/dist/token-CnlB0884.js.map +1 -0
- package/dist/{version-_Q1WpsQp.js → version-C8x2hQrZ.js} +14 -2
- package/dist/version-C8x2hQrZ.js.map +1 -0
- package/dist/{worker-dispatch-Bj1uYyG9.js → worker-dispatch-zW8Zi69V.js} +11 -7
- package/dist/worker-dispatch-zW8Zi69V.js.map +1 -0
- package/package.json +2 -2
- package/dist/attribution-settings-VOPvSn6v.js.map +0 -1
- package/dist/claude-OTHh7iKz.js.map +0 -1
- package/dist/engine-BKJKMAzh.js +0 -2
- package/dist/internal-plan-review-jRMpHu68.js.map +0 -1
- package/dist/internal-stop-hook-Ds9LwmZA.js.map +0 -1
- package/dist/lifecycle-DCrKbaIU.js +0 -2
- package/dist/lifecycle-KKSTKwd6.js +0 -2
- package/dist/mcp-workspace-header-CGJbNeHb.js.map +0 -1
- package/dist/paths-DN3Nio42.js +0 -2
- package/dist/paths-j2B7b0DZ.js.map +0 -1
- package/dist/peer-mcp-personas-DRL_xT4h.js.map +0 -1
- package/dist/plan-review-hook-Dk5zTNke.js.map +0 -1
- package/dist/prompt-submit-hook-lvTWwaTV.js.map +0 -1
- package/dist/serve-CM3OmF4I.js.map +0 -1
- package/dist/server-setup-C9r7jYnz.js.map +0 -1
- package/dist/stop-gate-hook-CaUrldLh.js.map +0 -1
- package/dist/token-CxPaBEUS.js.map +0 -1
- package/dist/version-_Q1WpsQp.js.map +0 -1
- package/dist/worker-dispatch-Bj1uYyG9.js.map +0 -1
|
@@ -1,13 +1,14 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { t as
|
|
3
|
-
import {
|
|
1
|
+
import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
|
|
2
|
+
import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
|
|
3
|
+
import { t as PATHS } from "./paths-wLC0InjX.js";
|
|
4
|
+
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-CnlB0884.js";
|
|
4
5
|
import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
|
|
5
|
-
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-
|
|
6
|
-
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-
|
|
6
|
+
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-B7CHqKlF.js";
|
|
7
|
+
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-ucs2SDST.js";
|
|
7
8
|
import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
|
|
8
|
-
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-
|
|
9
|
-
import {
|
|
10
|
-
import { t as liveExec } from "./orchestration-
|
|
9
|
+
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-D-rL81tT.js";
|
|
10
|
+
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DgJ6sW8N.js";
|
|
11
|
+
import { t as liveExec } from "./orchestration-CtM6FYNx.js";
|
|
11
12
|
import { createRequire } from "node:module";
|
|
12
13
|
import consola from "consola";
|
|
13
14
|
import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
|
|
@@ -162,6 +163,221 @@ function withOneMSuffix(id) {
|
|
|
162
163
|
if (oneMContextDisabled()) return id;
|
|
163
164
|
return catalogAdvertises1M(id) ? `${id}[1m]` : id;
|
|
164
165
|
}
|
|
166
|
+
/**
|
|
167
|
+
* Decorate a slug the USER named — a `-m` argument or a launcher default — with
|
|
168
|
+
* `[1m]` iff the model it actually RESOLVES to serves >=1M context.
|
|
169
|
+
*
|
|
170
|
+
* The difference from `withOneMSuffix` is the resolution step, and it exists
|
|
171
|
+
* because the two functions are handed different kinds of string. Every
|
|
172
|
+
* `withOneMSuffix` caller already holds a concrete catalog id from a catalog
|
|
173
|
+
* walk, so an exact-id match is both sufficient and the safer rule: inferring a
|
|
174
|
+
* match there could attach `[1m]` to a sibling that does not serve 1M. A lead
|
|
175
|
+
* slug is the opposite case — it is whatever the user typed, or an
|
|
176
|
+
* Anthropic-published dashed slug like `claude-opus-4-8` that the catalog
|
|
177
|
+
* carries in dotted form. Exact-id matching answers "no 1M" for those purely
|
|
178
|
+
* because it never found the entry, which is the silent under-accounting this
|
|
179
|
+
* function exists to stop.
|
|
180
|
+
*
|
|
181
|
+
* Resolving first also picks up the `-1m` SIBLING shape for free:
|
|
182
|
+
* `resolveModel`'s opus family preference maps `claude-opus-4-7` onto
|
|
183
|
+
* `claude-opus-4.7-1m-internal` when that is what the tier carries, and the
|
|
184
|
+
* sibling's own advertised window then answers the question. That is the same
|
|
185
|
+
* dual-signal conclusion `pickClaudeDefault` reaches for the family shorthand,
|
|
186
|
+
* so the two paths cannot disagree about a family both can be asked about.
|
|
187
|
+
*
|
|
188
|
+
* Idempotent: a slug that already carries the bracket is returned unchanged, so
|
|
189
|
+
* a user who pins `-m claude-opus-5[1m]` by hand does not get `[1m][1m]`. That
|
|
190
|
+
* early return deliberately does NOT re-validate the pin against the catalog.
|
|
191
|
+
* `-m claude-haiku-4-5[1m]` therefore survives even though Haiku 4.5 is a 200K
|
|
192
|
+
* model — the same as before this function existed, and `resolveModel` already
|
|
193
|
+
* warns loudly about exactly that case. Stripping a bracket the user typed
|
|
194
|
+
* would be the surprising behaviour, and it would be the only place in the
|
|
195
|
+
* launcher that overrides an explicit `-m`.
|
|
196
|
+
*
|
|
197
|
+
* A repeat can still arrive from the CLIENT side rather than from here: the
|
|
198
|
+
* `/model` picker rows are seeded already decorated, and Claude Code's alias
|
|
199
|
+
* path appends its own bracket (`getDefaultSonnetModel() + '[1m]'`), so
|
|
200
|
+
* selecting `sonnet[1m]` puts `claude-sonnet-5[1m][1m]` on the wire. That
|
|
201
|
+
* resolves to the same bare id — `resolveModel`'s strip recurses — and Claude
|
|
202
|
+
* Code's own detector is unanchored, so local accounting is right too. Pinned
|
|
203
|
+
* by a regression test in `tests/lib-utils.test.ts`.
|
|
204
|
+
*
|
|
205
|
+
* Degrades the same safe direction as everything else here. An unpopulated
|
|
206
|
+
* catalog makes `resolveModel` a pass-through and `catalogAdvertises1M` false,
|
|
207
|
+
* so the slug stays bare and Claude Code accounts at its conservative 200K
|
|
208
|
+
* default — under-accounting, never overflow.
|
|
209
|
+
*/
|
|
210
|
+
function withOneMSuffixForLead(slug) {
|
|
211
|
+
if (oneMContextDisabled()) return slug;
|
|
212
|
+
if (/\[1m\]$/i.test(slug)) return slug;
|
|
213
|
+
return catalogAdvertises1M(resolveModel(slug)) ? `${slug}[1m]` : slug;
|
|
214
|
+
}
|
|
215
|
+
//#endregion
|
|
216
|
+
//#region src/services/copilot/endpoint.ts
|
|
217
|
+
/**
|
|
218
|
+
* Catalog spellings that mean each of our two clients. Copilot is not
|
|
219
|
+
* self-consistent about the `/v1` prefix — the live catalog advertises
|
|
220
|
+
* `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
|
|
221
|
+
* fixtures carry both forms — so an exact-match on the bare spelling alone
|
|
222
|
+
* silently misses a real shape. `src/lib/model-validation.ts` already
|
|
223
|
+
* normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
|
|
224
|
+
*
|
|
225
|
+
* Matching is EXACT against this set, never a suffix/`includes` test: a
|
|
226
|
+
* `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
|
|
227
|
+
* client and must keep resolving to "serves neither".
|
|
228
|
+
*/
|
|
229
|
+
const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
|
|
230
|
+
const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
231
|
+
/**
|
|
232
|
+
* Decide which endpoint to call for a model from its catalog
|
|
233
|
+
* `supported_endpoints`. Prefers `/chat/completions` when available (the
|
|
234
|
+
* simpler, more widely-supported shape) and falls back to `/responses` for
|
|
235
|
+
* models that ONLY serve the Responses API — the gpt-5.x family except
|
|
236
|
+
* `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
|
|
237
|
+
* `*-codex` models). Returns undefined when the model serves neither, so a
|
|
238
|
+
* caller can skip it rather than 400 on `unsupported_api_for_model`.
|
|
239
|
+
*
|
|
240
|
+
* A model that OMITS `supported_endpoints` is treated as chat-eligible: the
|
|
241
|
+
* catalog historically omits the field for chat-default models, and
|
|
242
|
+
* excluding those would be a worse regression than the gap this guards.
|
|
243
|
+
*/
|
|
244
|
+
function pickEndpoint(model) {
|
|
245
|
+
const eps = model.supported_endpoints;
|
|
246
|
+
if (!eps || eps.length === 0) return "chat";
|
|
247
|
+
if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
|
|
248
|
+
if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
|
|
252
|
+
* "absent from the catalog" into "serves neither of our endpoints".
|
|
253
|
+
*
|
|
254
|
+
* This function deliberately has no default. The predecessor
|
|
255
|
+
* (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
|
|
256
|
+
* coerced both cases to "chat" — defensible for an unknown id, silently wrong
|
|
257
|
+
* for a catalog model serving only, say, `/v1/messages`: the caller would drive
|
|
258
|
+
* it through the chat client and get an opaque upstream 400 with no local
|
|
259
|
+
* signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
|
|
260
|
+
* treats that case correctly (`if (!endpoint) continue`); this makes the same
|
|
261
|
+
* distinction available to callers that resolve by id.
|
|
262
|
+
*
|
|
263
|
+
* Callers that legitimately want the chat default for an unknown id can still
|
|
264
|
+
* have it — they just have to write it, per case, on purpose.
|
|
265
|
+
*/
|
|
266
|
+
function resolveEndpointForModelId(id) {
|
|
267
|
+
const found = state.models?.data?.find((m) => m.id === id);
|
|
268
|
+
if (!found) return { kind: "unknown-model" };
|
|
269
|
+
const endpoint = pickEndpoint(found);
|
|
270
|
+
if (endpoint) return {
|
|
271
|
+
kind: "endpoint",
|
|
272
|
+
endpoint
|
|
273
|
+
};
|
|
274
|
+
return {
|
|
275
|
+
kind: "unreachable",
|
|
276
|
+
endpoints: found.supported_endpoints ?? []
|
|
277
|
+
};
|
|
278
|
+
}
|
|
279
|
+
//#endregion
|
|
280
|
+
//#region src/lib/anthropic-translate/classifier.ts
|
|
281
|
+
/**
|
|
282
|
+
* Routing classifier for `POST /v1/messages`.
|
|
283
|
+
*
|
|
284
|
+
* Claude Code speaks the Anthropic Messages wire format. Copilot only serves
|
|
285
|
+
* Claude models on its native `/v1/messages` endpoint; a gpt/gemini request
|
|
286
|
+
* sent there 400s. This classifier decides, from the RESOLVED model id and its
|
|
287
|
+
* catalog metadata, whether a request stays on the native passthrough
|
|
288
|
+
* (`createMessages`) or is diverted to the translation shim.
|
|
289
|
+
*
|
|
290
|
+
* Non-regression is the whole point — every Claude model (opus / sonnet / haiku /
|
|
291
|
+
* any `claude-*` or Anthropic-vendored id) MUST return "claude-passthrough" so
|
|
292
|
+
* its bytes reach `createMessages` unchanged — even if future catalog metadata
|
|
293
|
+
* were to (wrongly) advertise a `/responses` endpoint for it. The Claude check
|
|
294
|
+
* is therefore keyed off identity (id / vendor / family), NOT the endpoint.
|
|
295
|
+
*
|
|
296
|
+
* Non-Claude models are diverted to the translation shim by the endpoint the
|
|
297
|
+
* catalog says they serve: `/responses` models (gpt-5.5, gpt-5.3-codex, …) take
|
|
298
|
+
* the Responses path (`responses-shim`), `/chat/completions` models (gemini,
|
|
299
|
+
* and any chat-default model) take the chat path (`chat-shim`). The decision is
|
|
300
|
+
* derived from `pickEndpoint` (catalog `supported_endpoints`), never a
|
|
301
|
+
* hardcoded slug list, so it generalizes. Copilot only serves Claude models on
|
|
302
|
+
* its native `/v1/messages`, so diverting every non-Claude model to a shim is
|
|
303
|
+
* correct — a non-Claude request sent to `/v1/messages` would 400.
|
|
304
|
+
*/
|
|
305
|
+
/**
|
|
306
|
+
* Match "claude"/"anthropic" as a delimiter-bounded segment anywhere in a model
|
|
307
|
+
* id: at the start, or after a `/ _ . : -` path/version delimiter, and followed
|
|
308
|
+
* by end-of-string, another such delimiter, or a digit. This catches catalog
|
|
309
|
+
* aliases like `github/claude-3-7-sonnet` or `anthropic/…` whose vendor is
|
|
310
|
+
* "github" and whose family is empty — where the token only surfaces mid-id —
|
|
311
|
+
* while NOT firing on incidental substrings like `notclaude`. Deliberately
|
|
312
|
+
* over-inclusive at the boundaries: we fail CLOSED toward Claude (→ passthrough)
|
|
313
|
+
* so a real Claude model can never be diverted to the non-Claude shim.
|
|
314
|
+
*/
|
|
315
|
+
const CLAUDE_ID_RE = /(^|[/_.:-])(claude|anthropic)(?=$|[/_.:-]|\d)/i;
|
|
316
|
+
/**
|
|
317
|
+
* True when the target is a Claude / Anthropic model. Matches on any of:
|
|
318
|
+
* catalog vendor containing "anthropic", capability family containing "claude",
|
|
319
|
+
* or a Claude/anthropic path-segment (see `CLAUDE_ID_RE`) in ANY id we hold —
|
|
320
|
+
* the resolved id (`modelId`), the pre-resolution request id (`originalModelId`),
|
|
321
|
+
* or the catalog entry's own id (`model.id`). Conservative by design: when in
|
|
322
|
+
* doubt it returns true so a Claude request can never be diverted to the shim.
|
|
323
|
+
*/
|
|
324
|
+
function isClaudeModel(modelId, model, originalModelId) {
|
|
325
|
+
if (model) {
|
|
326
|
+
if ((model.vendor?.toLowerCase() ?? "").includes("anthropic")) return true;
|
|
327
|
+
if ((model.capabilities?.family?.toLowerCase() ?? "").includes("claude")) return true;
|
|
328
|
+
}
|
|
329
|
+
return [
|
|
330
|
+
modelId,
|
|
331
|
+
originalModelId,
|
|
332
|
+
model?.id
|
|
333
|
+
].some((id) => typeof id === "string" && CLAUDE_ID_RE.test(id));
|
|
334
|
+
}
|
|
335
|
+
/**
|
|
336
|
+
* Decide the route for a resolved model id + its catalog entry.
|
|
337
|
+
*
|
|
338
|
+
* - No model id, or a Claude model → "claude-passthrough" (existing behaviour).
|
|
339
|
+
* - A non-Claude model whose catalog endpoint is `/responses` → "responses-shim".
|
|
340
|
+
* - A non-Claude model whose catalog endpoint is `/chat/completions` (gemini and
|
|
341
|
+
* any chat-default model) → "chat-shim".
|
|
342
|
+
* - A non-Claude model absent from the catalog (so we can't confirm an endpoint)
|
|
343
|
+
* → "claude-passthrough" (unchanged; we don't divert what we can't classify).
|
|
344
|
+
* - A non-Claude model that IS in the catalog and serves NEITHER of the two shim
|
|
345
|
+
* endpoints (`pickEndpoint` → undefined) → "claude-passthrough" as well.
|
|
346
|
+
*
|
|
347
|
+
* Those last two land on the same route but are NOT the same answer, and the
|
|
348
|
+
* coincidence is deliberate rather than a collapsed default (contrast
|
|
349
|
+
* `resolveEndpointForModelId`, whose callers must tell them apart because
|
|
350
|
+
* guessing there produces an opaque upstream 400). Here neither shim is even a
|
|
351
|
+
* candidate: a shim can only speak `/responses` or `/chat/completions`, so
|
|
352
|
+
* diverting a model that serves neither would 400 just as surely. Passthrough
|
|
353
|
+
* is the better default because it is sometimes RIGHT — a non-Claude catalog
|
|
354
|
+
* model advertising `/v1/messages` is served by exactly the endpoint
|
|
355
|
+
* passthrough uses. It also preserves this module's fail-CLOSED-toward-Claude
|
|
356
|
+
* invariant: an unclassifiable model is never diverted.
|
|
357
|
+
*
|
|
358
|
+
* KNOWN GAP (audited, deliberately not fixed here): a non-Claude model serving
|
|
359
|
+
* only something we cannot speak at all (say `/embeddings`) also lands on
|
|
360
|
+
* passthrough and will still 400 upstream — `logEndpointMismatch(modelId,
|
|
361
|
+
* "/v1/messages")` logs it at the passthrough seam, but no local error is
|
|
362
|
+
* raised. Closing that needs a change in `src/routes/messages/handler.ts`,
|
|
363
|
+
* which this seam does not own. It is strictly narrower than the defect fixed
|
|
364
|
+
* in `resolveEndpointForModelId`: no such model is reachable as a Claude Code
|
|
365
|
+
* `/v1/messages` target today, whereas the plan worker's `claude-opus-5`
|
|
366
|
+
* default is.
|
|
367
|
+
*
|
|
368
|
+
* `originalModelId` is the optional pre-resolution request id; when supplied it
|
|
369
|
+
* is checked for Claude-likeness alongside the resolved id so an alias that
|
|
370
|
+
* resolves to a non-Claude-looking id can't slip past.
|
|
371
|
+
*/
|
|
372
|
+
function classifyMessagesRoute(modelId, model, originalModelId) {
|
|
373
|
+
if (!modelId) return "claude-passthrough";
|
|
374
|
+
if (isClaudeModel(modelId, model, originalModelId)) return "claude-passthrough";
|
|
375
|
+
if (!model) return "claude-passthrough";
|
|
376
|
+
const endpoint = pickEndpoint(model);
|
|
377
|
+
if (endpoint === "responses") return "responses-shim";
|
|
378
|
+
if (endpoint === "chat") return "chat-shim";
|
|
379
|
+
return "claude-passthrough";
|
|
380
|
+
}
|
|
165
381
|
//#endregion
|
|
166
382
|
//#region src/lib/port.ts
|
|
167
383
|
const DEFAULT_PORT = 8787;
|
|
@@ -208,11 +424,22 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
|
|
|
208
424
|
* This helper detects the catalog state at launch and only opts in
|
|
209
425
|
* when the backend can actually serve 1M.
|
|
210
426
|
*
|
|
211
|
-
*
|
|
212
|
-
*
|
|
213
|
-
*
|
|
214
|
-
* `
|
|
215
|
-
*
|
|
427
|
+
* This helper answers the question only for the OPUS families, because a
|
|
428
|
+
* family is what it is asked about (`-m 4.7` names no slug). Every other lead
|
|
429
|
+
* slug — `-m fast`, a full slug a power user pins, the implicit budget lead —
|
|
430
|
+
* goes through `withOneMSuffixForLead` (`./one-m-context`) instead, which
|
|
431
|
+
* resolves the slug first and then reads the resolved entry's advertised
|
|
432
|
+
* window. The two agree wherever both can be asked: a family that resolves to a
|
|
433
|
+
* 1M backend is 1M by either route.
|
|
434
|
+
*
|
|
435
|
+
* A previous revision of this comment claimed Sonnet and Haiku were left bare
|
|
436
|
+
* because "Copilot has no 1M backend for them". That was true when it was
|
|
437
|
+
* written and is now false for Sonnet: the live catalog advertises
|
|
438
|
+
* `max_context_window_tokens: 1_000_000` on both `claude-sonnet-5` and
|
|
439
|
+
* `claude-sonnet-4.6` (Haiku 4.5 really is 200K, and is left bare by the same
|
|
440
|
+
* catalog check rather than by a hardcoded family rule). Nothing here is
|
|
441
|
+
* family-gated any more — the catalog decides per model, so the next family
|
|
442
|
+
* that ships 1M is picked up without an edit.
|
|
216
443
|
*
|
|
217
444
|
* Must be called AFTER `cacheModels()` has populated `state.models`.
|
|
218
445
|
* Returns the bare slug if the catalog isn't populated (resolveModel
|
|
@@ -220,6 +447,88 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
|
|
|
220
447
|
* variant" — defaulting safe-side preserves the pre-change behavior).
|
|
221
448
|
*/
|
|
222
449
|
const DEFAULT_OPUS_FAMILY = "5";
|
|
450
|
+
/** The lead `-m fast` selects. Anthropic-published dashed slug that is also the
|
|
451
|
+
* Copilot catalog id verbatim, so `resolveModel` matches it exactly. */
|
|
452
|
+
const BUDGET_LEAD_MODEL = "claude-sonnet-5";
|
|
453
|
+
/** Small/fast tier for a budget lead, in the two forms this codebase needs.
|
|
454
|
+
*
|
|
455
|
+
* `SLUG` is the Anthropic-published DASHED form and is what goes into
|
|
456
|
+
* `ANTHROPIC_SMALL_FAST_MODEL` / `ANTHROPIC_DEFAULT_HAIKU_MODEL`: Claude Code's
|
|
457
|
+
* `/model` registry is keyed on Anthropic slugs, and seeding Copilot's dotted
|
|
458
|
+
* id there reproduces the documented `claude-opus-5` failure where the picker
|
|
459
|
+
* silently falls back to an older model. `CATALOG_ID` is Copilot's DOTTED id
|
|
460
|
+
* and is what the presence probe must test, because that is the id the catalog
|
|
461
|
+
* actually carries. `resolveModel` bridges the two at request time. */
|
|
462
|
+
const BUDGET_SMALL_FAST_SLUG = "claude-haiku-4-5";
|
|
463
|
+
const BUDGET_SMALL_FAST_CATALOG_ID = "claude-haiku-4.5";
|
|
464
|
+
/**
|
|
465
|
+
* Resolve the `-m` argument to the lead slug to launch with.
|
|
466
|
+
*
|
|
467
|
+
* - `fast` → `BUDGET_LEAD_MODEL` (budget mode)
|
|
468
|
+
* - `N.M` → the best variant of that Opus family, via `pickClaudeDefault`
|
|
469
|
+
* - a full slug → unchanged, including Copilot slugs a power user pins
|
|
470
|
+
* - absent → the ordinary default
|
|
471
|
+
*
|
|
472
|
+
* Every branch is `[1m]`-decorated against the live catalog, by
|
|
473
|
+
* `pickClaudeDefault` on the two Opus-family branches and by
|
|
474
|
+
* `withOneMSuffixForLead` on the other two. Pinning a MODEL is not a request to
|
|
475
|
+
* give up four fifths of its context window, which is what leaving the other
|
|
476
|
+
* two bare amounted to: `claude-sonnet-5` advertises a 1M window, so `-m fast`
|
|
477
|
+
* and `-m claude-sonnet-5` were both budgeted locally at Claude Code's 200K
|
|
478
|
+
* default and auto-compacted at roughly a fifth of the real window. The
|
|
479
|
+
* decoration is catalog-gated per model, so a genuinely 200K model
|
|
480
|
+
* (`claude-haiku-4.5`) still comes back bare.
|
|
481
|
+
*
|
|
482
|
+
* `fast` resolves to an ordinary slug rather than setting a mode flag, because
|
|
483
|
+
* budget mode is keyed off the RESOLVED lead everywhere it matters (the advisor
|
|
484
|
+
* escalation, the delegation prose, the small/fast tier). `-m fast` and
|
|
485
|
+
* `-m claude-sonnet-5` must therefore produce identical sessions, which a flag
|
|
486
|
+
* only one of the two set would break. The shared decoration is part of that
|
|
487
|
+
* identity: decorating one branch and not the other would reintroduce the
|
|
488
|
+
* divergence through the context budget instead of through a flag.
|
|
489
|
+
*
|
|
490
|
+
* Callers must keep treating any explicit `-m` as explicit: the
|
|
491
|
+
* `DEFAULT_CLAUDE_MODEL_FALLBACKS` walk applies to the implicit-default path
|
|
492
|
+
* only, so it cannot override a requested family (or `fast`) with an older Opus.
|
|
493
|
+
*/
|
|
494
|
+
function resolveLeadSlugArg(modelArg) {
|
|
495
|
+
const arg = modelArg?.trim();
|
|
496
|
+
if (!arg) return pickClaudeDefault();
|
|
497
|
+
if (arg.toLowerCase() === "fast") return withOneMSuffixForLead(BUDGET_LEAD_MODEL);
|
|
498
|
+
const opusFamilyShorthand = arg.match(/^(\d+\.\d+)$/)?.[1];
|
|
499
|
+
if (opusFamilyShorthand) return pickClaudeDefault(opusFamilyShorthand);
|
|
500
|
+
return withOneMSuffixForLead(arg);
|
|
501
|
+
}
|
|
502
|
+
/**
|
|
503
|
+
* True when `slug` names a Claude model that is NOT an Opus tier — the
|
|
504
|
+
* "budget lead" condition.
|
|
505
|
+
*
|
|
506
|
+
* Selecting sonnet or haiku as the lead is a decision to spend less while
|
|
507
|
+
* holding quality as far as possible, and three surfaces key off it: the
|
|
508
|
+
* advisor escalates to the Anthropic frontier (`resolveAdvisorModel`), the
|
|
509
|
+
* injected delegation prose puts the cheap agent tiers first
|
|
510
|
+
* (`buildNativeReachClauses`), and the small/fast tier drops to Haiku
|
|
511
|
+
* (`getClaudeCodeEnvVars`). One definition here so those three cannot disagree
|
|
512
|
+
* about what counts as a budget lead.
|
|
513
|
+
*
|
|
514
|
+
* Resolves before the family test so the Anthropic dashed form, Copilot's
|
|
515
|
+
* dotted form, and `pickClaudeDefault`'s literal `[1m]` suffix all classify
|
|
516
|
+
* alike. A non-Claude lead is not a budget lead: the concept is about picking a
|
|
517
|
+
* lighter tier WITHIN the Claude family, and the gpt/gemini shim models have
|
|
518
|
+
* their own cost profile that this switch says nothing about.
|
|
519
|
+
*
|
|
520
|
+
* CONTRACT: `slug` is an already-resolved LEAD SLUG, never a raw `-m` argument.
|
|
521
|
+
* `"fast"` and the `N.M` shorthand are not Claude slugs and would classify
|
|
522
|
+
* false here; run them through `resolveLeadSlugArg` first, which is what every
|
|
523
|
+
* caller does. Resolving internally instead would drag `pickClaudeDefault`'s
|
|
524
|
+
* catalog dependency into a pure predicate and make the same input answer
|
|
525
|
+
* differently before and after the catalog loads.
|
|
526
|
+
*/
|
|
527
|
+
function isBudgetClaudeLead(slug) {
|
|
528
|
+
if (!slug) return false;
|
|
529
|
+
if (!isClaudeModel(slug)) return false;
|
|
530
|
+
return !/opus/i.test(resolveModel(slug));
|
|
531
|
+
}
|
|
223
532
|
function pickClaudeDefault(opusFamily = DEFAULT_OPUS_FAMILY) {
|
|
224
533
|
const dotted = opusFamily.replace(/-/g, ".");
|
|
225
534
|
const bareSlug = `claude-opus-${dotted.replace(/\./g, "-")}`;
|
|
@@ -11888,6 +12197,123 @@ function objectProp(description, properties, required) {
|
|
|
11888
12197
|
function anyProp(description) {
|
|
11889
12198
|
return { description };
|
|
11890
12199
|
}
|
|
12200
|
+
//#endregion
|
|
12201
|
+
//#region src/lib/tree-sitter-assets/files.ts
|
|
12202
|
+
/** WASM assets code search loads from tree-sitter-wasms. */
|
|
12203
|
+
const TREE_SITTER_GRAMMAR_FILES = {
|
|
12204
|
+
typescript: "tree-sitter-typescript.wasm",
|
|
12205
|
+
tsx: "tree-sitter-tsx.wasm",
|
|
12206
|
+
javascript: "tree-sitter-javascript.wasm",
|
|
12207
|
+
python: "tree-sitter-python.wasm",
|
|
12208
|
+
go: "tree-sitter-go.wasm",
|
|
12209
|
+
rust: "tree-sitter-rust.wasm",
|
|
12210
|
+
java: "tree-sitter-java.wasm",
|
|
12211
|
+
c: "tree-sitter-c.wasm",
|
|
12212
|
+
cpp: "tree-sitter-cpp.wasm"
|
|
12213
|
+
};
|
|
12214
|
+
const TREE_SITTER_RUNTIME_FILE = "tree-sitter.wasm";
|
|
12215
|
+
//#endregion
|
|
12216
|
+
//#region src/lib/tree-sitter-assets/provision.ts
|
|
12217
|
+
const PUBLISH_ATTEMPTS = 3;
|
|
12218
|
+
let _provisioned$1 = false;
|
|
12219
|
+
let _inFlight$1;
|
|
12220
|
+
/**
|
|
12221
|
+
* Materialize every required WASM asset. Single-flight and success-cached;
|
|
12222
|
+
* transient failures remain retryable. Never throws to the launcher.
|
|
12223
|
+
*/
|
|
12224
|
+
function provisionTreeSitterAssets() {
|
|
12225
|
+
if (_provisioned$1) return Promise.resolve();
|
|
12226
|
+
if (_inFlight$1) return _inFlight$1;
|
|
12227
|
+
_inFlight$1 = Promise.resolve().then(() => provisionImpl()).then((complete) => {
|
|
12228
|
+
if (complete) _provisioned$1 = true;
|
|
12229
|
+
}).finally(() => {
|
|
12230
|
+
_inFlight$1 = void 0;
|
|
12231
|
+
});
|
|
12232
|
+
return _inFlight$1;
|
|
12233
|
+
}
|
|
12234
|
+
function provisionImpl() {
|
|
12235
|
+
try {
|
|
12236
|
+
const require = createRequire(import.meta.url);
|
|
12237
|
+
const grammarPackage = require.resolve("tree-sitter-wasms/package.json");
|
|
12238
|
+
const grammarRoot = path.join(path.dirname(grammarPackage), "out");
|
|
12239
|
+
const runtime = require.resolve("web-tree-sitter/tree-sitter.wasm");
|
|
12240
|
+
const destination = PATHS.TREE_SITTER_ASSETS_DIR;
|
|
12241
|
+
mkdirSync(destination, { recursive: true });
|
|
12242
|
+
let complete = publishAsset(runtime, path.join(destination, TREE_SITTER_RUNTIME_FILE));
|
|
12243
|
+
for (const filename of Object.values(TREE_SITTER_GRAMMAR_FILES)) complete = publishAsset(path.join(grammarRoot, filename), path.join(destination, filename)) && complete;
|
|
12244
|
+
return complete;
|
|
12245
|
+
} catch (err) {
|
|
12246
|
+
consola.debug("[tree-sitter-assets] provisioning skipped:", err);
|
|
12247
|
+
return false;
|
|
12248
|
+
}
|
|
12249
|
+
}
|
|
12250
|
+
/**
|
|
12251
|
+
* Publish through a unique same-directory temporary file. Never remove the
|
|
12252
|
+
* destination first: a concurrent parser must see either the prior complete
|
|
12253
|
+
* file or the new complete file, never a missing/partial path. A losing racer
|
|
12254
|
+
* accepts the winner when its bytes match the source.
|
|
12255
|
+
*/
|
|
12256
|
+
function publishAsset(source, destination) {
|
|
12257
|
+
let bytes;
|
|
12258
|
+
try {
|
|
12259
|
+
bytes = readFileSync(source);
|
|
12260
|
+
} catch {
|
|
12261
|
+
return false;
|
|
12262
|
+
}
|
|
12263
|
+
if (matchesContent(destination, bytes)) return true;
|
|
12264
|
+
for (let attempt = 0; attempt < PUBLISH_ATTEMPTS; attempt++) {
|
|
12265
|
+
const tmp = `${destination}.${process.pid}-${attempt}.tmp`;
|
|
12266
|
+
try {
|
|
12267
|
+
writeFileSync(tmp, bytes);
|
|
12268
|
+
renameSync(tmp, destination);
|
|
12269
|
+
return true;
|
|
12270
|
+
} catch {
|
|
12271
|
+
try {
|
|
12272
|
+
rmSync(tmp, { force: true });
|
|
12273
|
+
} catch {}
|
|
12274
|
+
if (matchesContent(destination, bytes)) return true;
|
|
12275
|
+
}
|
|
12276
|
+
}
|
|
12277
|
+
return false;
|
|
12278
|
+
}
|
|
12279
|
+
/**
|
|
12280
|
+
* Whether the published file is byte-identical to the source.
|
|
12281
|
+
*
|
|
12282
|
+
* Content, not size: a grammar upgrade that happens to keep the same byte
|
|
12283
|
+
* length would otherwise never propagate, and the stale copy would shadow the
|
|
12284
|
+
* package's newer file permanently — a silent, self-perpetuating wrong answer.
|
|
12285
|
+
* The size check first keeps the common case to one `stat`.
|
|
12286
|
+
*/
|
|
12287
|
+
function matchesContent(file, bytes) {
|
|
12288
|
+
try {
|
|
12289
|
+
const stat = statSync(file);
|
|
12290
|
+
if (!stat.isFile() || stat.size !== bytes.byteLength) return false;
|
|
12291
|
+
return readFileSync(file).equals(bytes);
|
|
12292
|
+
} catch {
|
|
12293
|
+
return false;
|
|
12294
|
+
}
|
|
12295
|
+
}
|
|
12296
|
+
/**
|
|
12297
|
+
* Whether the stable directory holds the COMPLETE asset set: the runtime plus
|
|
12298
|
+
* every grammar.
|
|
12299
|
+
*
|
|
12300
|
+
* Callers must gate on the whole set, never on the one file they are about to
|
|
12301
|
+
* read. web-tree-sitter enforces a language-ABI version between the runtime and
|
|
12302
|
+
* the grammars, so adopting a stable runtime while grammars still come from the
|
|
12303
|
+
* package tree (or the reverse) can pair mismatched builds. That surfaces as a
|
|
12304
|
+
* caught `Language.load` failure, which silently disables structural ranking —
|
|
12305
|
+
* the exact silent degradation this whole change is meant to end.
|
|
12306
|
+
*/
|
|
12307
|
+
function stableTreeSitterAssetsComplete() {
|
|
12308
|
+
const dir = PATHS.TREE_SITTER_ASSETS_DIR;
|
|
12309
|
+
return [TREE_SITTER_RUNTIME_FILE, ...Object.values(TREE_SITTER_GRAMMAR_FILES)].every((name) => {
|
|
12310
|
+
try {
|
|
12311
|
+
return statSync(path.join(dir, name)).isFile();
|
|
12312
|
+
} catch {
|
|
12313
|
+
return false;
|
|
12314
|
+
}
|
|
12315
|
+
});
|
|
12316
|
+
}
|
|
11891
12317
|
/**
|
|
11892
12318
|
* Extension → grammar key. Grammars not in this map skip structural
|
|
11893
12319
|
* parsing (the hit falls back to the regex SYMBOL_REGEX heuristic for
|
|
@@ -11915,22 +12341,11 @@ const EXTENSION_TO_LANG = {
|
|
|
11915
12341
|
".hxx": "cpp"
|
|
11916
12342
|
};
|
|
11917
12343
|
/**
|
|
11918
|
-
* Grammar key → wasm filename
|
|
11919
|
-
*
|
|
11920
|
-
*
|
|
11921
|
-
* codegen).
|
|
12344
|
+
* Grammar key → wasm filename. The stable APP_DIR copy is preferred; the
|
|
12345
|
+
* original `node_modules/tree-sitter-wasms/out/` directory remains the
|
|
12346
|
+
* first-run fallback while background provisioning completes.
|
|
11922
12347
|
*/
|
|
11923
|
-
const GRAMMAR_FILES =
|
|
11924
|
-
typescript: "tree-sitter-typescript.wasm",
|
|
11925
|
-
tsx: "tree-sitter-tsx.wasm",
|
|
11926
|
-
javascript: "tree-sitter-javascript.wasm",
|
|
11927
|
-
python: "tree-sitter-python.wasm",
|
|
11928
|
-
go: "tree-sitter-go.wasm",
|
|
11929
|
-
rust: "tree-sitter-rust.wasm",
|
|
11930
|
-
java: "tree-sitter-java.wasm",
|
|
11931
|
-
c: "tree-sitter-c.wasm",
|
|
11932
|
-
cpp: "tree-sitter-cpp.wasm"
|
|
11933
|
-
};
|
|
12348
|
+
const GRAMMAR_FILES = TREE_SITTER_GRAMMAR_FILES;
|
|
11934
12349
|
/**
|
|
11935
12350
|
* Per-language definition-shape node types. When a matched identifier
|
|
11936
12351
|
* sits inside one of these nodes AND is at the node's "name" position,
|
|
@@ -12064,12 +12479,16 @@ function getLanguageKeyForPath(filePath) {
|
|
|
12064
12479
|
}
|
|
12065
12480
|
let _grammarBundle;
|
|
12066
12481
|
/**
|
|
12067
|
-
* Resolve the
|
|
12068
|
-
* `require.resolve`
|
|
12069
|
-
* fallback
|
|
12070
|
-
*
|
|
12482
|
+
* Resolve the grammar directory. The stable APP_DIR copy wins only when the
|
|
12483
|
+
* COMPLETE set is present; otherwise `require.resolve` supplies the
|
|
12484
|
+
* package-tree fallback for first launch or a best-effort provisioning failure.
|
|
12485
|
+
*
|
|
12486
|
+
* All-or-nothing on purpose — see `stableTreeSitterAssetsComplete()`: mixing a
|
|
12487
|
+
* stable runtime with package-tree grammars can pair mismatched ABI builds and
|
|
12488
|
+
* silently disable structural ranking.
|
|
12071
12489
|
*/
|
|
12072
12490
|
function resolveGrammarRoot() {
|
|
12491
|
+
if (stableTreeSitterAssetsComplete()) return PATHS.TREE_SITTER_ASSETS_DIR;
|
|
12073
12492
|
try {
|
|
12074
12493
|
const pkgPath = __require.resolve("tree-sitter-wasms/package.json");
|
|
12075
12494
|
return path$1.join(path$1.dirname(pkgPath), "out");
|
|
@@ -12078,6 +12497,15 @@ function resolveGrammarRoot() {
|
|
|
12078
12497
|
}
|
|
12079
12498
|
}
|
|
12080
12499
|
/**
|
|
12500
|
+
* web-tree-sitter normally resolves this sidecar relative to its JS module.
|
|
12501
|
+
* Prefer the durable copy, but only under the same all-present gate the
|
|
12502
|
+
* grammars use, so the runtime and the grammars always come from one install.
|
|
12503
|
+
*/
|
|
12504
|
+
function parserInitOptions() {
|
|
12505
|
+
if (!stableTreeSitterAssetsComplete()) return void 0;
|
|
12506
|
+
return { locateFile: () => path$1.join(PATHS.TREE_SITTER_ASSETS_DIR, TREE_SITTER_RUNTIME_FILE) };
|
|
12507
|
+
}
|
|
12508
|
+
/**
|
|
12081
12509
|
* Pre-load all grammars at module-init time so the first search
|
|
12082
12510
|
* doesn't pay a ~500ms cold-start cost. The Promise is captured at
|
|
12083
12511
|
* import time and awaited per-call; per-grammar failures are caught
|
|
@@ -12088,7 +12516,7 @@ function getGrammarBundle() {
|
|
|
12088
12516
|
_grammarBundle = { ready: (async () => {
|
|
12089
12517
|
const out = /* @__PURE__ */ new Map();
|
|
12090
12518
|
try {
|
|
12091
|
-
await Parser.init();
|
|
12519
|
+
await Parser.init(parserInitOptions());
|
|
12092
12520
|
} catch (err) {
|
|
12093
12521
|
consola.warn(`[code_search] tree-sitter Parser.init failed; structural ranking disabled: ${err.message}`);
|
|
12094
12522
|
return out;
|
|
@@ -13271,16 +13699,19 @@ const STRUCTURAL_CACHE_MAX = 64;
|
|
|
13271
13699
|
const SYMBOL_REGEX = /^(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:public\s+|private\s+|protected\s+|static\s+|abstract\s+|readonly\s+)*(?:function|class|interface|type|enum|def|fn|trait|impl|module|namespace|const|let|var)\s+[A-Za-z_$]/;
|
|
13272
13700
|
let _rgResolution;
|
|
13273
13701
|
/**
|
|
13274
|
-
*
|
|
13702
|
+
* Four-tier resolution. Memoized. Mirrors cc-backup
|
|
13275
13703
|
* `src/utils/ripgrep.ts:31-65`.
|
|
13276
13704
|
*
|
|
13277
13705
|
* 1. System rg on PATH — use the literal command name `"rg"` (NOT
|
|
13278
13706
|
* the absolute path). This leverages NoDefaultCurrentDirectory-
|
|
13279
13707
|
* InExePath on Windows, preventing PATH-hijacking via a
|
|
13280
13708
|
* malicious ./rg.exe in the proxy's cwd.
|
|
13281
|
-
* 2.
|
|
13709
|
+
* 2. Router-owned toolbelt copy under APP_DIR. Its absolute path is
|
|
13710
|
+
* safe because it cannot resolve to a planted binary in the cwd,
|
|
13711
|
+
* and it survives a temp-hosted bunx package tree being reaped.
|
|
13712
|
+
* 3. Bundled via `@vscode/ripgrep` — falls back to the per-platform
|
|
13282
13713
|
* binary that `optionalDependencies` installed.
|
|
13283
|
-
*
|
|
13714
|
+
* 4. Throw — surfaced to the caller as an MCP isError response.
|
|
13284
13715
|
*/
|
|
13285
13716
|
function resolveRipgrep() {
|
|
13286
13717
|
if (_rgResolution) return _rgResolution;
|
|
@@ -13291,6 +13722,14 @@ function resolveRipgrep() {
|
|
|
13291
13722
|
};
|
|
13292
13723
|
return _rgResolution;
|
|
13293
13724
|
}
|
|
13725
|
+
const toolbeltPath = path$1.join(PATHS.TOOLBELT_BIN_DIR, process.platform === "win32" ? "rg.exe" : "rg");
|
|
13726
|
+
if (existsSync(toolbeltPath)) {
|
|
13727
|
+
_rgResolution = {
|
|
13728
|
+
rgPath: toolbeltPath,
|
|
13729
|
+
source: "toolbelt"
|
|
13730
|
+
};
|
|
13731
|
+
return _rgResolution;
|
|
13732
|
+
}
|
|
13294
13733
|
try {
|
|
13295
13734
|
const mod = __require("@vscode/ripgrep");
|
|
13296
13735
|
if (mod.rgPath && existsSync(mod.rgPath)) {
|
|
@@ -17206,14 +17645,19 @@ function findPackageRoot(startDir, maxHops = 10) {
|
|
|
17206
17645
|
}
|
|
17207
17646
|
}
|
|
17208
17647
|
/**
|
|
17209
|
-
* Resolve the github-router package root. Uses
|
|
17210
|
-
* 1.
|
|
17211
|
-
*
|
|
17212
|
-
*
|
|
17648
|
+
* Resolve the github-router package root. Uses sources in order:
|
|
17649
|
+
* 1. The explicit root baked into the relocated hook launcher's argv. From
|
|
17650
|
+
* `<APP_DIR>/hooks/`, neither entrypoint walk can find the package and cwd
|
|
17651
|
+
* is the user's workspace, so this source must win when present.
|
|
17652
|
+
* 2. process.argv[1] — the entrypoint script, walks up from there.
|
|
17653
|
+
* 3. import.meta.url of THIS module, walks up from there.
|
|
17654
|
+
* 4. process.cwd() as last resort.
|
|
17213
17655
|
*
|
|
17214
|
-
* Robust across bun (src/main.ts) and node (dist/main.js)
|
|
17656
|
+
* Robust across relocated hooks, bun (src/main.ts), and node (dist/main.js).
|
|
17215
17657
|
*/
|
|
17216
17658
|
function packageRoot() {
|
|
17659
|
+
const explicit = explicitPackageRoot();
|
|
17660
|
+
if (explicit) return explicit;
|
|
17217
17661
|
const entryPath = typeof process$1?.argv?.[1] === "string" ? process$1.argv[1] : void 0;
|
|
17218
17662
|
if (entryPath) {
|
|
17219
17663
|
const fromEntry = findPackageRoot(path.dirname(entryPath));
|
|
@@ -18470,7 +18914,7 @@ function logAudit$1(record) {
|
|
|
18470
18914
|
try {
|
|
18471
18915
|
const fs = await import("node:fs/promises");
|
|
18472
18916
|
const path = await import("node:path");
|
|
18473
|
-
const { PATHS } = await import("./paths-
|
|
18917
|
+
const { PATHS } = await import("./paths-CV9K7Xqm.js");
|
|
18474
18918
|
const dir = path.join(PATHS.APP_DIR, "browser-mcp");
|
|
18475
18919
|
await fs.mkdir(dir, { recursive: true });
|
|
18476
18920
|
const line = JSON.stringify({
|
|
@@ -19849,70 +20293,6 @@ function detectAgentCall(input) {
|
|
|
19849
20293
|
});
|
|
19850
20294
|
}
|
|
19851
20295
|
//#endregion
|
|
19852
|
-
//#region src/services/copilot/endpoint.ts
|
|
19853
|
-
/**
|
|
19854
|
-
* Catalog spellings that mean each of our two clients. Copilot is not
|
|
19855
|
-
* self-consistent about the `/v1` prefix — the live catalog advertises
|
|
19856
|
-
* `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
|
|
19857
|
-
* fixtures carry both forms — so an exact-match on the bare spelling alone
|
|
19858
|
-
* silently misses a real shape. `src/lib/model-validation.ts` already
|
|
19859
|
-
* normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
|
|
19860
|
-
*
|
|
19861
|
-
* Matching is EXACT against this set, never a suffix/`includes` test: a
|
|
19862
|
-
* `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
|
|
19863
|
-
* client and must keep resolving to "serves neither".
|
|
19864
|
-
*/
|
|
19865
|
-
const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
|
|
19866
|
-
const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
19867
|
-
/**
|
|
19868
|
-
* Decide which endpoint to call for a model from its catalog
|
|
19869
|
-
* `supported_endpoints`. Prefers `/chat/completions` when available (the
|
|
19870
|
-
* simpler, more widely-supported shape) and falls back to `/responses` for
|
|
19871
|
-
* models that ONLY serve the Responses API — the gpt-5.x family except
|
|
19872
|
-
* `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
|
|
19873
|
-
* `*-codex` models). Returns undefined when the model serves neither, so a
|
|
19874
|
-
* caller can skip it rather than 400 on `unsupported_api_for_model`.
|
|
19875
|
-
*
|
|
19876
|
-
* A model that OMITS `supported_endpoints` is treated as chat-eligible: the
|
|
19877
|
-
* catalog historically omits the field for chat-default models, and
|
|
19878
|
-
* excluding those would be a worse regression than the gap this guards.
|
|
19879
|
-
*/
|
|
19880
|
-
function pickEndpoint(model) {
|
|
19881
|
-
const eps = model.supported_endpoints;
|
|
19882
|
-
if (!eps || eps.length === 0) return "chat";
|
|
19883
|
-
if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
|
|
19884
|
-
if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
|
|
19885
|
-
}
|
|
19886
|
-
/**
|
|
19887
|
-
* `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
|
|
19888
|
-
* "absent from the catalog" into "serves neither of our endpoints".
|
|
19889
|
-
*
|
|
19890
|
-
* This function deliberately has no default. The predecessor
|
|
19891
|
-
* (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
|
|
19892
|
-
* coerced both cases to "chat" — defensible for an unknown id, silently wrong
|
|
19893
|
-
* for a catalog model serving only, say, `/v1/messages`: the caller would drive
|
|
19894
|
-
* it through the chat client and get an opaque upstream 400 with no local
|
|
19895
|
-
* signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
|
|
19896
|
-
* treats that case correctly (`if (!endpoint) continue`); this makes the same
|
|
19897
|
-
* distinction available to callers that resolve by id.
|
|
19898
|
-
*
|
|
19899
|
-
* Callers that legitimately want the chat default for an unknown id can still
|
|
19900
|
-
* have it — they just have to write it, per case, on purpose.
|
|
19901
|
-
*/
|
|
19902
|
-
function resolveEndpointForModelId(id) {
|
|
19903
|
-
const found = state.models?.data?.find((m) => m.id === id);
|
|
19904
|
-
if (!found) return { kind: "unknown-model" };
|
|
19905
|
-
const endpoint = pickEndpoint(found);
|
|
19906
|
-
if (endpoint) return {
|
|
19907
|
-
kind: "endpoint",
|
|
19908
|
-
endpoint
|
|
19909
|
-
};
|
|
19910
|
-
return {
|
|
19911
|
-
kind: "unreachable",
|
|
19912
|
-
endpoints: found.supported_endpoints ?? []
|
|
19913
|
-
};
|
|
19914
|
-
}
|
|
19915
|
-
//#endregion
|
|
19916
20296
|
//#region src/lib/browser-mcp/compressor.ts
|
|
19917
20297
|
/**
|
|
19918
20298
|
* Static fallback chain for the inner compressor. Order is preference:
|
|
@@ -22715,7 +23095,7 @@ const EMPTY_USAGE = {
|
|
|
22715
23095
|
total: 0
|
|
22716
23096
|
}
|
|
22717
23097
|
};
|
|
22718
|
-
const DEFAULT_MODEL
|
|
23098
|
+
const DEFAULT_MODEL = {
|
|
22719
23099
|
id: "unknown",
|
|
22720
23100
|
name: "unknown",
|
|
22721
23101
|
api: "unknown",
|
|
@@ -22737,7 +23117,7 @@ function createMutableAgentState(initialState) {
|
|
|
22737
23117
|
let messages = initialState?.messages?.slice() ?? [];
|
|
22738
23118
|
return {
|
|
22739
23119
|
systemPrompt: initialState?.systemPrompt ?? "",
|
|
22740
|
-
model: initialState?.model ?? DEFAULT_MODEL
|
|
23120
|
+
model: initialState?.model ?? DEFAULT_MODEL,
|
|
22741
23121
|
thinkingLevel: initialState?.thinkingLevel ?? "off",
|
|
22742
23122
|
get tools() {
|
|
22743
23123
|
return tools;
|
|
@@ -23462,18 +23842,174 @@ function resolveModelAndThinking(opts) {
|
|
|
23462
23842
|
if (!clamp) clamp = allowed[0];
|
|
23463
23843
|
return mkOk(clamp);
|
|
23464
23844
|
}
|
|
23845
|
+
const CATALOG_PRICE_SCALE = 1e9;
|
|
23846
|
+
const TOKENS_PER_MILLION = 1e6;
|
|
23847
|
+
/**
|
|
23848
|
+
* Last-resort per-1M-token prices, recorded from the live catalog on
|
|
23849
|
+
* 2026-08-12. The LIVE catalog always wins; this only fills in when
|
|
23850
|
+
* `state.models` is unpopulated, which in practice means the startup catalog
|
|
23851
|
+
* fetch failed. Without it the injected roster degrades to bare names and the
|
|
23852
|
+
* model loses the cost signal entirely for that session.
|
|
23853
|
+
*
|
|
23854
|
+
* A hardcoded copy of a value that HAS a live source is a second source of
|
|
23855
|
+
* truth, and this one has already been observed to drift: two figures written
|
|
23856
|
+
* from memory into a commit message (`gpt-5.3-codex` 400/1600, `gpt-5.5`
|
|
23857
|
+
* 500/2000) were both wrong against the live catalog (175/1400 and 500/3000).
|
|
23858
|
+
* That is exactly the silent-misroute failure this table risks, so
|
|
23859
|
+
* `warnOnTokenPriceDrift()` compares it against the live catalog once at
|
|
23860
|
+
* startup and logs any disagreement rather than letting a stale number sit
|
|
23861
|
+
* here indefinitely.
|
|
23862
|
+
*
|
|
23863
|
+
* This is NOT the same trade as `INDICATIVE_TOKENS_PER_SECOND`: throughput
|
|
23864
|
+
* cannot be derived from the catalog at all, so hardcoding is the only option
|
|
23865
|
+
* there. Price can, so hardcoding is strictly a degraded fallback.
|
|
23866
|
+
*/
|
|
23867
|
+
const FALLBACK_TOKEN_PRICES = Object.freeze({
|
|
23868
|
+
"gpt-5.6-luna": {
|
|
23869
|
+
in: 20,
|
|
23870
|
+
out: 120
|
|
23871
|
+
},
|
|
23872
|
+
"gpt-5.6-terra": {
|
|
23873
|
+
in: 200,
|
|
23874
|
+
out: 1200
|
|
23875
|
+
},
|
|
23876
|
+
"gpt-5.4-mini": {
|
|
23877
|
+
in: 75,
|
|
23878
|
+
out: 450
|
|
23879
|
+
},
|
|
23880
|
+
"claude-sonnet-5": {
|
|
23881
|
+
in: 200,
|
|
23882
|
+
out: 1e3
|
|
23883
|
+
},
|
|
23884
|
+
"gpt-5.3-codex": {
|
|
23885
|
+
in: 175,
|
|
23886
|
+
out: 1400
|
|
23887
|
+
},
|
|
23888
|
+
"claude-haiku-4.5": {
|
|
23889
|
+
in: 100,
|
|
23890
|
+
out: 500
|
|
23891
|
+
},
|
|
23892
|
+
"claude-opus-5": {
|
|
23893
|
+
in: 500,
|
|
23894
|
+
out: 2500
|
|
23895
|
+
},
|
|
23896
|
+
"gpt-5.6-sol": {
|
|
23897
|
+
in: 500,
|
|
23898
|
+
out: 3e3
|
|
23899
|
+
},
|
|
23900
|
+
"grok-4.5": {
|
|
23901
|
+
in: 200,
|
|
23902
|
+
out: 600
|
|
23903
|
+
},
|
|
23904
|
+
"gpt-5.5": {
|
|
23905
|
+
in: 500,
|
|
23906
|
+
out: 3e3
|
|
23907
|
+
},
|
|
23908
|
+
"gemini-3.6-flash": {
|
|
23909
|
+
in: 150,
|
|
23910
|
+
out: 750
|
|
23911
|
+
},
|
|
23912
|
+
"gemini-3.5-flash": {
|
|
23913
|
+
in: 150,
|
|
23914
|
+
out: 900
|
|
23915
|
+
},
|
|
23916
|
+
"gemini-3.1-pro-preview": {
|
|
23917
|
+
in: 200,
|
|
23918
|
+
out: 1200
|
|
23919
|
+
}
|
|
23920
|
+
});
|
|
23921
|
+
/**
|
|
23922
|
+
* Compare every `FALLBACK_TOKEN_PRICES` entry against the live catalog and warn
|
|
23923
|
+
* on disagreement. Call once after the catalog is populated. Makes fallback
|
|
23924
|
+
* staleness VISIBLE instead of silent: a stale entry only ever surfaces on the
|
|
23925
|
+
* degraded path, where nobody is looking, so without this it could be wrong for
|
|
23926
|
+
* months. Warn-only by design — a price mismatch must never block a launch.
|
|
23927
|
+
*/
|
|
23928
|
+
function warnOnTokenPriceDrift() {
|
|
23929
|
+
for (const [id, hardcoded] of Object.entries(FALLBACK_TOKEN_PRICES)) {
|
|
23930
|
+
const live = livePricesFor(id);
|
|
23931
|
+
if (!live) continue;
|
|
23932
|
+
if (live.in !== hardcoded.in || live.out !== hardcoded.out) consola.warn(`[model-resolve] FALLBACK_TOKEN_PRICES is stale for ${id}: hardcoded ${hardcoded.in}/${hardcoded.out}, live catalog ${live.in}/${live.out}. Update the table in src/lib/worker-agent/model-resolve.ts.`);
|
|
23933
|
+
}
|
|
23934
|
+
}
|
|
23935
|
+
/** Live-catalog price lookup with no fallback. Split out so the drift check can
|
|
23936
|
+
* compare against the catalog without the fallback masking a disagreement. */
|
|
23937
|
+
function livePricesFor(modelId) {
|
|
23938
|
+
const prices = state.models?.data.find((model) => model.id === modelId)?.billing?.token_prices;
|
|
23939
|
+
if (!prices || typeof prices.batch_size !== "number" || !Number.isSafeInteger(prices.batch_size) || prices.batch_size <= 0 || typeof prices.input_price !== "number" || !Number.isFinite(prices.input_price) || prices.input_price < 0 || typeof prices.output_price !== "number" || !Number.isFinite(prices.output_price) || prices.output_price < 0) return;
|
|
23940
|
+
const toPerMillion = (price) => price / CATALOG_PRICE_SCALE * TOKENS_PER_MILLION / prices.batch_size;
|
|
23941
|
+
return {
|
|
23942
|
+
in: toPerMillion(prices.input_price),
|
|
23943
|
+
out: toPerMillion(prices.output_price)
|
|
23944
|
+
};
|
|
23945
|
+
}
|
|
23946
|
+
/**
|
|
23947
|
+
* A model's per-1M-token prices: live catalog first, then the dated fallback
|
|
23948
|
+
* table. Still returns undefined for a model in neither, so a caller never
|
|
23949
|
+
* mistakes a guess for a fact — the fallback covers models we have actually
|
|
23950
|
+
* recorded, not every id.
|
|
23951
|
+
*/
|
|
23952
|
+
function catalogTokenPrices(modelId) {
|
|
23953
|
+
return livePricesFor(modelId) ?? FALLBACK_TOKEN_PRICES[modelId];
|
|
23954
|
+
}
|
|
23955
|
+
/**
|
|
23956
|
+
* Approximate output tokens/sec, median of n=3 per model, measured 2026-08-12
|
|
23957
|
+
* through this proxy. Reproduce with `bun scripts/bench-model-speed.ts` — the
|
|
23958
|
+
* harness is committed precisely so these numbers can be re-derived and
|
|
23959
|
+
* challenged instead of being trusted. Rounded coarsely on purpose: run-to-run
|
|
23960
|
+
* variance is large (`gpt-5.6-sol` measured 22 in an early n=1 pass and 74 at
|
|
23961
|
+
* n=3), so any digit beyond the leading one or two would be false precision.
|
|
23962
|
+
*
|
|
23963
|
+
* Wall clock includes time-to-first-token, which is why an early n=1 pass put
|
|
23964
|
+
* `gemini-3.1-pro-preview` at 9: that response emitted only 66 tokens, so TTFT
|
|
23965
|
+
* dominated. Reasoning tokens are timed but may not appear in `output_tokens`,
|
|
23966
|
+
* so heavy-reasoning models are penalised here.
|
|
23967
|
+
*
|
|
23968
|
+
* This is a deliberately hardcoded, coarse speed hint, indicative and never a
|
|
23969
|
+
* per-call benchmark: a recoverable speed retry is safer than a quality score
|
|
23970
|
+
* that silently misroutes.
|
|
23971
|
+
*
|
|
23972
|
+
* NOT the whole picture for agent work. The benchmark also measures p50 latency
|
|
23973
|
+
* to a trivial tool call, which is the workload an agent model actually spends
|
|
23974
|
+
* its turns on, and the ordering differs from raw generation: `gpt-5.6-sol`
|
|
23975
|
+
* generates at 75 but takes ~4.3s to reach a tool call, while `gpt-5.6-luna`
|
|
23976
|
+
* takes ~0.9s. That figure is deliberately NOT surfaced to the model, because a
|
|
23977
|
+
* second speed axis invites optimising a routing choice that policy already
|
|
23978
|
+
* settles (see the decorrelation note below).
|
|
23979
|
+
*/
|
|
23980
|
+
const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
|
|
23981
|
+
"gpt-5.6-luna": 120,
|
|
23982
|
+
"gpt-5.6-terra": 100,
|
|
23983
|
+
"gpt-5.4-mini": 100,
|
|
23984
|
+
"claude-sonnet-5": 100,
|
|
23985
|
+
"gpt-5.3-codex": 85,
|
|
23986
|
+
"claude-haiku-4.5": 85,
|
|
23987
|
+
"claude-opus-5": 80,
|
|
23988
|
+
"gpt-5.6-sol": 75,
|
|
23989
|
+
"grok-4.5": 70,
|
|
23990
|
+
"gpt-5.5": 65,
|
|
23991
|
+
"gemini-3.6-flash": 45,
|
|
23992
|
+
"gemini-3.5-flash": 40,
|
|
23993
|
+
"gemini-3.1-pro-preview": 25
|
|
23994
|
+
});
|
|
23995
|
+
/** Returns the approximate, indicative output speed when it was measured. */
|
|
23996
|
+
function indicativeTokensPerSecond(modelId) {
|
|
23997
|
+
return INDICATIVE_TOKENS_PER_SECOND[modelId];
|
|
23998
|
+
}
|
|
23465
23999
|
/** Worker-usable models need a big enough window to be worth delegating to. */
|
|
23466
24000
|
const CATALOG_MIN_CONTEXT = 2e5;
|
|
23467
24001
|
/**
|
|
23468
24002
|
* Derived view of the live catalog: every model a worker could actually be
|
|
23469
24003
|
* pointed at, with the metadata needed to choose between them.
|
|
23470
24004
|
*
|
|
23471
|
-
*
|
|
24005
|
+
* Derived facts only, with one explicitly-labelled exception:
|
|
24006
|
+
* `INDICATIVE_TOKENS_PER_SECOND` is a dated, coarse measurement whose speed
|
|
24007
|
+
* signal is recoverable by retrying a slow selection. A one-liner like "strong
|
|
23472
24008
|
* reasoning, weak long-context recall" cannot be computed from catalog
|
|
23473
24009
|
* metadata — it is editorial, it goes stale silently as vendors ship, and the
|
|
23474
24010
|
* asymmetry is brutal: a MISSING characterization costs one suboptimal pick
|
|
23475
24011
|
* the model recovers from, while a WRONG one misroutes invisibly at the call
|
|
23476
|
-
* site. So this ships facts and
|
|
24012
|
+
* site. So this ships facts, the recoverable speed hint, and no quality score.
|
|
23477
24013
|
*
|
|
23478
24014
|
* It exists because the hardcoded chains cannot discover anything. Models are
|
|
23479
24015
|
* live in the catalog that appear nowhere in `src/` — nobody evaluated them
|
|
@@ -23499,13 +24035,16 @@ function buildCatalogView() {
|
|
|
23499
24035
|
if (ctx < CATALOG_MIN_CONTEXT) continue;
|
|
23500
24036
|
const efforts = (supports.reasoning_effort ?? []).filter((effort) => WORKER_THINKING_LEVELS.includes(effort));
|
|
23501
24037
|
if (efforts.length === 0) continue;
|
|
24038
|
+
const prices = catalogTokenPrices(model.id);
|
|
24039
|
+
const tps = indicativeTokensPerSecond(model.id);
|
|
23502
24040
|
rows.push({
|
|
23503
24041
|
id: model.id,
|
|
23504
24042
|
vendor: model.vendor,
|
|
23505
24043
|
ctx,
|
|
23506
24044
|
...limits?.max_output_tokens ? { maxOut: limits.max_output_tokens } : {},
|
|
23507
24045
|
efforts,
|
|
23508
|
-
...
|
|
24046
|
+
...prices ?? {},
|
|
24047
|
+
...tps === void 0 ? {} : { tps }
|
|
23509
24048
|
});
|
|
23510
24049
|
}
|
|
23511
24050
|
return rows.sort((a, b) => a.id.localeCompare(b.id));
|
|
@@ -26511,13 +27050,12 @@ function geminiAvailable(source = state) {
|
|
|
26511
27050
|
* one walk instead of hand-copying it. Ids are matched EXACTLY against
|
|
26512
27051
|
* `catalog.id` — no slug translation, matching the pre-existing behavior.
|
|
26513
27052
|
*
|
|
26514
|
-
* `minContextTokens` is OPT-IN because the
|
|
26515
|
-
*
|
|
26516
|
-
*
|
|
26517
|
-
*
|
|
26518
|
-
*
|
|
26519
|
-
*
|
|
26520
|
-
* Claude Code's 200K default with no signal that anything changed.
|
|
27053
|
+
* `minContextTokens` is OPT-IN because the constraint is genuinely per-agent:
|
|
27054
|
+
* the conditional cheaper-tier agents promise 1M end to end. Enforcing the
|
|
27055
|
+
* floor here rather than by comment is what stops a chain silently degrading
|
|
27056
|
+
* when an id's advertised window shrinks upstream — `withOneMSuffix` would then
|
|
27057
|
+
* just omit the `[1m]` bracket, and the agent would be budgeted at Claude Code's
|
|
27058
|
+
* 200K default with no signal that anything changed.
|
|
26521
27059
|
*/
|
|
26522
27060
|
function firstPresentInCatalog(chain, opts) {
|
|
26523
27061
|
const models = state.models?.data;
|
|
@@ -26607,61 +27145,47 @@ function scribeModel() {
|
|
|
26607
27145
|
* (same behavior as before `scout` existed) rather than to an expensive
|
|
26608
27146
|
* impostor wearing the cheap agent's name.
|
|
26609
27147
|
*
|
|
26610
|
-
* `gpt-5.6-luna`
|
|
26611
|
-
*
|
|
26612
|
-
*
|
|
26613
|
-
*
|
|
26614
|
-
*
|
|
27148
|
+
* `gpt-5.6-luna` leads because it is the cheapest 1M-context model in the
|
|
27149
|
+
* catalog; `gemini-3.6-flash` remains the cross-vendor fallback so an OpenAI-side
|
|
27150
|
+
* outage does not remove the scout. Both entries must continue advertising at
|
|
27151
|
+
* least 1M context so Claude Code's `[1m]` accounting remains honest if an
|
|
27152
|
+
* upstream catalog entry shrinks.
|
|
26615
27153
|
*
|
|
26616
|
-
*
|
|
26617
|
-
*
|
|
26618
|
-
*
|
|
26619
|
-
*
|
|
26620
|
-
*
|
|
27154
|
+
* This chain deliberately uses literal ids rather than `EXPLORE_DEFAULT_MODEL`:
|
|
27155
|
+
* the explore worker default and scout's cross-vendor fallback are independent
|
|
27156
|
+
* policies, so retuning one must not silently collapse the other. There is no
|
|
27157
|
+
* 400K last resort. On a catalog carrying neither chain member, `scout` is
|
|
27158
|
+
* dropped rather than inheriting the lead or presenting a narrower-context agent.
|
|
26621
27159
|
*/
|
|
27160
|
+
const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.6-flash"]);
|
|
26622
27161
|
function scoutModel() {
|
|
26623
|
-
return firstPresentInCatalog(
|
|
26624
|
-
EXPLORE_DEFAULT_MODEL,
|
|
26625
|
-
"gpt-5.6-luna",
|
|
26626
|
-
DEFAULT_MODEL
|
|
26627
|
-
], { requireToolCalls: true });
|
|
26628
|
-
}
|
|
26629
|
-
/** Model for `generic` — the mid-tier catch-all. Absent → the agent is dropped.
|
|
26630
|
-
*
|
|
26631
|
-
* `gpt-5.6-sol` is deliberately NOT in this chain: the OpenAI frontier coder is
|
|
26632
|
-
* already `implementer`'s job, and a catch-all that quietly costs frontier
|
|
26633
|
-
* rates is the opposite of what this agent is for. Both entries are 1M+ and
|
|
26634
|
-
* mid-to-high capability, which is the most the description may claim. */
|
|
26635
|
-
function genericModel() {
|
|
26636
|
-
return firstPresentInCatalog(["gpt-5.6-terra", REVIEW_DEFAULT_MODEL], {
|
|
27162
|
+
return firstPresentInCatalog(SCOUT_MODEL_CHAIN, {
|
|
26637
27163
|
requireToolCalls: true,
|
|
26638
27164
|
minContextTokens: ONE_M_TOKENS
|
|
26639
27165
|
});
|
|
26640
27166
|
}
|
|
26641
|
-
/** Model for `
|
|
27167
|
+
/** Model for `implementer-fast` — the cheaper implementation tier. Absent →
|
|
27168
|
+
* the agent is dropped.
|
|
26642
27169
|
*
|
|
26643
|
-
*
|
|
26644
|
-
*
|
|
26645
|
-
*
|
|
26646
|
-
*
|
|
26647
|
-
|
|
26648
|
-
|
|
26649
|
-
function genericFastModel() {
|
|
26650
|
-
return firstPresentInCatalog([EXPLORE_DEFAULT_MODEL, "gemini-3.5-flash"], {
|
|
27170
|
+
* `gpt-5.6-sol` is deliberately NOT in this chain: changes needing frontier
|
|
27171
|
+
* judgment already belong to `implementer`, while this agent handles
|
|
27172
|
+
* well-specified, mechanical changes at a lower tier. Both entries are 1M+;
|
|
27173
|
+
* their different speed and effort properties stay out of shared claims. */
|
|
27174
|
+
function implementerFastModel() {
|
|
27175
|
+
return firstPresentInCatalog(["gpt-5.6-terra", REVIEW_DEFAULT_MODEL], {
|
|
26651
27176
|
requireToolCalls: true,
|
|
26652
27177
|
minContextTokens: ONE_M_TOKENS
|
|
26653
27178
|
});
|
|
26654
27179
|
}
|
|
26655
|
-
/** Model for `
|
|
27180
|
+
/** Model for `general-purpose-fast` — the fast, cheapest catch-all. Absent →
|
|
27181
|
+
* dropped.
|
|
26656
27182
|
*
|
|
26657
|
-
* Single-entry by design
|
|
26658
|
-
*
|
|
26659
|
-
*
|
|
26660
|
-
*
|
|
26661
|
-
* `
|
|
26662
|
-
|
|
26663
|
-
* rather than a mini one. */
|
|
26664
|
-
function genericCheapModel() {
|
|
27183
|
+
* Single-entry by design. `gpt-5.6-luna` is the cheapest model in the live
|
|
27184
|
+
* catalog and measured fastest among the catch-all candidates, while carrying
|
|
27185
|
+
* 1.05M context and the full `none..max` effort ladder. No
|
|
27186
|
+
* `-mini`/`-lite`/`-haiku` model in the catalog serves 1M, which is why this
|
|
27187
|
+
* catch-all uses a `gpt-5.6-*` slug rather than a mini one. */
|
|
27188
|
+
function generalPurposeFastModel() {
|
|
26665
27189
|
return firstPresentInCatalog(["gpt-5.6-luna"], {
|
|
26666
27190
|
requireToolCalls: true,
|
|
26667
27191
|
minContextTokens: ONE_M_TOKENS
|
|
@@ -26671,15 +27195,14 @@ function genericCheapModel() {
|
|
|
26671
27195
|
* Gate for the worker tools (`explore`, `review`, `implement`).
|
|
26672
27196
|
*
|
|
26673
27197
|
* Returns true iff BOTH:
|
|
26674
|
-
* 1. Copilot's live catalog (`state.models?.data`) contains the
|
|
26675
|
-
* worker
|
|
26676
|
-
*
|
|
26677
|
-
*
|
|
26678
|
-
*
|
|
26679
|
-
*
|
|
26680
|
-
*
|
|
26681
|
-
*
|
|
26682
|
-
* tools, since explore/review still work.)
|
|
27198
|
+
* 1. Copilot's live catalog (`state.models?.data`) contains any model in the
|
|
27199
|
+
* ordered worker gate chain (`gpt-5.6-luna` → `gpt-5.4-mini`) and that
|
|
27200
|
+
* entry advertises `capabilities.supports.tool_calls === true`. Luna leads
|
|
27201
|
+
* on qualifying tiers; mini preserves the worker surface on individual
|
|
27202
|
+
* trial and education catalogs. The catalog is the entitlement signal.
|
|
27203
|
+
* The worker loop is function-calling, so a model without tool calls is
|
|
27204
|
+
* unusable. Per-mode defaults are NOT gated here — an absent mode default
|
|
27205
|
+
* surfaces a clean resolve error rather than disabling all worker tools.
|
|
26683
27206
|
* 2. The operator hasn't set `GH_ROUTER_DISABLE_WORKER_TOOLS=1`
|
|
26684
27207
|
* (opt-out — workers ship enabled by default per plan).
|
|
26685
27208
|
*
|
|
@@ -26688,17 +27211,12 @@ function genericCheapModel() {
|
|
|
26688
27211
|
* validation in the engine, which surfaces a clean `isError`
|
|
26689
27212
|
* envelope with the catalog's eligible model ids on mismatch.
|
|
26690
27213
|
*
|
|
26691
|
-
* `
|
|
26692
|
-
*
|
|
26693
|
-
* of truth.
|
|
27214
|
+
* `WORKER_DEFAULT_MODEL_CHAIN` is imported from `src/lib/worker-agent` so the
|
|
27215
|
+
* engine owns the single source of truth for both gating and fallback order.
|
|
26694
27216
|
*/
|
|
26695
27217
|
function workerToolsEnabled() {
|
|
26696
27218
|
if (process.env.GH_ROUTER_DISABLE_WORKER_TOOLS === "1") return false;
|
|
26697
|
-
|
|
26698
|
-
if (!models) return false;
|
|
26699
|
-
const found = models.find((m) => m.id === DEFAULT_MODEL);
|
|
26700
|
-
if (!found) return false;
|
|
26701
|
-
return found.capabilities?.supports?.tool_calls === true;
|
|
27219
|
+
return firstPresentInCatalog(DEFAULT_MODEL_CHAIN, { requireToolCalls: true }) != null;
|
|
26702
27220
|
}
|
|
26703
27221
|
/**
|
|
26704
27222
|
* Gate for the compound L2 browser tools (`browser_act`, `browser_observe`,
|
|
@@ -26805,10 +27323,10 @@ function artifactToolsEnabled() {
|
|
|
26805
27323
|
* browser is on disk. The browse agent drives the SAME Chrome/Edge
|
|
26806
27324
|
* bridge as the raw `browser_*` tools, so it can't be useful without
|
|
26807
27325
|
* that surface enabled.
|
|
26808
|
-
* 2. The browse default model (`BROWSE_DEFAULT_MODEL`, `gpt-5.
|
|
27326
|
+
* 2. The browse default model (`BROWSE_DEFAULT_MODEL`, `gpt-5.6-luna`)
|
|
26809
27327
|
* is in Copilot's live catalog AND `pickEndpoint()` resolves a
|
|
26810
27328
|
* reachable endpoint for it. Unlike `workerToolsEnabled()` (which
|
|
26811
|
-
* checks `tool_calls` on the
|
|
27329
|
+
* checks `tool_calls` on the shared gate sentinel), the browse default is
|
|
26812
27330
|
* a `/responses`-only gpt-5.x model — `pickEndpoint` is the right
|
|
26813
27331
|
* reachability probe (it returns undefined only when the model
|
|
26814
27332
|
* serves neither chat nor responses).
|
|
@@ -27353,8 +27871,26 @@ function logTelemetry(t) {
|
|
|
27353
27871
|
function toolAcceptsWorkspace(tool) {
|
|
27354
27872
|
return tool.capability === "worker" || tool.toolNameHttp === "code" || tool.toolNameHttp === "run_workflow";
|
|
27355
27873
|
}
|
|
27874
|
+
/**
|
|
27875
|
+
* Fold the per-session `X-GH-Workspace` header into `args.workspace` when the
|
|
27876
|
+
* caller left it empty, and REPORT which of the two the tool ended up with.
|
|
27877
|
+
*
|
|
27878
|
+
* The return value is the load-bearing part. This function mutates `args`, so
|
|
27879
|
+
* once it has run a header-derived workspace is byte-indistinguishable from one
|
|
27880
|
+
* the caller chose — and those two cases warrant very different treatment. A
|
|
27881
|
+
* caller that named a directory has told us where it is; a header is a
|
|
27882
|
+
* connection-level default that may be stale (it is computed by a helper Claude
|
|
27883
|
+
* Code runs, and the calling agent may since have moved into a git worktree).
|
|
27884
|
+
* `runWorkerToolCall` uses the distinction to decide what to tell the caller
|
|
27885
|
+
* about the tree it actually ran in, so the provenance must survive the merge.
|
|
27886
|
+
*/
|
|
27356
27887
|
function applySessionWorkspace(args, sessionWorkspace, tool) {
|
|
27357
|
-
if (
|
|
27888
|
+
if (args.workspace !== void 0 && args.workspace !== "") return "argument";
|
|
27889
|
+
if ((!tool || toolAcceptsWorkspace(tool)) && typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace)) {
|
|
27890
|
+
args.workspace = sessionWorkspace;
|
|
27891
|
+
return "session";
|
|
27892
|
+
}
|
|
27893
|
+
return "absent";
|
|
27358
27894
|
}
|
|
27359
27895
|
async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
27360
27896
|
const params = body.params ?? {};
|
|
@@ -27388,7 +27924,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
|
27388
27924
|
personaContext = typeof args.context === "string" ? args.context : void 0;
|
|
27389
27925
|
if (args.imagePaths !== void 0) {
|
|
27390
27926
|
if (!Array.isArray(args.imagePaths) || args.imagePaths.some((v) => typeof v !== "string")) return rpcError(body.id, RPC_INVALID_PARAMS, "tools/call: arguments.imagePaths must be an array of strings");
|
|
27391
|
-
const
|
|
27927
|
+
const imageRoot = typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace) ? sessionWorkspace : process.cwd();
|
|
27928
|
+
const loaded = await loadPeerImages(args.imagePaths, imageRoot);
|
|
27392
27929
|
if (!loaded.ok) return rpcError(body.id, RPC_INVALID_PARAMS, `tools/call: ${loaded.error}`);
|
|
27393
27930
|
personaImages = loaded.images;
|
|
27394
27931
|
}
|
|
@@ -27426,8 +27963,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
|
27426
27963
|
const telemetryName = persona ? persona.agentName : nonPersonaTool.toolNameHttp;
|
|
27427
27964
|
const telemetryModel = persona ? persona.model : "(non-persona)";
|
|
27428
27965
|
try {
|
|
27429
|
-
|
|
27430
|
-
const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal);
|
|
27966
|
+
const workspaceSource = nonPersonaTool ? applySessionWorkspace(args, sessionWorkspace, nonPersonaTool) : "absent";
|
|
27967
|
+
const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal, { workspaceSource });
|
|
27431
27968
|
logTelemetry({
|
|
27432
27969
|
name: telemetryName,
|
|
27433
27970
|
model: telemetryModel,
|
|
@@ -27753,6 +28290,98 @@ function handleMcpDelete(c) {
|
|
|
27753
28290
|
return c.body(null, 200);
|
|
27754
28291
|
}
|
|
27755
28292
|
//#endregion
|
|
28293
|
+
//#region src/lib/reasoning-effort.ts
|
|
28294
|
+
/**
|
|
28295
|
+
* Reasoning-effort bucketing + clamping shared by the `/v1/messages` handler
|
|
28296
|
+
* (adaptive-thinking translation) and the Anthropic-translation shim
|
|
28297
|
+
* (thinking-budget → Responses `reasoning.effort`).
|
|
28298
|
+
*
|
|
28299
|
+
* Extracted from `src/routes/messages/handler.ts` so a `~/lib/*` module can
|
|
28300
|
+
* depend on it without importing route code (and without forming a
|
|
28301
|
+
* handler → shim → handler import cycle). `handler.ts` re-exports these for
|
|
28302
|
+
* backward compatibility with existing imports/tests.
|
|
28303
|
+
*/
|
|
28304
|
+
/**
|
|
28305
|
+
* Copilot's reasoning-effort tiers, lowest to highest.
|
|
28306
|
+
*
|
|
28307
|
+
* Both ends were added after the fact and both are load-bearing:
|
|
28308
|
+
*
|
|
28309
|
+
* `none` is advertised by every gpt-5.x entry in the live catalog. While it was
|
|
28310
|
+
* missing here it was treated as an UNRECOGNIZED value, so a client asking for
|
|
28311
|
+
* the MINIMUM on a model that does not offer it (gemini advertises only
|
|
28312
|
+
* low/medium/high) was anchored at the unknown-value tier and clamped to
|
|
28313
|
+
* `high` — the maximum. Listing it makes the clamp resolve to `low` instead,
|
|
28314
|
+
* which is what "nearest supported tier" should always have meant.
|
|
28315
|
+
*
|
|
28316
|
+
* `max` is advertised by `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`,
|
|
28317
|
+
* `claude-opus-5` and the 4.7/4.8 Opus lines, and Claude Code's effort picker
|
|
28318
|
+
* offers it for any model whose entry allows it. Listing it lets an explicit
|
|
28319
|
+
* selection pass through, and lets `clampEffort` land on it for a model that
|
|
28320
|
+
* advertises nothing lower.
|
|
28321
|
+
*
|
|
28322
|
+
* `bucketEffort` deliberately reaches neither end — see below.
|
|
28323
|
+
*/
|
|
28324
|
+
const EFFORT_ORDER = [
|
|
28325
|
+
"none",
|
|
28326
|
+
"low",
|
|
28327
|
+
"medium",
|
|
28328
|
+
"high",
|
|
28329
|
+
"xhigh",
|
|
28330
|
+
"max"
|
|
28331
|
+
];
|
|
28332
|
+
/** Anchor for an effort value that is not a recognized tier at all.
|
|
28333
|
+
*
|
|
28334
|
+
* Deliberately NOT the top of `EFFORT_ORDER`. Callers reach this only when the
|
|
28335
|
+
* incoming value is unrecognized, which is a guess — and resolving a guess to
|
|
28336
|
+
* the most expensive tier a model advertises would silently spend more than the
|
|
28337
|
+
* caller could have meant. Anchoring here and clamping DOWN keeps the behavior
|
|
28338
|
+
* identical to before `max` joined the ladder, while `max` stays reachable by
|
|
28339
|
+
* explicit, valid selection. */
|
|
28340
|
+
const UNKNOWN_EFFORT_ANCHOR = "xhigh";
|
|
28341
|
+
/**
|
|
28342
|
+
* Bucket a thinking budget into a Copilot reasoning-effort string.
|
|
28343
|
+
* `<2000`→low, `<8000`→medium, `<24000`→high, else→xhigh.
|
|
28344
|
+
* Defaults missing/non-numeric budgets to 8000 ("high").
|
|
28345
|
+
*
|
|
28346
|
+
* The ceiling stays at `xhigh` even though `max` exists: Anthropic's
|
|
28347
|
+
* `budget_tokens` is unbounded above, so any threshold chosen for a `max`
|
|
28348
|
+
* bucket would silently re-tier existing callers whose budgets already map to
|
|
28349
|
+
* `xhigh`. `max` is reachable only by explicit selection
|
|
28350
|
+
* (`output_config.effort`), which is an unambiguous request rather than an
|
|
28351
|
+
* inference from a token count.
|
|
28352
|
+
*/
|
|
28353
|
+
function bucketEffort(budget) {
|
|
28354
|
+
const n = typeof budget === "number" && Number.isFinite(budget) ? budget : 8e3;
|
|
28355
|
+
if (n < 2e3) return "low";
|
|
28356
|
+
if (n < 8e3) return "medium";
|
|
28357
|
+
if (n < 24e3) return "high";
|
|
28358
|
+
return "xhigh";
|
|
28359
|
+
}
|
|
28360
|
+
/**
|
|
28361
|
+
* Clamp a bucketed effort to the closest value in `supported`. Ties resolve to
|
|
28362
|
+
* the lower-tier option (per EFFORT_ORDER).
|
|
28363
|
+
*
|
|
28364
|
+
* Iterates EFFORT_ORDER (canonical low→xhigh) so the first match on a given
|
|
28365
|
+
* distance is always the lower-tier value, regardless of input order in
|
|
28366
|
+
* `supported`.
|
|
28367
|
+
*/
|
|
28368
|
+
function clampEffort(bucketed, supported) {
|
|
28369
|
+
if (supported.includes(bucketed)) return bucketed;
|
|
28370
|
+
const targetIdx = EFFORT_ORDER.indexOf(bucketed);
|
|
28371
|
+
let best;
|
|
28372
|
+
let bestDist = Infinity;
|
|
28373
|
+
for (let i = 0; i < EFFORT_ORDER.length; i++) {
|
|
28374
|
+
const value = EFFORT_ORDER[i];
|
|
28375
|
+
if (!supported.includes(value)) continue;
|
|
28376
|
+
const dist = Math.abs(i - targetIdx);
|
|
28377
|
+
if (dist < bestDist) {
|
|
28378
|
+
bestDist = dist;
|
|
28379
|
+
best = value;
|
|
28380
|
+
}
|
|
28381
|
+
}
|
|
28382
|
+
return best ?? bucketed;
|
|
28383
|
+
}
|
|
28384
|
+
//#endregion
|
|
27756
28385
|
//#region src/lib/stream-relay.ts
|
|
27757
28386
|
const ENCODER$1 = new TextEncoder();
|
|
27758
28387
|
/**
|
|
@@ -28240,10 +28869,15 @@ function rememberThinkingHistoryRepair(fingerprint) {
|
|
|
28240
28869
|
* re-call Copilot for the next turn — stream onto the SAME
|
|
28241
28870
|
* SSE connection (no new message_start; the original one is
|
|
28242
28871
|
* still open). Loop up to ADVISOR_MAX_TURNS times.
|
|
28243
|
-
* 4.
|
|
28244
|
-
* family than the main loop (gpt-5.6-sol
|
|
28245
|
-
*
|
|
28246
|
-
*
|
|
28872
|
+
* 4. Lead-aware model choice: route the advisor call to a different model
|
|
28873
|
+
* family than the main loop (gpt-5.6-sol) so the user gets a true "second
|
|
28874
|
+
* set of eyes" instead of Opus reviewing Opus (gemini-critic finding). When
|
|
28875
|
+
* the LEAD is a lighter Claude tier the choice inverts and the advisor
|
|
28876
|
+
* escalates to `ADVISOR_ESCALATION_MODEL` instead — see that constant for
|
|
28877
|
+
* why trading the cross-lab property is the right call on that path.
|
|
28878
|
+
* 5. Effort follows the Claude Code effort picker (`resolveAdvisorEffort`)
|
|
28879
|
+
* rather than a hardcoded constant, floored so a low picker cannot render
|
|
28880
|
+
* the consultation useless.
|
|
28247
28881
|
*
|
|
28248
28882
|
* The translate-loop is bounded to a single user request — no
|
|
28249
28883
|
* persistent state across requests is needed (unlike Phase G's
|
|
@@ -28267,6 +28901,185 @@ const ADVISOR_CLIENT_TOOL_NAME = "advisor";
|
|
|
28267
28901
|
* model — Opus 4.6/Sonnet 4.6 typically). */
|
|
28268
28902
|
const ADVISOR_DEFAULT_MODEL = "gpt-5.6-sol";
|
|
28269
28903
|
const ADVISOR_DEFAULT_EFFORT = "xhigh";
|
|
28904
|
+
/** The Anthropic frontier model the advisor escalates to when the LEAD is a
|
|
28905
|
+
* lighter Claude tier (sonnet, haiku).
|
|
28906
|
+
*
|
|
28907
|
+
* Selecting a lighter lead is a decision to work on a budget while holding
|
|
28908
|
+
* quality: the lead does the legwork and escalates for direction. Without this,
|
|
28909
|
+
* a budget lead has no transcript-aware path to the strongest Anthropic
|
|
28910
|
+
* reasoner at all — `opus_critic` is stateless and sees one artifact, and the
|
|
28911
|
+
* `plan` worker is read-only and never sees the transcript.
|
|
28912
|
+
*
|
|
28913
|
+
* This deliberately trades the advisor's cross-lab property on that path. The
|
|
28914
|
+
* advisor is not this repo's review instrument: it catches drift and momentum
|
|
28915
|
+
* and inherits the lead's framing by design, while the fresh-context critics
|
|
28916
|
+
* (`codex_critic`, `gemini_critic`, `codex_reviewer`, `gemini_reviewer`) are
|
|
28917
|
+
* the decorrelation instrument and are untouched. `GH_ROUTER_ADVISOR_MODEL`
|
|
28918
|
+
* keeps a cross-lab advisor one env var away for anyone who wants it back. */
|
|
28919
|
+
const ADVISOR_ESCALATION_MODEL = "claude-opus-5";
|
|
28920
|
+
/** Floor for the advisor's reasoning effort.
|
|
28921
|
+
*
|
|
28922
|
+
* The advisor follows the Claude Code effort picker (see `resolveAdvisorEffort`)
|
|
28923
|
+
* so dialing the picker down makes it cheaper, but it does NOT follow it all the
|
|
28924
|
+
* way to the bottom of `EFFORT_ORDER`. The advisor fires a handful of times per
|
|
28925
|
+
* session (`ADVISOR_MAX_TURNS`, typically 1-3), so it is not the budget line —
|
|
28926
|
+
* the lead's own turns are — while an advisor reasoning at `none`/`low` cannot
|
|
28927
|
+
* do the job the consultation exists for. The picker therefore governs the
|
|
28928
|
+
* `high..max` range. */
|
|
28929
|
+
const ADVISOR_MIN_EFFORT = "high";
|
|
28930
|
+
/** Output cap for the Anthropic-branch advisor call when the catalog carries no
|
|
28931
|
+
* limits for the resolved model. The value the branch used unconditionally
|
|
28932
|
+
* before it became reachable, kept so a catalog-less path is no worse off. */
|
|
28933
|
+
const ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS = 4096;
|
|
28934
|
+
/** Catalog spellings that mean the Responses API. Copilot is inconsistent about
|
|
28935
|
+
* the `/v1` prefix, so both are matched — mirroring `CHAT_ENDPOINTS` /
|
|
28936
|
+
* `RESPONSES_ENDPOINTS` in `src/services/copilot/endpoint.ts`. */
|
|
28937
|
+
const ADVISOR_RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
28938
|
+
/**
|
|
28939
|
+
* Which transport the advisor dispatches on: `/responses` (with
|
|
28940
|
+
* `reasoning.effort`) or `/v1/messages`.
|
|
28941
|
+
*
|
|
28942
|
+
* Catalog-first, name-regex second, and BOTH tests run against the bare id as
|
|
28943
|
+
* well as the given one. `pickEndpoint` is deliberately not reused: it answers
|
|
28944
|
+
* "chat or responses" for the two tool-calling clients and would send
|
|
28945
|
+
* `claude-opus-5` — which advertises `/v1/messages` AND `/chat/completions` — to
|
|
28946
|
+
* chat. The advisor's question is narrower: does this model serve `/responses`?
|
|
28947
|
+
*
|
|
28948
|
+
* The bare-id fallback is what makes `GH_ROUTER_ADVISOR_MODEL` safe. That pin is
|
|
28949
|
+
* accepted verbatim, so an operator can write a vendor-namespaced value like
|
|
28950
|
+
* `openai/gpt-5.6-sol`. Such an id is in no catalog and fails the start-anchored
|
|
28951
|
+
* name regex, so a catalog-only fix still posted it to `/v1/messages` and 400'd
|
|
28952
|
+
* — exported and directly tested for that exact input, because an earlier
|
|
28953
|
+
* version of this function claimed to handle it and did not.
|
|
28954
|
+
*/
|
|
28955
|
+
function advisorUsesResponses(resolvedAdvisorModel) {
|
|
28956
|
+
const bare = resolvedAdvisorModel.slice(resolvedAdvisorModel.lastIndexOf("/") + 1);
|
|
28957
|
+
const endpoints = (state.models?.data?.find((m) => m.id === resolvedAdvisorModel || m.id === bare))?.supported_endpoints;
|
|
28958
|
+
if (endpoints && endpoints.length > 0) return endpoints.some((e) => ADVISOR_RESPONSES_ENDPOINTS.has(e));
|
|
28959
|
+
return /^(gpt-|o\d|.*codex)/i.test(bare);
|
|
28960
|
+
}
|
|
28961
|
+
/** True when the model advertises a usable reasoning-effort ladder. */
|
|
28962
|
+
function advertisedEffortLadder(resolvedAdvisorModel) {
|
|
28963
|
+
const supported = state.models?.data?.find((m) => m.id === resolvedAdvisorModel)?.capabilities?.supports?.reasoning_effort;
|
|
28964
|
+
return Array.isArray(supported) && supported.length > 0 ? supported : void 0;
|
|
28965
|
+
}
|
|
28966
|
+
/** True when the advisor should escalate to `ADVISOR_ESCALATION_MODEL` for this
|
|
28967
|
+
* lead: a Claude lead that is NOT already an Opus tier, on a catalog that
|
|
28968
|
+
* actually carries the escalation model.
|
|
28969
|
+
*
|
|
28970
|
+
* The catalog probe mirrors `standInToolEnabled`'s: never name a model the
|
|
28971
|
+
* account cannot reach. A non-Claude lead never gets here in practice (the
|
|
28972
|
+
* advisor tool is stripped for those before the request reaches this module),
|
|
28973
|
+
* but the check is explicit rather than assumed.
|
|
28974
|
+
*
|
|
28975
|
+
* The probe compares the BARE constant rather than `resolveModel`-ing it first,
|
|
28976
|
+
* which is deliberate and not an oversight: `claude-opus-5` is a single-segment
|
|
28977
|
+
* slug whose dashed and dotted spellings are identical, so resolution is a
|
|
28978
|
+
* no-op, and `resolveModel` WARNS on an id it cannot find — routing this probe
|
|
28979
|
+
* through it would emit that warning on every advisor request for anyone whose
|
|
28980
|
+
* catalog lacks opus-5, which is exactly the tier this returns false for.
|
|
28981
|
+
* `standInToolEnabled` compares the same id the same way. */
|
|
28982
|
+
function shouldEscalateAdvisor(leadModel) {
|
|
28983
|
+
if (!isBudgetClaudeLead(leadModel)) return false;
|
|
28984
|
+
return state.models?.data?.some((m) => m.id === "claude-opus-5") ?? false;
|
|
28985
|
+
}
|
|
28986
|
+
/**
|
|
28987
|
+
* Pick the advisor model for one request from the LEAD model that request is
|
|
28988
|
+
* running on.
|
|
28989
|
+
*
|
|
28990
|
+
* Resolved per request rather than at launch because the lead changes
|
|
28991
|
+
* mid-session via the `/model` picker; launch-time env plumbing would pin the
|
|
28992
|
+
* advisor to whatever was selected at spawn.
|
|
28993
|
+
*
|
|
28994
|
+
* Precedence:
|
|
28995
|
+
* 1. `GH_ROUTER_ADVISOR_MODEL` (trimmed) — the operator pin, checked first so
|
|
28996
|
+
* it works on every lead.
|
|
28997
|
+
* 2. A lighter Claude lead with the escalation model in the catalog.
|
|
28998
|
+
* 3. `ADVISOR_DEFAULT_MODEL`.
|
|
28999
|
+
*
|
|
29000
|
+
* Step 3 returns the LITERAL constant rather than walking the OpenAI frontier
|
|
29001
|
+
* chain. An Opus lead must resolve to exactly what it resolves to today, and a
|
|
29002
|
+
* frontier walk could yield `gpt-5.5` on a catalog missing `gpt-5.6-sol` —
|
|
29003
|
+
* a silent change to the one path that is required not to move.
|
|
29004
|
+
*/
|
|
29005
|
+
/**
|
|
29006
|
+
* Map an operator pin onto the id the catalog actually carries.
|
|
29007
|
+
*
|
|
29008
|
+
* `GH_ROUTER_ADVISOR_MODEL` is free-form, and the natural thing to write is a
|
|
29009
|
+
* vendor-namespaced id like `openai/gpt-5.6-sol`. Copilot's catalog carries the
|
|
29010
|
+
* bare `gpt-5.6-sol`, so forwarding the namespaced form verbatim gets a 400
|
|
29011
|
+
* `model_not_supported` and the advisor silently degrades to its
|
|
29012
|
+
* "[Advisor unavailable: ...]" fallback — measured, not theorised: choosing the
|
|
29013
|
+
* transport correctly was NOT sufficient, because the id itself was still
|
|
29014
|
+
* wrong on the wire.
|
|
29015
|
+
*
|
|
29016
|
+
* An exact catalog hit wins first, so a real id containing a slash could never
|
|
29017
|
+
* be mangled. Only when the pin is absent from the catalog do we try its last
|
|
29018
|
+
* path segment, and only when THAT is present do we rewrite. A pin that matches
|
|
29019
|
+
* nothing is passed through untouched: the catalog may simply not be loaded
|
|
29020
|
+
* yet, and inventing an id would be worse than letting upstream reject it.
|
|
29021
|
+
*/
|
|
29022
|
+
function normalizeAdvisorPin(pinned) {
|
|
29023
|
+
const models = state.models?.data;
|
|
29024
|
+
if (!models) return pinned;
|
|
29025
|
+
if (models.some((m) => m.id === pinned)) return pinned;
|
|
29026
|
+
const bare = pinned.slice(pinned.lastIndexOf("/") + 1);
|
|
29027
|
+
return bare !== pinned && models.some((m) => m.id === bare) ? bare : pinned;
|
|
29028
|
+
}
|
|
29029
|
+
function resolveAdvisorModel(leadModel) {
|
|
29030
|
+
const pinned = process.env.GH_ROUTER_ADVISOR_MODEL?.trim();
|
|
29031
|
+
if (pinned) return {
|
|
29032
|
+
model: normalizeAdvisorPin(pinned),
|
|
29033
|
+
escalated: false
|
|
29034
|
+
};
|
|
29035
|
+
if (leadModel && shouldEscalateAdvisor(leadModel)) return {
|
|
29036
|
+
model: ADVISOR_ESCALATION_MODEL,
|
|
29037
|
+
escalated: true
|
|
29038
|
+
};
|
|
29039
|
+
return {
|
|
29040
|
+
model: ADVISOR_DEFAULT_MODEL,
|
|
29041
|
+
escalated: false
|
|
29042
|
+
};
|
|
29043
|
+
}
|
|
29044
|
+
/**
|
|
29045
|
+
* Resolve the advisor's reasoning effort from the ORIGINAL request body, so the
|
|
29046
|
+
* advisor thinks at the level selected in the Claude Code effort picker instead
|
|
29047
|
+
* of a hardcoded constant.
|
|
29048
|
+
*
|
|
29049
|
+
* The source is the RAW pre-`resolveModelInBody` body, deliberately. By the time
|
|
29050
|
+
* the handler holds a parsed body, `translateThinking` has already bucketed
|
|
29051
|
+
* `thinking.budget_tokens` into `output_config.effort` AND clamped it to the
|
|
29052
|
+
* LEAD model's allowlist — so that value encodes "what the lead could do", not
|
|
29053
|
+
* "what the user picked". Re-clamping it against the advisor cannot recover the
|
|
29054
|
+
* difference: a `max` pick on a lead whose ceiling is `high` would reach an
|
|
29055
|
+
* xhigh-capable advisor as `high`.
|
|
29056
|
+
*
|
|
29057
|
+
* Precedence mirrors the repo-wide rule that an explicit client effort wins:
|
|
29058
|
+
* 1. `output_config.effort`
|
|
29059
|
+
* 2. `bucketEffort(thinking.budget_tokens)`
|
|
29060
|
+
* 3. `ADVISOR_DEFAULT_EFFORT` — a request expressing no preference behaves
|
|
29061
|
+
* exactly as it did before the picker was honored at all.
|
|
29062
|
+
*
|
|
29063
|
+
* Then floor, THEN clamp. The order is load-bearing: a model whose ceiling sits
|
|
29064
|
+
* below `ADVISOR_MIN_EFFORT` must still receive a value it accepts, so the clamp
|
|
29065
|
+
* is allowed to pull back under the floor. Flipping the two would forward an
|
|
29066
|
+
* effort upstream rejects.
|
|
29067
|
+
*/
|
|
29068
|
+
function resolveAdvisorEffort(rawRequestBody, advisorModel) {
|
|
29069
|
+
let requested = ADVISOR_DEFAULT_EFFORT;
|
|
29070
|
+
if (rawRequestBody) try {
|
|
29071
|
+
const body = JSON.parse(rawRequestBody);
|
|
29072
|
+
const oc = body.output_config;
|
|
29073
|
+
const explicit = oc && typeof oc === "object" ? oc.effort : void 0;
|
|
29074
|
+
const thinking = body.thinking;
|
|
29075
|
+
if (typeof explicit === "string" && EFFORT_ORDER.includes(explicit)) requested = explicit;
|
|
29076
|
+
else if (thinking && typeof thinking === "object" && thinking.type === "enabled") requested = bucketEffort(thinking.budget_tokens);
|
|
29077
|
+
} catch {}
|
|
29078
|
+
const floored = EFFORT_ORDER.indexOf(requested) < EFFORT_ORDER.indexOf(ADVISOR_MIN_EFFORT) ? ADVISOR_MIN_EFFORT : requested;
|
|
29079
|
+
const supported = state.models?.data?.find((m) => m.id === resolveModel(advisorModel))?.capabilities?.supports?.reasoning_effort;
|
|
29080
|
+
if (!Array.isArray(supported) || supported.length === 0) return floored;
|
|
29081
|
+
return clampEffort(floored, supported);
|
|
29082
|
+
}
|
|
28270
29083
|
/** ADVISOR_TOOL_INSTRUCTIONS verbatim from cc-backup
|
|
28271
29084
|
* src/utils/advisor.ts — describes when the model should invoke
|
|
28272
29085
|
* the advisor. Long-form prose; see source for justification. */
|
|
@@ -28362,8 +29175,12 @@ const ADVISOR_FALLBACK_MAX_TOKENS = 24e4;
|
|
|
28362
29175
|
* our o200k count and Copilot's full-payload count. The transcript token
|
|
28363
29176
|
* budget is `max_prompt_tokens - reserve`. Generous on purpose: a 400
|
|
28364
29177
|
* `model_max_prompt_tokens_exceeded` degrades to a silent advisor
|
|
28365
|
-
* fallback, and the
|
|
28366
|
-
* gpt-5.6-sol
|
|
29178
|
+
* fallback, and the window given up is marginal against either advisor
|
|
29179
|
+
* model's real prompt window (`claude-opus-5` 936k, `gpt-5.6-sol` ~1M off
|
|
29180
|
+
* the live catalog). Sized as a fraction of the smaller of the two, not as
|
|
29181
|
+
* "irrelevant next to ~1M" — that framing assumed the advisor was always
|
|
29182
|
+
* the cheap side of the pair, which stopped being true once a budget lead
|
|
29183
|
+
* escalates to Opus. */
|
|
28367
29184
|
const ADVISOR_PROMPT_TOKEN_RESERVE = 8e3;
|
|
28368
29185
|
/**
|
|
28369
29186
|
* Derive the TOKEN budget for the rendered transcript from the advisor
|
|
@@ -28474,9 +29291,9 @@ function truncateTailToUnits(text, maxUnits, measure) {
|
|
|
28474
29291
|
* Anthropic's own ADVISOR ("see the whole task + every tool call +
|
|
28475
29292
|
* every result").
|
|
28476
29293
|
*/
|
|
28477
|
-
async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
29294
|
+
async function runAdvisor(conversation, advisorModel, advisorEffort, signal, advisorEscalated = false) {
|
|
28478
29295
|
if (signal?.aborted) throw new Error("advisor call aborted before dispatch");
|
|
28479
|
-
const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance.";
|
|
29296
|
+
const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance." + (advisorEscalated ? " The requesting agent is running a lighter, faster model than you. Give a directive recommendation and commit to the decision rather than laying out options for it to weigh." : "");
|
|
28480
29297
|
const resolvedAdvisorModel = resolveModel(advisorModel);
|
|
28481
29298
|
let measure;
|
|
28482
29299
|
let maxUnits;
|
|
@@ -28491,7 +29308,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
|
28491
29308
|
maxUnits = ADVISOR_MAX_CONVERSATION_CHARS;
|
|
28492
29309
|
}
|
|
28493
29310
|
const conversationText = renderConversationAsText(conversation, maxUnits, measure);
|
|
28494
|
-
if (
|
|
29311
|
+
if (advisorUsesResponses(resolvedAdvisorModel)) {
|
|
28495
29312
|
const payload = {
|
|
28496
29313
|
model: resolvedAdvisorModel,
|
|
28497
29314
|
instructions: advisorSystem,
|
|
@@ -28526,15 +29343,22 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
|
|
|
28526
29343
|
if (!text) throw new Error(`Advisor model ${resolvedAdvisorModel} returned empty assistant output`);
|
|
28527
29344
|
return text;
|
|
28528
29345
|
}
|
|
29346
|
+
const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
|
|
29347
|
+
const limits = advisorEntry?.capabilities?.limits;
|
|
29348
|
+
const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
|
|
28529
29349
|
const advisorBody = JSON.stringify({
|
|
28530
29350
|
model: resolvedAdvisorModel,
|
|
28531
|
-
max_tokens:
|
|
29351
|
+
max_tokens: maxTokens,
|
|
28532
29352
|
system: advisorSystem,
|
|
28533
29353
|
messages: [{
|
|
28534
29354
|
role: "user",
|
|
28535
29355
|
content: conversationText
|
|
28536
29356
|
}],
|
|
28537
|
-
stream: false
|
|
29357
|
+
stream: false,
|
|
29358
|
+
...advisorEntry?.capabilities?.supports?.adaptive_thinking ? {
|
|
29359
|
+
thinking: { type: "adaptive" },
|
|
29360
|
+
...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
|
|
29361
|
+
} : {}
|
|
28538
29362
|
});
|
|
28539
29363
|
const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
|
|
28540
29364
|
signal,
|
|
@@ -28589,6 +29413,7 @@ function sseEvent(type, data) {
|
|
|
28589
29413
|
function buildAdvisorStream(opts) {
|
|
28590
29414
|
const advisorModel = opts.advisorModel ?? "gpt-5.6-sol";
|
|
28591
29415
|
const advisorEffort = opts.advisorEffort ?? "xhigh";
|
|
29416
|
+
const advisorEscalated = opts.advisorEscalated ?? false;
|
|
28592
29417
|
const aborter = opts.externalAborter ?? new AbortController();
|
|
28593
29418
|
let conversation = [...opts.initialConversation];
|
|
28594
29419
|
return new ReadableStream({
|
|
@@ -28838,7 +29663,7 @@ function buildAdvisorStream(opts) {
|
|
|
28838
29663
|
const advisorConversation = conversation;
|
|
28839
29664
|
const advisorTexts = await Promise.all(advisorToolUses.map(async () => {
|
|
28840
29665
|
try {
|
|
28841
|
-
return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal);
|
|
29666
|
+
return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal, advisorEscalated);
|
|
28842
29667
|
} catch (err) {
|
|
28843
29668
|
if (aborter.signal.aborted) throw err;
|
|
28844
29669
|
const msg = err instanceof Error ? err.message : String(err);
|
|
@@ -31561,32 +32386,33 @@ async function saveOverflowPatch(dir) {
|
|
|
31561
32386
|
*/
|
|
31562
32387
|
const WORKTREE_REGISTRY = new WorktreeRegistry();
|
|
31563
32388
|
registerExitHandlers$1(WORKTREE_REGISTRY);
|
|
31564
|
-
/** Worker-availability
|
|
31565
|
-
*
|
|
31566
|
-
*
|
|
31567
|
-
*
|
|
31568
|
-
*
|
|
31569
|
-
*
|
|
31570
|
-
*
|
|
31571
|
-
* `
|
|
31572
|
-
*
|
|
31573
|
-
|
|
31574
|
-
const DEFAULT_MODEL = "gpt-5.4-mini";
|
|
32389
|
+
/** Worker-availability gate + unmatched-mode fallback chain. Luna leads where
|
|
32390
|
+
* the live catalog grants access: it is cheaper, faster, and has a larger context
|
|
32391
|
+
* window than mini. Mini remains the broad-tier fallback because individual-trial
|
|
32392
|
+
* and education catalogs may omit Luna. The catalog itself is the entitlement
|
|
32393
|
+
* signal; `billing.restricted_to` describes model policy, not the user's tier.
|
|
32394
|
+
*
|
|
32395
|
+
* `workerToolsEnabled()` admits the worker surface when either entry is present
|
|
32396
|
+
* with `tool_calls`. `resolveDefaultModel()` picks the first usable live entry for
|
|
32397
|
+
* an unmatched worker mode. Per-mode defaults remain independent of the gate. */
|
|
32398
|
+
const DEFAULT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gpt-5.4-mini"]);
|
|
31575
32399
|
const DEFAULT_THINKING = "xhigh";
|
|
31576
|
-
|
|
31577
|
-
|
|
31578
|
-
|
|
31579
|
-
|
|
31580
|
-
|
|
31581
|
-
* `
|
|
31582
|
-
*
|
|
31583
|
-
*
|
|
31584
|
-
*
|
|
31585
|
-
*
|
|
31586
|
-
|
|
31587
|
-
|
|
31588
|
-
*
|
|
31589
|
-
|
|
32400
|
+
function resolveDefaultModel() {
|
|
32401
|
+
const models = state.models?.data ?? [];
|
|
32402
|
+
return DEFAULT_MODEL_CHAIN.find((id) => models.some((model) => model.id === id && model.capabilities?.supports?.tool_calls === true)) ?? DEFAULT_MODEL_CHAIN[0];
|
|
32403
|
+
}
|
|
32404
|
+
/** Default model for the READ-ONLY `explore` mode. `gpt-5.6-luna` at `high`
|
|
32405
|
+
* (via `EXPLORE_DEFAULT_THINKING`) is the measured strict improvement over the
|
|
32406
|
+
* former Gemini Flash default: lower token cost, faster generation and tool-call
|
|
32407
|
+
* latency, a larger context window, and the full reasoning-effort ladder. `high`
|
|
32408
|
+
* is therefore a real selected tier rather than a clamp. Like `implement`'s
|
|
32409
|
+
* gpt-5.6-sol this per-mode default is NOT a `workerToolsEnabled` gate input — if
|
|
32410
|
+
* absent on a thin catalog, `explore` errors helpfully at call time rather than
|
|
32411
|
+
* vanishing the whole worker surface. The caller can override model and thinking
|
|
32412
|
+
* per call via the `model` / `thinking` args. */
|
|
32413
|
+
const EXPLORE_DEFAULT_MODEL = "gpt-5.6-luna";
|
|
32414
|
+
/** Default thinking for `explore`. Explicit rather than inherited from
|
|
32415
|
+
* `DEFAULT_THINKING` so the explore effort cannot drift with the fallback. */
|
|
31590
32416
|
const EXPLORE_DEFAULT_THINKING = "high";
|
|
31591
32417
|
/** Default model + thinking for the READ-ONLY `review` mode.
|
|
31592
32418
|
* `gemini-3.1-pro-preview` at `xhigh` (clamped to `high` at call time — gemini
|
|
@@ -31603,41 +32429,44 @@ const EXPLORE_DEFAULT_THINKING = "high";
|
|
|
31603
32429
|
const REVIEW_DEFAULT_MODEL = "gemini-3.1-pro-preview";
|
|
31604
32430
|
const REVIEW_DEFAULT_THINKING = "xhigh";
|
|
31605
32431
|
/** Default model + thinking for the READ+WRITE `implement` mode. `gpt-5.6-sol`
|
|
31606
|
-
* at `
|
|
31607
|
-
*
|
|
31608
|
-
*
|
|
31609
|
-
*
|
|
32432
|
+
* at `high` — a time-to-outcome default for the 1M+ context model, routed
|
|
32433
|
+
* through `/responses` by the stream-fn endpoint split. Any caller can restore
|
|
32434
|
+
* a higher tier per call via `thinking` or per session via `worker_defaults`;
|
|
32435
|
+
* precedence is per-call > session > built-in. */
|
|
31610
32436
|
const IMPLEMENT_DEFAULT_MODEL = "gpt-5.6-sol";
|
|
31611
|
-
const IMPLEMENT_DEFAULT_THINKING = "
|
|
31612
|
-
/** `test` starts with the same built-in pair as `implement`, but
|
|
31613
|
-
* independent
|
|
32437
|
+
const IMPLEMENT_DEFAULT_THINKING = "high";
|
|
32438
|
+
/** `test` starts with the same time-to-outcome built-in pair as `implement`, but
|
|
32439
|
+
* remains independent so either mode can restore a higher tier per call via
|
|
32440
|
+
* `thinking` or per session via `worker_defaults`. Resolution precedence is
|
|
32441
|
+
* per-call > session > built-in. */
|
|
31614
32442
|
const TEST_DEFAULT_MODEL = "gpt-5.6-sol";
|
|
31615
|
-
const TEST_DEFAULT_THINKING = "
|
|
31616
|
-
/** Default model for `browse` mode. `gpt-5.
|
|
31617
|
-
*
|
|
31618
|
-
*
|
|
31619
|
-
*
|
|
31620
|
-
*
|
|
31621
|
-
* the
|
|
31622
|
-
*
|
|
32443
|
+
const TEST_DEFAULT_THINKING = "high";
|
|
32444
|
+
/** Default model for `browse` mode. `gpt-5.6-luna` has the same measured image
|
|
32445
|
+
* ceiling as `gpt-5.4-mini`, so screenshot-heavy sessions retain their input
|
|
32446
|
+
* capacity, and endpoint routing is derived from the live catalog. Luna's full
|
|
32447
|
+
* reasoning-effort ladder preserves the independent `high` browse default.
|
|
32448
|
+
* This is deliberately a per-mode default even though it also leads
|
|
32449
|
+
* `DEFAULT_MODEL_CHAIN`; the general worker gate remains tier-adaptive while
|
|
32450
|
+
* browse still applies its independent reachability gate. Caller can override
|
|
31623
32451
|
* per call via the `model` arg.
|
|
31624
32452
|
*
|
|
31625
32453
|
* Exported so the MCP browse handler reads the same constant — drift
|
|
31626
32454
|
* between the two would ship a tool whose docs disagree with its runtime
|
|
31627
32455
|
* default. */
|
|
31628
|
-
const BROWSE_DEFAULT_MODEL = "gpt-5.
|
|
32456
|
+
const BROWSE_DEFAULT_MODEL = "gpt-5.6-luna";
|
|
31629
32457
|
/** Default thinking for `browse`. Higher than the page-driving workload
|
|
31630
32458
|
* strictly needs, but the termination discipline benefits from it. */
|
|
31631
32459
|
const BROWSE_DEFAULT_THINKING = "high";
|
|
31632
|
-
/** Default model + thinking for the read-only `plan` mode. `claude-opus-
|
|
31633
|
-
* at `
|
|
31634
|
-
*
|
|
31635
|
-
*
|
|
31636
|
-
*
|
|
31637
|
-
*
|
|
31638
|
-
*
|
|
31639
|
-
*
|
|
31640
|
-
*
|
|
32460
|
+
/** Default model + thinking for the read-only `plan` mode. `claude-opus-5`
|
|
32461
|
+
* at `high` favours time-to-outcome while retaining the strongest planning
|
|
32462
|
+
* model rather than the lightweight `gpt-5.6-luna` explore default. Any
|
|
32463
|
+
* caller can restore a higher tier per call via `thinking` or per session via
|
|
32464
|
+
* `worker_defaults`; precedence is per-call > session > built-in. Uses the
|
|
32465
|
+
* DOTTED Copilot catalog id (the worker resolver exact-matches `catalog.id`, it
|
|
32466
|
+
* does NOT translate the Anthropic dashed slug; `claude-opus-5` is a
|
|
32467
|
+
* single-segment slug so dotted == dashed). Falls back to a helpful unknown-model
|
|
32468
|
+
* error at call time if opus-5 isn't in the catalog (e.g. a non-enterprise tier),
|
|
32469
|
+
* exactly like `implement`'s `gpt-5.6-sol`. */
|
|
31641
32470
|
const PLAN_DEFAULT_MODEL = "claude-opus-5";
|
|
31642
32471
|
const BUILT_IN_MODE_DEFAULTS = Object.freeze({
|
|
31643
32472
|
explore: {
|
|
@@ -31650,7 +32479,7 @@ const BUILT_IN_MODE_DEFAULTS = Object.freeze({
|
|
|
31650
32479
|
},
|
|
31651
32480
|
plan: {
|
|
31652
32481
|
model: PLAN_DEFAULT_MODEL,
|
|
31653
|
-
thinking: "
|
|
32482
|
+
thinking: "high"
|
|
31654
32483
|
},
|
|
31655
32484
|
implement: {
|
|
31656
32485
|
model: IMPLEMENT_DEFAULT_MODEL,
|
|
@@ -31665,10 +32494,10 @@ const BUILT_IN_MODE_DEFAULTS = Object.freeze({
|
|
|
31665
32494
|
thinking: BROWSE_DEFAULT_THINKING
|
|
31666
32495
|
}
|
|
31667
32496
|
});
|
|
31668
|
-
/** Resolve the effective mode ladder without changing the gate
|
|
32497
|
+
/** Resolve the effective mode ladder without changing the worker gate. */
|
|
31669
32498
|
function resolveModeDefaults(mode, ignoreSessionDefaults = false) {
|
|
31670
32499
|
const builtIn = BUILT_IN_MODE_DEFAULTS[mode] ?? {
|
|
31671
|
-
model:
|
|
32500
|
+
model: resolveDefaultModel(),
|
|
31672
32501
|
thinking: DEFAULT_THINKING
|
|
31673
32502
|
};
|
|
31674
32503
|
const override = ignoreSessionDefaults ? {} : getWorkerSessionDefault(mode);
|
|
@@ -33720,7 +34549,7 @@ const PERSONAS_READ = Object.freeze([
|
|
|
33720
34549
|
toolNameHttp: "codex_reviewer",
|
|
33721
34550
|
model: "gpt-5.3-codex",
|
|
33722
34551
|
endpoint: "/v1/responses",
|
|
33723
|
-
description: "Line-level code reviewer backed by gpt-5.3-codex (OpenAI, ≈272K-token input window), a code
|
|
34552
|
+
description: "Line-level code reviewer backed by gpt-5.3-codex (OpenAI, ≈272K-token input window), a code specialist for line-level review. It reviews concrete diffs, files, or function bodies and returns findings with severity, file:line locations, issue impact, and a minimal suggested fix. Use when the artifact is actual code and the goal is bug, edge-case, security, concurrency, resource, or idiom review. Not for architecture or tradeoff review, use codex_critic or gemini_critic; pass the diff or file content verbatim.",
|
|
33724
34553
|
baseInstructions: REVIEWER_BASE,
|
|
33725
34554
|
agentPrompt: "",
|
|
33726
34555
|
writeCapable: false,
|
|
@@ -33912,13 +34741,8 @@ function buildPeerAwarenessSnippet(opts) {
|
|
|
33912
34741
|
const codexCliClause = opts.codexCli ? " `mcp__codex-cli__codex` dispatches to `codex-implementer` (gpt-5.3-codex with workspace-write) for end-to-end coding tasks." : "";
|
|
33913
34742
|
const para2Parts = [`\`mcp__${searchKey}__code\` is the one-stop code search (no extra model call). Its DEFAULT mode (or \`mode:"semantic"\`) ranks by MEANING via ColBERT over a per-workspace index, the first thing to reach for on intent/concept questions ("where is retry/backoff handled", "how does auth work"); when that index isn't ready it transparently falls back to lexical (the response \`source\` says which engine ran). Forced modes cover the rest: \`lexical\` (BM25F-ranked + tree-sitter, best for exact symbols), \`exact\`, \`regex\`, \`complete\` (exhaustive set), \`ast_pattern\`+\`ast_lang\` for multi-line AST shapes, \`scan\` for a whole-workspace symbol outline, \`multiline\` for cross-line regex. Multiple queries can run in a single turn. The index covers code-shaped files; for unstructured files (logs, \`.csv\`, \`.env*\`, config-only wiring), \`grep\`/\`glob\` still apply.`];
|
|
33914
34743
|
if (opts.workerToolsAvailable) para2Parts.push(`\`worker-*\` are background Agent subagents (subagent_type) that run the matching worker in its own context and deliver the result as a completion notification, so a long run never blocks the turn: \`worker-explore\` (read-only research), \`worker-review\` (reads the code to verify a change or claim), \`worker-plan\` (ordered implementation plan), \`worker-implement\` (edit/write/bash; ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file; for in-place edits use the \`implementer\` subagent), \`worker-test\` (independent test author; also always worktree-isolated)${opts.browseAvailable ? ", `worker-browse` (autonomous browser agent driving a real browser)" : ""}. The raw \`mcp__${workersKey}__*\` tools they call are guarded (a direct main-thread call is redirected to the matching agent); Workers themselves have \`code_search\`.`);
|
|
33915
|
-
const
|
|
33916
|
-
|
|
33917
|
-
opts.genericFastAvailable === false ? void 0 : "`generic-fast`",
|
|
33918
|
-
opts.genericCheapAvailable === false ? void 0 : "`generic-cheap`"
|
|
33919
|
-
].filter((n) => n != null);
|
|
33920
|
-
const catchAllClause = catchAllNames.length > 0 ? ` Catch-alls on non-lead models, for work no specialist fits, cheapest last: ${catchAllNames.join(", ")}.` : "";
|
|
33921
|
-
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (you know what to build), \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
34744
|
+
const catchAllClause = opts.generalPurposeFastAvailable === false ? "" : " Catch-all on a fast, economical non-lead model for work no specialist fits: `general-purpose-fast`.";
|
|
34745
|
+
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
33922
34746
|
if (opts.workerToolsAvailable) para2Parts.push(`For a bounded, well-scoped implementation, prefer the \`implementer\` subagent over \`worker-implement\`; reach for \`worker-implement\` only when you specifically need git-worktree isolation, parallel variants, or a throwaway experiment.`);
|
|
33923
34747
|
if (opts.workerToolsAvailable) para2Parts.push(`\`mcp__${orchestrateKey}__decompose\` composes an open-ended ask into a typed, VERIFIED workflow IR (a strong driver decorrelated by a cross-lab critic, so the decompose step isn't a single point of failure), and \`mcp__${orchestrateKey}__run_workflow\` executes that IR through a frozen kernel delivering max(orchestrated, baseline) over a sealed executable gate, so it never ships worse than a plain single-model run. \`mcp__${orchestrateKey}__verify_workflow\` checks an IR's floor invariants before you run it, and \`mcp__${orchestrateKey}__attest_step\` audits that a finished run's producers were each checked by a different lab. They suit non-trivial, role-separated asks; a trivial ask does not need them.`);
|
|
33924
34748
|
else para2Parts.push(`\`mcp__${orchestrateKey}__verify_workflow\` statically checks a workflow IR's floor invariants and \`mcp__${orchestrateKey}__attest_step\` audits a run's cross-lab lineage (the \`decompose\`/\`run_workflow\` composer + kernel need the worker backend, unavailable here).`);
|
|
@@ -33951,15 +34775,27 @@ function buildPeerAwarenessSnippet(opts) {
|
|
|
33951
34775
|
*/
|
|
33952
34776
|
function buildPeerAwarenessSummary(opts) {
|
|
33953
34777
|
const key = (g) => opts.groupKeys?.[g] ?? GROUP_META[g].preferredKey;
|
|
33954
|
-
const
|
|
33955
|
-
|
|
33956
|
-
|
|
33957
|
-
|
|
34778
|
+
const renderNative = (name) => {
|
|
34779
|
+
const modelId = opts.nativeAgentModels?.[name];
|
|
34780
|
+
if (!modelId) return `\`${name}\``;
|
|
34781
|
+
const prices = catalogTokenPrices(modelId);
|
|
34782
|
+
const tps = indicativeTokensPerSecond(modelId);
|
|
34783
|
+
if (!prices || tps == null) return `\`${name}\``;
|
|
34784
|
+
return `\`${name}\` ${prices.in}/${prices.out} ~${tps}t/s`;
|
|
34785
|
+
};
|
|
34786
|
+
const summaryNativeNames = [
|
|
34787
|
+
renderNative("implementer"),
|
|
34788
|
+
opts.implementerFastAvailable === false ? void 0 : renderNative("implementer-fast"),
|
|
34789
|
+
renderNative("reviewer"),
|
|
34790
|
+
renderNative("brainstorm"),
|
|
34791
|
+
opts.scoutAvailable === false ? void 0 : renderNative("scout"),
|
|
34792
|
+
renderNative("scribe"),
|
|
34793
|
+
opts.generalPurposeFastAvailable === false ? void 0 : renderNative("general-purpose-fast")
|
|
33958
34794
|
].filter((n) => n != null);
|
|
33959
34795
|
const lines = [
|
|
33960
34796
|
"## Injected capabilities (summary)",
|
|
33961
34797
|
"",
|
|
33962
|
-
|
|
34798
|
+
`${summaryNativeNames.some((n) => /\d/.test(n)) ? "Native subagents (Task), own context. Cost is per 1M tokens in/out, tok/s approximate:" : "Native subagents (Task), each in its own context:"} ${summaryNativeNames.join(", ")}. Each agent's own description states when it applies. They read the repo and can run things; the peer critics below cannot, so reach for \`reviewer\` when an assessment needs execution or repo context and for a critic when you already hold the artifact.`,
|
|
33963
34799
|
`A layer of MCP tools, background workers, and skills is injected into this session. Cross-lab peer critics under \`mcp__${key("peers")}__*\` (plus the \`peer-review-coordinator\` subagent) review plans and diffs adversarially, and Claude Code's built-in \`advisor\` catches approach drift. \`mcp__${key("search")}__code\` is meaning-first code search and \`mcp__${key("search")}__web\` returns citable web sources.`
|
|
33964
34800
|
];
|
|
33965
34801
|
if (opts.workerToolsAvailable) lines.push(`Background \`worker-*\` agents (explore, review, plan, implement, test${opts.browseAvailable ? ", browse" : ""}) run delegated work in their own context without blocking your turn, and \`mcp__${key("orchestrate")}__*\` composes, verifies, and runs floor-raising workflows.`);
|
|
@@ -34262,7 +35098,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34262
35098
|
toolNameHttp: "explore",
|
|
34263
35099
|
group: "workers",
|
|
34264
35100
|
capability: "worker",
|
|
34265
|
-
description: "Runs as the background `worker-explore` agent. Dispatch via the Agent tool (subagent_type: worker-explore) so the turn is never blocked; the result arrives as a completion notification. Read-only investigation by an autonomous worker (Pi runtime; default model `
|
|
35101
|
+
description: "Runs as the background `worker-explore` agent. Dispatch via the Agent tool (subagent_type: worker-explore) so the turn is never blocked; the result arrives as a completion notification. Read-only investigation by an autonomous worker (Pi runtime; default model `gpt-5.6-luna` at high reasoning, override via the `model` arg with any Copilot-catalog model that advertises `tool_calls`). It has read, glob, grep, semantic-first code search, web search, fetch_url, advisor, update_plan, and read-only toolbelt tools, and it returns a single text answer. Use for bounded research, repo discovery, dependency investigation, or multi-file reading that would otherwise consume the lead context window. Not for implementation, test authoring, or verification of a concrete diff; use implement, test, or review for those scopes. Brief the investigation goal and constraints, not step-by-step tool semantics. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34266
35102
|
inputSchema: {
|
|
34267
35103
|
type: "object",
|
|
34268
35104
|
required: ["prompt"],
|
|
@@ -34274,7 +35110,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34274
35110
|
},
|
|
34275
35111
|
model: {
|
|
34276
35112
|
type: "string",
|
|
34277
|
-
description: "Optional Copilot catalog model id (defaults to
|
|
35113
|
+
description: "Optional Copilot catalog model id (defaults to gpt-5.6-luna). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
|
|
34278
35114
|
},
|
|
34279
35115
|
thinking: {
|
|
34280
35116
|
type: "string",
|
|
@@ -34283,7 +35119,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34283
35119
|
},
|
|
34284
35120
|
workspace: {
|
|
34285
35121
|
type: "string",
|
|
34286
|
-
description: "
|
|
35122
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34287
35123
|
},
|
|
34288
35124
|
maxWallClockMs: {
|
|
34289
35125
|
type: "integer",
|
|
@@ -34291,11 +35127,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34291
35127
|
}
|
|
34292
35128
|
}
|
|
34293
35129
|
},
|
|
34294
|
-
async handler(args, signal) {
|
|
35130
|
+
async handler(args, signal, ctx) {
|
|
34295
35131
|
return runWorkerToolCall({
|
|
34296
35132
|
mode: "explore",
|
|
34297
35133
|
args,
|
|
34298
|
-
signal
|
|
35134
|
+
signal,
|
|
35135
|
+
ctx
|
|
34299
35136
|
});
|
|
34300
35137
|
}
|
|
34301
35138
|
},
|
|
@@ -34303,7 +35140,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34303
35140
|
toolNameHttp: "implement",
|
|
34304
35141
|
group: "workers",
|
|
34305
35142
|
capability: "worker",
|
|
34306
|
-
description: "Runs as the background `worker-implement` agent. Dispatch via the Agent tool (subagent_type: worker-implement) so the turn is never blocked; the result arrives as a completion notification. Delegates a scoped coding task to an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at
|
|
35143
|
+
description: "Runs as the background `worker-implement` agent. Dispatch via the Agent tool (subagent_type: worker-implement) so the turn is never blocked; the result arrives as a completion notification. Delegates a scoped coding task to an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the explore read-only tools plus edit, write, bash, and codex_review, and it returns its final text with any changed files or worktree diff. Use for bounded implementation work that may take a while or benefits from isolated worker context. Not for pure research, planning, review, or independent test authoring; use explore, plan, review, or test for those scopes. ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place edits, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34307
35144
|
inputSchema: {
|
|
34308
35145
|
type: "object",
|
|
34309
35146
|
required: ["prompt"],
|
|
@@ -34319,7 +35156,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34319
35156
|
},
|
|
34320
35157
|
model: {
|
|
34321
35158
|
type: "string",
|
|
34322
|
-
description: "Optional Copilot catalog model id (defaults to gpt-5.6-sol). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `
|
|
35159
|
+
description: "Optional Copilot catalog model id (defaults to gpt-5.6-sol). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
|
|
34323
35160
|
},
|
|
34324
35161
|
thinking: {
|
|
34325
35162
|
type: "string",
|
|
@@ -34328,7 +35165,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34328
35165
|
},
|
|
34329
35166
|
workspace: {
|
|
34330
35167
|
type: "string",
|
|
34331
|
-
description: "
|
|
35168
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (implement always runs in a worktree)."
|
|
34332
35169
|
},
|
|
34333
35170
|
maxWallClockMs: {
|
|
34334
35171
|
type: "integer",
|
|
@@ -34336,11 +35173,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34336
35173
|
}
|
|
34337
35174
|
}
|
|
34338
35175
|
},
|
|
34339
|
-
async handler(args, signal) {
|
|
35176
|
+
async handler(args, signal, ctx) {
|
|
34340
35177
|
return runWorkerToolCall({
|
|
34341
35178
|
mode: "implement",
|
|
34342
35179
|
args,
|
|
34343
|
-
signal
|
|
35180
|
+
signal,
|
|
35181
|
+
ctx
|
|
34344
35182
|
});
|
|
34345
35183
|
}
|
|
34346
35184
|
},
|
|
@@ -34348,7 +35186,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34348
35186
|
toolNameHttp: "review",
|
|
34349
35187
|
group: "workers",
|
|
34350
35188
|
capability: "worker",
|
|
34351
|
-
description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification.
|
|
35189
|
+
description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification. Code review by an autonomous worker (Pi runtime; default model `gemini-3.1-pro-preview`, default thinking xhigh clamped to high for that model, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has explore's read-and-search tools PLUS `bash`, so it verifies claims by running them — reproducing a failure or running the build or suite — rather than only reading, and returns severity-ranked findings with `file:line` citations. It gets no edit/write tools unless you pass `worktree: true`, but `bash` runs real commands in the workspace, so a build or test it invokes can touch the tree. Use for reviewing a change, diff, or correctness claim when the reviewer should read the code itself rather than trusting a pasted artifact. Not for architecture critique, implementation, or test authoring; use codex_critic or gemini_critic for design review, implement for edits, and test for independent test creation. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34352
35190
|
inputSchema: {
|
|
34353
35191
|
type: "object",
|
|
34354
35192
|
required: ["prompt"],
|
|
@@ -34360,16 +35198,20 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34360
35198
|
},
|
|
34361
35199
|
model: {
|
|
34362
35200
|
type: "string",
|
|
34363
|
-
description: "Optional Copilot catalog model id (defaults to gemini-3.1-pro-preview). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `
|
|
35201
|
+
description: "Optional Copilot catalog model id (defaults to gemini-3.1-pro-preview). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
|
|
34364
35202
|
},
|
|
34365
35203
|
thinking: {
|
|
34366
35204
|
type: "string",
|
|
34367
35205
|
enum: WORKER_THINKING_LEVELS,
|
|
34368
35206
|
description: "Optional reasoning depth (defaults to xhigh, clamped to high for the default review model). Silently clamped to the model's allowed range; \"off\" drops the parameter entirely."
|
|
34369
35207
|
},
|
|
35208
|
+
worktree: {
|
|
35209
|
+
type: "boolean",
|
|
35210
|
+
description: "Optional. When true, the review runs in an isolated git worktree replaying the workspace's working tree (dirty tracked changes and untracked-not-ignored files), which additionally grants `edit`/`write` so the reviewer can author a throwaway probe test to prove a claim. Default false: the reviewer reads and runs commands in the workspace itself. Prefer the default when verifying needs the build to work — a fresh worktree does not carry IGNORED files, so installed dependencies are absent. Requires a git repository."
|
|
35211
|
+
},
|
|
34370
35212
|
workspace: {
|
|
34371
35213
|
type: "string",
|
|
34372
|
-
description: "
|
|
35214
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34373
35215
|
},
|
|
34374
35216
|
maxWallClockMs: {
|
|
34375
35217
|
type: "integer",
|
|
@@ -34377,11 +35219,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34377
35219
|
}
|
|
34378
35220
|
}
|
|
34379
35221
|
},
|
|
34380
|
-
async handler(args, signal) {
|
|
35222
|
+
async handler(args, signal, ctx) {
|
|
34381
35223
|
return runWorkerToolCall({
|
|
34382
35224
|
mode: "review",
|
|
34383
35225
|
args,
|
|
34384
|
-
signal
|
|
35226
|
+
signal,
|
|
35227
|
+
ctx
|
|
34385
35228
|
});
|
|
34386
35229
|
}
|
|
34387
35230
|
},
|
|
@@ -34389,7 +35232,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34389
35232
|
toolNameHttp: "plan",
|
|
34390
35233
|
group: "workers",
|
|
34391
35234
|
capability: "worker",
|
|
34392
|
-
description: "Runs as the background `worker-plan` agent. Dispatch via the Agent tool (subagent_type: worker-plan) so the turn is never blocked; the result arrives as a completion notification. Read-only implementation planning by an autonomous worker (Pi runtime; default model `claude-opus-5` at
|
|
35235
|
+
description: "Runs as the background `worker-plan` agent. Dispatch via the Agent tool (subagent_type: worker-plan) so the turn is never blocked; the result arrives as a completion notification. Read-only implementation planning by an autonomous worker (Pi runtime; default model `claude-opus-5` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read-only toolset as explore and returns a concrete, ordered implementation plan covering files, approach, risks, and how acceptance criteria will be verified. Use before coding when the task needs repo-grounded sequencing or acceptance criteria translated into implementation steps. Not for editing files, running an implementation, writing tests, or adversarial review; use implement, test, or review for those scopes. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34393
35236
|
inputSchema: {
|
|
34394
35237
|
type: "object",
|
|
34395
35238
|
required: ["prompt"],
|
|
@@ -34410,7 +35253,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34410
35253
|
},
|
|
34411
35254
|
workspace: {
|
|
34412
35255
|
type: "string",
|
|
34413
|
-
description: "
|
|
35256
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
|
|
34414
35257
|
},
|
|
34415
35258
|
maxWallClockMs: {
|
|
34416
35259
|
type: "integer",
|
|
@@ -34418,11 +35261,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34418
35261
|
}
|
|
34419
35262
|
}
|
|
34420
35263
|
},
|
|
34421
|
-
async handler(args, signal) {
|
|
35264
|
+
async handler(args, signal, ctx) {
|
|
34422
35265
|
return runWorkerToolCall({
|
|
34423
35266
|
mode: "plan",
|
|
34424
35267
|
args,
|
|
34425
|
-
signal
|
|
35268
|
+
signal,
|
|
35269
|
+
ctx
|
|
34426
35270
|
});
|
|
34427
35271
|
}
|
|
34428
35272
|
},
|
|
@@ -34430,7 +35274,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34430
35274
|
toolNameHttp: "test",
|
|
34431
35275
|
group: "workers",
|
|
34432
35276
|
capability: "worker",
|
|
34433
|
-
description: "Runs as the background `worker-test` agent. Dispatch via the Agent tool (subagent_type: worker-test) so the turn is never blocked; the result arrives as a completion notification. Independent adversarial test authoring by an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at
|
|
35277
|
+
description: "Runs as the background `worker-test` agent. Dispatch via the Agent tool (subagent_type: worker-test) so the turn is never blocked; the result arrives as a completion notification. Independent adversarial test authoring by an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read/write toolset as implement and writes tests that try to break the implementation through edge cases, error paths, and acceptance criteria, then runs them and reports pass/fail. Use when a separate test author should challenge an implementation without modifying the production code to make tests pass. Not for implementing fixes, broad research, or code review; use implement, explore, or review for those scopes. ALWAYS runs in an isolated git worktree and returns the test diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place test authoring, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34434
35278
|
inputSchema: {
|
|
34435
35279
|
type: "object",
|
|
34436
35280
|
required: ["prompt"],
|
|
@@ -34455,7 +35299,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34455
35299
|
},
|
|
34456
35300
|
workspace: {
|
|
34457
35301
|
type: "string",
|
|
34458
|
-
description: "
|
|
35302
|
+
description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (test always runs in a worktree)."
|
|
34459
35303
|
},
|
|
34460
35304
|
maxWallClockMs: {
|
|
34461
35305
|
type: "integer",
|
|
@@ -34463,11 +35307,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34463
35307
|
}
|
|
34464
35308
|
}
|
|
34465
35309
|
},
|
|
34466
|
-
async handler(args, signal) {
|
|
35310
|
+
async handler(args, signal, ctx) {
|
|
34467
35311
|
return runWorkerToolCall({
|
|
34468
35312
|
mode: "test",
|
|
34469
35313
|
args,
|
|
34470
|
-
signal
|
|
35314
|
+
signal,
|
|
35315
|
+
ctx
|
|
34471
35316
|
});
|
|
34472
35317
|
}
|
|
34473
35318
|
},
|
|
@@ -34677,7 +35522,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
|
|
|
34677
35522
|
toolNameHttp: "browse",
|
|
34678
35523
|
group: "workers",
|
|
34679
35524
|
capability: "browse_agent",
|
|
34680
|
-
description: "Runs as the background `worker-browse` agent. Dispatch via the Agent tool (subagent_type: worker-browse) so the turn is never blocked; the result arrives as a completion notification. A Pi-driven autonomous browser worker (default model `gpt-5.
|
|
35525
|
+
description: "Runs as the background `worker-browse` agent. Dispatch via the Agent tool (subagent_type: worker-browse) so the turn is never blocked; the result arrives as a completion notification. A Pi-driven autonomous browser worker (default model `gpt-5.6-luna`) drives a real browser to accomplish `task`, keeps raw DOM and page snapshots inside its own context, and returns a single text result. Use for delegated multi-step web tasks such as comparing prices, logging into a dashboard, or summarizing pages when the lead does not need to steer each click. Not for direct in-context browser control, screenshots, or precise element interactions; use the `browser` MCP tools for those. Pass `sessionId` to continue a prior browse session, or omit it for a fresh isolated session; multiple calls run as parallel sessions on the shared browser. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
|
|
34681
35526
|
inputSchema: {
|
|
34682
35527
|
type: "object",
|
|
34683
35528
|
required: ["task"],
|
|
@@ -34790,10 +35635,12 @@ function assertMcpToolSurfaceConsistent() {
|
|
|
34790
35635
|
/**
|
|
34791
35636
|
* Shared closure body for the two worker MCP tools. Validates the
|
|
34792
35637
|
* minimal arg shape (prompt required + optional knobs typed), then
|
|
34793
|
-
* forwards to `runWorkerAgent`.
|
|
34794
|
-
*
|
|
34795
|
-
*
|
|
34796
|
-
*
|
|
35638
|
+
* forwards to `runWorkerAgent`. `workspace` comes from the caller's
|
|
35639
|
+
* argument or, failing that, the per-connection session header the
|
|
35640
|
+
* boundary folds in; with neither, the call is REFUSED rather than
|
|
35641
|
+
* defaulted to the proxy's launch cwd (see the resolution block below
|
|
35642
|
+
* for why that default was a bug, and `GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE`
|
|
35643
|
+
* for the raw-client escape hatch). The engine performs every
|
|
34797
35644
|
* deeper validation (model existence, thinking clamp, worktree
|
|
34798
35645
|
* provisioning, semaphore acquisition, workspace realpath +
|
|
34799
35646
|
* accessibility) and never throws — its `{text, isError?}` envelope
|
|
@@ -34806,7 +35653,7 @@ function assertMcpToolSurfaceConsistent() {
|
|
|
34806
35653
|
* client that ignores the schema.
|
|
34807
35654
|
*/
|
|
34808
35655
|
async function runWorkerToolCall(call) {
|
|
34809
|
-
const { mode, args, signal } = call;
|
|
35656
|
+
const { mode, args, signal, ctx } = call;
|
|
34810
35657
|
const prompt = typeof args.prompt === "string" ? args.prompt : "";
|
|
34811
35658
|
if (!prompt) return {
|
|
34812
35659
|
content: [{
|
|
@@ -34838,7 +35685,7 @@ async function runWorkerToolCall(call) {
|
|
|
34838
35685
|
}
|
|
34839
35686
|
let worktree;
|
|
34840
35687
|
let worktreeNote = "";
|
|
34841
|
-
if (mode === "implement" || mode === "test") {
|
|
35688
|
+
if (mode === "implement" || mode === "test" || mode === "review") {
|
|
34842
35689
|
if (args.worktree !== void 0 && typeof args.worktree !== "boolean") return {
|
|
34843
35690
|
content: [{
|
|
34844
35691
|
type: "text",
|
|
@@ -34846,11 +35693,16 @@ async function runWorkerToolCall(call) {
|
|
|
34846
35693
|
}],
|
|
34847
35694
|
isError: true
|
|
34848
35695
|
};
|
|
34849
|
-
worktree = true;
|
|
34850
|
-
|
|
35696
|
+
if (mode === "review") worktree = args.worktree === true ? true : void 0;
|
|
35697
|
+
else {
|
|
35698
|
+
worktree = true;
|
|
35699
|
+
if (args.worktree === false) worktreeNote = `[note: worker_${mode} always runs in an isolated git worktree; the requested worktree:false was overridden. For in-place edits, use the \`implementer\` subagent.]
|
|
34851
35700
|
|
|
34852
35701
|
`;
|
|
35702
|
+
}
|
|
34853
35703
|
}
|
|
35704
|
+
const allowProxyCwd = process.env.GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE === "1";
|
|
35705
|
+
const callerGaveWorkspace = args.workspace !== void 0;
|
|
34854
35706
|
let workspace;
|
|
34855
35707
|
if (args.workspace !== void 0) {
|
|
34856
35708
|
if (typeof args.workspace !== "string" || args.workspace.length === 0) return {
|
|
@@ -34875,7 +35727,16 @@ async function runWorkerToolCall(call) {
|
|
|
34875
35727
|
}],
|
|
34876
35728
|
isError: true
|
|
34877
35729
|
};
|
|
34878
|
-
else workspace = process.cwd();
|
|
35730
|
+
else if (allowProxyCwd) workspace = process.cwd();
|
|
35731
|
+
else return {
|
|
35732
|
+
content: [{
|
|
35733
|
+
type: "text",
|
|
35734
|
+
text: `worker_${mode}: a workspace is required. Nothing in this call said which directory to run in, and the proxy's own launch directory is not a safe guess — it is where the proxy was started, which may be a different checkout or git worktree than the one you are working in. Re-issue this call with \`workspace\` set to the absolute path of your current working directory. (Operators running a raw MCP client that cannot send one can restore the old launch-cwd default with GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE=1.)`
|
|
35735
|
+
}],
|
|
35736
|
+
isError: true
|
|
35737
|
+
};
|
|
35738
|
+
const effectiveSource = ctx?.workspaceSource ?? (callerGaveWorkspace ? "argument" : "absent");
|
|
35739
|
+
const workspaceNote = effectiveSource === "argument" ? "" : `[workspace: ${workspace} (${effectiveSource === "session" ? "from your session's working directory; pass `workspace` explicitly if you are running somewhere else, such as a git worktree" : "the proxy's launch directory"})]\n\n`;
|
|
34879
35740
|
let maxWallClockMs;
|
|
34880
35741
|
let clampNote = "";
|
|
34881
35742
|
if (args.maxWallClockMs !== void 0) {
|
|
@@ -34903,7 +35764,7 @@ async function runWorkerToolCall(call) {
|
|
|
34903
35764
|
maxWallClockMs,
|
|
34904
35765
|
signal
|
|
34905
35766
|
});
|
|
34906
|
-
const notePrefix = `${clampNote}${worktreeNote}`;
|
|
35767
|
+
const notePrefix = `${workspaceNote}${clampNote}${worktreeNote}`;
|
|
34907
35768
|
return {
|
|
34908
35769
|
content: [{
|
|
34909
35770
|
type: "text",
|
|
@@ -35136,6 +35997,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
|
|
|
35136
35997
|
return [...new Set(names)];
|
|
35137
35998
|
}
|
|
35138
35999
|
//#endregion
|
|
35139
|
-
export {
|
|
36000
|
+
export { handleMcpDelete as $, resolveLeadSlugArg as $t, satisfiesMinVersion as A, hasSupportedBrowserInstalled as At, rememberThinkingHistoryRepair as B, collapsePathKeys as Bt, availableToolCommands as C, resolveMcpToolTimeoutMs as Ct, vscodeRipgrepPath as D, readResponseBodyCapped as Dt, toolbeltSkipSet as E, MAX_RESPONSE_BODY_BYTES as Et, injectAdvisorTool as F, warmTreeSitterPool as Ft, isControllerClosedError as G, DEFAULT_CODEX_MODEL as Gt, repairRejectedThinkingHistory as H, BUDGET_SMALL_FAST_CATALOG_ID as Ht, isAdvisorRequested as I, provisionTreeSitterAssets as It, relayAnthropicStream as J, UPSTREAM_FETCH_TIMEOUT_MS as Jt, logStreamError as K, DEFAULT_CODEX_MODEL_FALLBACKS as Kt, resolveAdvisorEffort as L, CONDENSED_OPERATING_SEQUENCE as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, provisionAndIndexColbert as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, extractTarGzMember as Nt, TOOLBELT_TOOLS$1 as O, parseJsonOrDiagnose as Ot, buildAdvisorStream as P, extractZipMember as Pt, clampEffort as Q, pickClaudeDefault as Qt, resolveAdvisorModel as R, DEFINITION_OF_GREATNESS as Rt, buildEnv as S, warnOnTokenPriceDrift as St, toolbeltEnabled as T, createChatCompletions as Tt, buildAnthropicErrorEvent as U, BUDGET_SMALL_FAST_SLUG as Ut, repairKnownThinkingHistory as V, toolbeltPathOverride as Vt, buildOpenAIErrorEvent as W, DEFAULT_CLAUDE_MODEL_FALLBACKS as Wt, UNKNOWN_EFFORT_ANCHOR as X, generateRandomPort as Xt, EFFORT_ORDER as Y, UPSTREAM_INACTIVITY_TIMEOUT_MS as Yt, bucketEffort as Z, isBudgetClaudeLead as Zt, appendPlanReminder as _, shimDefaultsToXhigh as _t, buildPeerAwarenessSnippet as a, withInstallLock as an, browserCompoundToolsEnabled as at, resolveWorkerRunOpts as b, getTokenCount as bt, personasFor as c, geminiAvailable as ct, EXPLORE_DEFAULT_MODEL as d, nativeSubagentModel as dt, upstreamAllowH2 as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, reviewerModel as ft, TEST_DEFAULT_MODEL as g, workerToolsEnabled as gt, REVIEW_DEFAULT_MODEL as h, standInToolEnabled as ht, buildAgentPrompt as i, withOneMSuffixForLead as in, browseAgentEnabled as it, searchWeb as j, colbertDegradedWarning as jt, assetFor as k, provisionBrowserAssets as kt, BROWSE_DEFAULT_MODEL as l, generalPurposeFastModel as lt, PLAN_DEFAULT_MODEL as m, scribeModel as mt, MCP_GROUPS as n, classifyMessagesRoute as nn, artifactToolsEnabled as nt, buildPeerAwarenessSummary as o, browserToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, scoutModel as pt, readIteratorWithTimeout as q, DEFAULT_PORT as qt, assertMcpToolSurfaceConsistent as r, withOneMSuffix as rn, brainstormModel as rt, enumerateInjectedMcpToolNames as s, fleetToolsEnabled as st, GROUP_META as t, upstreamMaxConnections as tn, agentToolsEnabled as tt, DEFAULT_MODEL_CHAIN as u, implementerFastModel as ut, resolveDefaultModel as v, countTokens as vt, buildToolbeltAwareness as w, createResponses as wt, runWorkerAgent as x, assembleResponsesPayload as xt, resolveModeDefaults as y, createMessages as yt, formatThinkingRepairDecline as z, shouldUseInsecureTls as zt };
|
|
35140
36001
|
|
|
35141
|
-
//# sourceMappingURL=peer-mcp-personas-
|
|
36002
|
+
//# sourceMappingURL=peer-mcp-personas-V6stFvpq.js.map
|