github-router 0.3.276 → 0.3.285

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/dist/{attribution-settings-VOPvSn6v.js → attribution-settings-B8M2fvhz.js} +44 -45
  2. package/dist/attribution-settings-B8M2fvhz.js.map +1 -0
  3. package/dist/{auth-CYoRwhC9.js → auth-VUL2Zxvw.js} +3 -3
  4. package/dist/{auth-CYoRwhC9.js.map → auth-VUL2Zxvw.js.map} +1 -1
  5. package/dist/browser-ext/manifest.json +1 -1
  6. package/dist/{check-usage-DDUyg8ZB.js → check-usage-CcLdFPGr.js} +4 -4
  7. package/dist/{check-usage-DDUyg8ZB.js.map → check-usage-CcLdFPGr.js.map} +1 -1
  8. package/dist/{claude-OTHh7iKz.js → claude-e-JIQGhR.js} +55 -38
  9. package/dist/claude-e-JIQGhR.js.map +1 -0
  10. package/dist/{codex-tCxK30Su.js → codex-DSnVZ9oK.js} +5 -5
  11. package/dist/{codex-tCxK30Su.js.map → codex-DSnVZ9oK.js.map} +1 -1
  12. package/dist/{debug-CJgsxoVw.js → debug-CtzYWxpJ.js} +2 -2
  13. package/dist/{debug-CJgsxoVw.js.map → debug-CtzYWxpJ.js.map} +1 -1
  14. package/dist/engine-DMveEa9x.js +2 -0
  15. package/dist/{gate-discovery-nf7PGgnS.js → gate-discovery-BCFwLm0q.js} +5 -5
  16. package/dist/{gate-discovery-nf7PGgnS.js.map → gate-discovery-BCFwLm0q.js.map} +1 -1
  17. package/dist/{get-copilot-usage-D1OAH0EU.js → get-copilot-usage-B-EDAqQb.js} +2 -2
  18. package/dist/{get-copilot-usage-D1OAH0EU.js.map → get-copilot-usage-B-EDAqQb.js.map} +1 -1
  19. package/dist/hooks.mjs +37870 -0
  20. package/dist/hooks.sha256 +1 -0
  21. package/dist/{internal-artifact-open-BCz9lnq0.js → internal-artifact-open-Dj5Nlq0L.js} +2 -2
  22. package/dist/{internal-artifact-open-BCz9lnq0.js.map → internal-artifact-open-Dj5Nlq0L.js.map} +1 -1
  23. package/dist/{internal-first-mate-guard-7v2CtMA6.js → internal-first-mate-guard-B7p4NttK.js} +1 -1
  24. package/dist/{internal-first-mate-guard-La0tIjTU.js → internal-first-mate-guard-DOgFVki5.js} +5 -4
  25. package/dist/{internal-first-mate-guard-La0tIjTU.js.map → internal-first-mate-guard-DOgFVki5.js.map} +1 -1
  26. package/dist/{internal-plan-review-jRMpHu68.js → internal-plan-review-DRFHET_B.js} +11 -4
  27. package/dist/internal-plan-review-DRFHET_B.js.map +1 -0
  28. package/dist/{internal-prompt-submit-CqWRFCuv.js → internal-prompt-submit-f1LR2P2G.js} +4 -4
  29. package/dist/{internal-prompt-submit-CqWRFCuv.js.map → internal-prompt-submit-f1LR2P2G.js.map} +1 -1
  30. package/dist/{internal-session-bind-BiHWDxuC.js → internal-session-bind-DhZhxJ3T.js} +2 -2
  31. package/dist/{internal-session-bind-BiHWDxuC.js.map → internal-session-bind-DhZhxJ3T.js.map} +1 -1
  32. package/dist/{internal-stop-hook-Ds9LwmZA.js → internal-stop-hook-Cf-7w3RH.js} +68 -20
  33. package/dist/internal-stop-hook-Cf-7w3RH.js.map +1 -0
  34. package/dist/{internal-stop-review-BOFj7R3q.js → internal-stop-review-BBsLcbPG.js} +2 -2
  35. package/dist/{internal-stop-review-BOFj7R3q.js.map → internal-stop-review-BBsLcbPG.js.map} +1 -1
  36. package/dist/{internal-worker-guard-Lu8VHj5k.js → internal-worker-guard-CKgYFiFO.js} +2 -2
  37. package/dist/{internal-worker-guard-Lu8VHj5k.js.map → internal-worker-guard-CKgYFiFO.js.map} +1 -1
  38. package/dist/{internal-workspace-header-BRMz0Yql.js → internal-workspace-header-OYgHEnFt.js} +2 -2
  39. package/dist/{internal-workspace-header-BRMz0Yql.js.map → internal-workspace-header-OYgHEnFt.js.map} +1 -1
  40. package/dist/lib/tree-sitter-pool/lifecycle-C7W8D2lx.js +165 -0
  41. package/dist/lib/tree-sitter-pool/lifecycle-Dx0SnP7Z.js +157 -0
  42. package/dist/lib/tree-sitter-pool/worker.js +287 -14
  43. package/dist/{lifecycle-LA4rAuAL.js → lifecycle-B7CHqKlF.js} +2 -2
  44. package/dist/{lifecycle-LA4rAuAL.js.map → lifecycle-B7CHqKlF.js.map} +1 -1
  45. package/dist/lifecycle-CbmHMMSD.js +2 -0
  46. package/dist/{lifecycle-AGXnd-ZY.js → lifecycle-D-rL81tT.js} +2 -2
  47. package/dist/{lifecycle-AGXnd-ZY.js.map → lifecycle-D-rL81tT.js.map} +1 -1
  48. package/dist/lifecycle-DTcZwa7U.js +2 -0
  49. package/dist/lifecycle-K9oVtGde.mjs +16 -0
  50. package/dist/main.js +18 -18
  51. package/dist/{mcp-workspace-header-CGJbNeHb.js → mcp-workspace-header-ucs2SDST.js} +5 -6
  52. package/dist/mcp-workspace-header-ucs2SDST.js.map +1 -0
  53. package/dist/{models-4Q45Q2Dw.js → models-C16mBK2M.js} +3 -3
  54. package/dist/{models-4Q45Q2Dw.js.map → models-C16mBK2M.js.map} +1 -1
  55. package/dist/{orchestration-CpkG8ScN.js → orchestration-CtM6FYNx.js} +2 -2
  56. package/dist/{orchestration-CpkG8ScN.js.map → orchestration-CtM6FYNx.js.map} +1 -1
  57. package/dist/package-root-B-osctCk.js +29 -0
  58. package/dist/package-root-B-osctCk.js.map +1 -0
  59. package/dist/paths-CV9K7Xqm.js +2 -0
  60. package/dist/{paths-j2B7b0DZ.js → paths-wLC0InjX.js} +28 -4
  61. package/dist/paths-wLC0InjX.js.map +1 -0
  62. package/dist/{peer-mcp-personas-DRL_xT4h.js → peer-mcp-personas-V6stFvpq.js} +1158 -297
  63. package/dist/peer-mcp-personas-V6stFvpq.js.map +1 -0
  64. package/dist/{plan-review-hook-Dk5zTNke.js → plan-review-hook-Lf9ISdF6.js} +5 -6
  65. package/dist/plan-review-hook-Lf9ISdF6.js.map +1 -0
  66. package/dist/{prompt-submit-hook-lvTWwaTV.js → prompt-submit-hook-BlijaOn7.js} +6 -11
  67. package/dist/prompt-submit-hook-BlijaOn7.js.map +1 -0
  68. package/dist/{provision-HzZ547dW.js → provision-BpL6gZIt.js} +4 -4
  69. package/dist/{provision-HzZ547dW.js.map → provision-BpL6gZIt.js.map} +1 -1
  70. package/dist/self-invocation-CKMjcA5F.js +380 -0
  71. package/dist/self-invocation-CKMjcA5F.js.map +1 -0
  72. package/dist/{serve-CM3OmF4I.js → serve-CXUf7RtJ.js} +34 -27
  73. package/dist/serve-CXUf7RtJ.js.map +1 -0
  74. package/dist/{server-setup-C9r7jYnz.js → server-setup-Bppjt9xO.js} +81 -207
  75. package/dist/server-setup-Bppjt9xO.js.map +1 -0
  76. package/dist/{start-KAUCsDao.js → start-C1-jrHfU.js} +3 -3
  77. package/dist/{start-KAUCsDao.js.map → start-C1-jrHfU.js.map} +1 -1
  78. package/dist/{stop-gate-hook-CaUrldLh.js → stop-gate-hook-DgJ6sW8N.js} +100 -52
  79. package/dist/stop-gate-hook-DgJ6sW8N.js.map +1 -0
  80. package/dist/{stop-gate-policy-lAabj_pP.js → stop-gate-policy-DG5yYWGn.js} +2 -2
  81. package/dist/{stop-gate-policy-lAabj_pP.js.map → stop-gate-policy-DG5yYWGn.js.map} +1 -1
  82. package/dist/{token-CxPaBEUS.js → token-CnlB0884.js} +2 -2
  83. package/dist/token-CnlB0884.js.map +1 -0
  84. package/dist/{version-_Q1WpsQp.js → version-C8x2hQrZ.js} +14 -2
  85. package/dist/version-C8x2hQrZ.js.map +1 -0
  86. package/dist/{worker-dispatch-Bj1uYyG9.js → worker-dispatch-zW8Zi69V.js} +11 -7
  87. package/dist/worker-dispatch-zW8Zi69V.js.map +1 -0
  88. package/package.json +2 -2
  89. package/dist/attribution-settings-VOPvSn6v.js.map +0 -1
  90. package/dist/claude-OTHh7iKz.js.map +0 -1
  91. package/dist/engine-BKJKMAzh.js +0 -2
  92. package/dist/internal-plan-review-jRMpHu68.js.map +0 -1
  93. package/dist/internal-stop-hook-Ds9LwmZA.js.map +0 -1
  94. package/dist/lifecycle-DCrKbaIU.js +0 -2
  95. package/dist/lifecycle-KKSTKwd6.js +0 -2
  96. package/dist/mcp-workspace-header-CGJbNeHb.js.map +0 -1
  97. package/dist/paths-DN3Nio42.js +0 -2
  98. package/dist/paths-j2B7b0DZ.js.map +0 -1
  99. package/dist/peer-mcp-personas-DRL_xT4h.js.map +0 -1
  100. package/dist/plan-review-hook-Dk5zTNke.js.map +0 -1
  101. package/dist/prompt-submit-hook-lvTWwaTV.js.map +0 -1
  102. package/dist/serve-CM3OmF4I.js.map +0 -1
  103. package/dist/server-setup-C9r7jYnz.js.map +0 -1
  104. package/dist/stop-gate-hook-CaUrldLh.js.map +0 -1
  105. package/dist/token-CxPaBEUS.js.map +0 -1
  106. package/dist/version-_Q1WpsQp.js.map +0 -1
  107. package/dist/worker-dispatch-Bj1uYyG9.js.map +0 -1
@@ -1,13 +1,14 @@
1
- import { t as getPackageVersion } from "./version-_Q1WpsQp.js";
2
- import { t as PATHS } from "./paths-j2B7b0DZ.js";
3
- import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-CxPaBEUS.js";
1
+ import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
2
+ import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
3
+ import { t as PATHS } from "./paths-wLC0InjX.js";
4
+ import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-CnlB0884.js";
4
5
  import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
5
- import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-LA4rAuAL.js";
6
- import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-CGJbNeHb.js";
6
+ import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-B7CHqKlF.js";
7
+ import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-ucs2SDST.js";
7
8
  import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
8
- import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-AGXnd-ZY.js";
9
- import { h as runGateChecks, m as sealedGateIds, p as resolveSealedGate } from "./stop-gate-hook-CaUrldLh.js";
10
- import { t as liveExec } from "./orchestration-CpkG8ScN.js";
9
+ import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-D-rL81tT.js";
10
+ import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DgJ6sW8N.js";
11
+ import { t as liveExec } from "./orchestration-CtM6FYNx.js";
11
12
  import { createRequire } from "node:module";
12
13
  import consola from "consola";
13
14
  import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
@@ -162,6 +163,221 @@ function withOneMSuffix(id) {
162
163
  if (oneMContextDisabled()) return id;
163
164
  return catalogAdvertises1M(id) ? `${id}[1m]` : id;
164
165
  }
166
+ /**
167
+ * Decorate a slug the USER named — a `-m` argument or a launcher default — with
168
+ * `[1m]` iff the model it actually RESOLVES to serves >=1M context.
169
+ *
170
+ * The difference from `withOneMSuffix` is the resolution step, and it exists
171
+ * because the two functions are handed different kinds of string. Every
172
+ * `withOneMSuffix` caller already holds a concrete catalog id from a catalog
173
+ * walk, so an exact-id match is both sufficient and the safer rule: inferring a
174
+ * match there could attach `[1m]` to a sibling that does not serve 1M. A lead
175
+ * slug is the opposite case — it is whatever the user typed, or an
176
+ * Anthropic-published dashed slug like `claude-opus-4-8` that the catalog
177
+ * carries in dotted form. Exact-id matching answers "no 1M" for those purely
178
+ * because it never found the entry, which is the silent under-accounting this
179
+ * function exists to stop.
180
+ *
181
+ * Resolving first also picks up the `-1m` SIBLING shape for free:
182
+ * `resolveModel`'s opus family preference maps `claude-opus-4-7` onto
183
+ * `claude-opus-4.7-1m-internal` when that is what the tier carries, and the
184
+ * sibling's own advertised window then answers the question. That is the same
185
+ * dual-signal conclusion `pickClaudeDefault` reaches for the family shorthand,
186
+ * so the two paths cannot disagree about a family both can be asked about.
187
+ *
188
+ * Idempotent: a slug that already carries the bracket is returned unchanged, so
189
+ * a user who pins `-m claude-opus-5[1m]` by hand does not get `[1m][1m]`. That
190
+ * early return deliberately does NOT re-validate the pin against the catalog.
191
+ * `-m claude-haiku-4-5[1m]` therefore survives even though Haiku 4.5 is a 200K
192
+ * model — the same as before this function existed, and `resolveModel` already
193
+ * warns loudly about exactly that case. Stripping a bracket the user typed
194
+ * would be the surprising behaviour, and it would be the only place in the
195
+ * launcher that overrides an explicit `-m`.
196
+ *
197
+ * A repeat can still arrive from the CLIENT side rather than from here: the
198
+ * `/model` picker rows are seeded already decorated, and Claude Code's alias
199
+ * path appends its own bracket (`getDefaultSonnetModel() + '[1m]'`), so
200
+ * selecting `sonnet[1m]` puts `claude-sonnet-5[1m][1m]` on the wire. That
201
+ * resolves to the same bare id — `resolveModel`'s strip recurses — and Claude
202
+ * Code's own detector is unanchored, so local accounting is right too. Pinned
203
+ * by a regression test in `tests/lib-utils.test.ts`.
204
+ *
205
+ * Degrades the same safe direction as everything else here. An unpopulated
206
+ * catalog makes `resolveModel` a pass-through and `catalogAdvertises1M` false,
207
+ * so the slug stays bare and Claude Code accounts at its conservative 200K
208
+ * default — under-accounting, never overflow.
209
+ */
210
+ function withOneMSuffixForLead(slug) {
211
+ if (oneMContextDisabled()) return slug;
212
+ if (/\[1m\]$/i.test(slug)) return slug;
213
+ return catalogAdvertises1M(resolveModel(slug)) ? `${slug}[1m]` : slug;
214
+ }
215
+ //#endregion
216
+ //#region src/services/copilot/endpoint.ts
217
+ /**
218
+ * Catalog spellings that mean each of our two clients. Copilot is not
219
+ * self-consistent about the `/v1` prefix — the live catalog advertises
220
+ * `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
221
+ * fixtures carry both forms — so an exact-match on the bare spelling alone
222
+ * silently misses a real shape. `src/lib/model-validation.ts` already
223
+ * normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
224
+ *
225
+ * Matching is EXACT against this set, never a suffix/`includes` test: a
226
+ * `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
227
+ * client and must keep resolving to "serves neither".
228
+ */
229
+ const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
230
+ const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
231
+ /**
232
+ * Decide which endpoint to call for a model from its catalog
233
+ * `supported_endpoints`. Prefers `/chat/completions` when available (the
234
+ * simpler, more widely-supported shape) and falls back to `/responses` for
235
+ * models that ONLY serve the Responses API — the gpt-5.x family except
236
+ * `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
237
+ * `*-codex` models). Returns undefined when the model serves neither, so a
238
+ * caller can skip it rather than 400 on `unsupported_api_for_model`.
239
+ *
240
+ * A model that OMITS `supported_endpoints` is treated as chat-eligible: the
241
+ * catalog historically omits the field for chat-default models, and
242
+ * excluding those would be a worse regression than the gap this guards.
243
+ */
244
+ function pickEndpoint(model) {
245
+ const eps = model.supported_endpoints;
246
+ if (!eps || eps.length === 0) return "chat";
247
+ if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
248
+ if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
249
+ }
250
+ /**
251
+ * `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
252
+ * "absent from the catalog" into "serves neither of our endpoints".
253
+ *
254
+ * This function deliberately has no default. The predecessor
255
+ * (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
256
+ * coerced both cases to "chat" — defensible for an unknown id, silently wrong
257
+ * for a catalog model serving only, say, `/v1/messages`: the caller would drive
258
+ * it through the chat client and get an opaque upstream 400 with no local
259
+ * signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
260
+ * treats that case correctly (`if (!endpoint) continue`); this makes the same
261
+ * distinction available to callers that resolve by id.
262
+ *
263
+ * Callers that legitimately want the chat default for an unknown id can still
264
+ * have it — they just have to write it, per case, on purpose.
265
+ */
266
+ function resolveEndpointForModelId(id) {
267
+ const found = state.models?.data?.find((m) => m.id === id);
268
+ if (!found) return { kind: "unknown-model" };
269
+ const endpoint = pickEndpoint(found);
270
+ if (endpoint) return {
271
+ kind: "endpoint",
272
+ endpoint
273
+ };
274
+ return {
275
+ kind: "unreachable",
276
+ endpoints: found.supported_endpoints ?? []
277
+ };
278
+ }
279
+ //#endregion
280
+ //#region src/lib/anthropic-translate/classifier.ts
281
+ /**
282
+ * Routing classifier for `POST /v1/messages`.
283
+ *
284
+ * Claude Code speaks the Anthropic Messages wire format. Copilot only serves
285
+ * Claude models on its native `/v1/messages` endpoint; a gpt/gemini request
286
+ * sent there 400s. This classifier decides, from the RESOLVED model id and its
287
+ * catalog metadata, whether a request stays on the native passthrough
288
+ * (`createMessages`) or is diverted to the translation shim.
289
+ *
290
+ * Non-regression is the whole point — every Claude model (opus / sonnet / haiku /
291
+ * any `claude-*` or Anthropic-vendored id) MUST return "claude-passthrough" so
292
+ * its bytes reach `createMessages` unchanged — even if future catalog metadata
293
+ * were to (wrongly) advertise a `/responses` endpoint for it. The Claude check
294
+ * is therefore keyed off identity (id / vendor / family), NOT the endpoint.
295
+ *
296
+ * Non-Claude models are diverted to the translation shim by the endpoint the
297
+ * catalog says they serve: `/responses` models (gpt-5.5, gpt-5.3-codex, …) take
298
+ * the Responses path (`responses-shim`), `/chat/completions` models (gemini,
299
+ * and any chat-default model) take the chat path (`chat-shim`). The decision is
300
+ * derived from `pickEndpoint` (catalog `supported_endpoints`), never a
301
+ * hardcoded slug list, so it generalizes. Copilot only serves Claude models on
302
+ * its native `/v1/messages`, so diverting every non-Claude model to a shim is
303
+ * correct — a non-Claude request sent to `/v1/messages` would 400.
304
+ */
305
+ /**
306
+ * Match "claude"/"anthropic" as a delimiter-bounded segment anywhere in a model
307
+ * id: at the start, or after a `/ _ . : -` path/version delimiter, and followed
308
+ * by end-of-string, another such delimiter, or a digit. This catches catalog
309
+ * aliases like `github/claude-3-7-sonnet` or `anthropic/…` whose vendor is
310
+ * "github" and whose family is empty — where the token only surfaces mid-id —
311
+ * while NOT firing on incidental substrings like `notclaude`. Deliberately
312
+ * over-inclusive at the boundaries: we fail CLOSED toward Claude (→ passthrough)
313
+ * so a real Claude model can never be diverted to the non-Claude shim.
314
+ */
315
+ const CLAUDE_ID_RE = /(^|[/_.:-])(claude|anthropic)(?=$|[/_.:-]|\d)/i;
316
+ /**
317
+ * True when the target is a Claude / Anthropic model. Matches on any of:
318
+ * catalog vendor containing "anthropic", capability family containing "claude",
319
+ * or a Claude/anthropic path-segment (see `CLAUDE_ID_RE`) in ANY id we hold —
320
+ * the resolved id (`modelId`), the pre-resolution request id (`originalModelId`),
321
+ * or the catalog entry's own id (`model.id`). Conservative by design: when in
322
+ * doubt it returns true so a Claude request can never be diverted to the shim.
323
+ */
324
+ function isClaudeModel(modelId, model, originalModelId) {
325
+ if (model) {
326
+ if ((model.vendor?.toLowerCase() ?? "").includes("anthropic")) return true;
327
+ if ((model.capabilities?.family?.toLowerCase() ?? "").includes("claude")) return true;
328
+ }
329
+ return [
330
+ modelId,
331
+ originalModelId,
332
+ model?.id
333
+ ].some((id) => typeof id === "string" && CLAUDE_ID_RE.test(id));
334
+ }
335
+ /**
336
+ * Decide the route for a resolved model id + its catalog entry.
337
+ *
338
+ * - No model id, or a Claude model → "claude-passthrough" (existing behaviour).
339
+ * - A non-Claude model whose catalog endpoint is `/responses` → "responses-shim".
340
+ * - A non-Claude model whose catalog endpoint is `/chat/completions` (gemini and
341
+ * any chat-default model) → "chat-shim".
342
+ * - A non-Claude model absent from the catalog (so we can't confirm an endpoint)
343
+ * → "claude-passthrough" (unchanged; we don't divert what we can't classify).
344
+ * - A non-Claude model that IS in the catalog and serves NEITHER of the two shim
345
+ * endpoints (`pickEndpoint` → undefined) → "claude-passthrough" as well.
346
+ *
347
+ * Those last two land on the same route but are NOT the same answer, and the
348
+ * coincidence is deliberate rather than a collapsed default (contrast
349
+ * `resolveEndpointForModelId`, whose callers must tell them apart because
350
+ * guessing there produces an opaque upstream 400). Here neither shim is even a
351
+ * candidate: a shim can only speak `/responses` or `/chat/completions`, so
352
+ * diverting a model that serves neither would 400 just as surely. Passthrough
353
+ * is the better default because it is sometimes RIGHT — a non-Claude catalog
354
+ * model advertising `/v1/messages` is served by exactly the endpoint
355
+ * passthrough uses. It also preserves this module's fail-CLOSED-toward-Claude
356
+ * invariant: an unclassifiable model is never diverted.
357
+ *
358
+ * KNOWN GAP (audited, deliberately not fixed here): a non-Claude model serving
359
+ * only something we cannot speak at all (say `/embeddings`) also lands on
360
+ * passthrough and will still 400 upstream — `logEndpointMismatch(modelId,
361
+ * "/v1/messages")` logs it at the passthrough seam, but no local error is
362
+ * raised. Closing that needs a change in `src/routes/messages/handler.ts`,
363
+ * which this seam does not own. It is strictly narrower than the defect fixed
364
+ * in `resolveEndpointForModelId`: no such model is reachable as a Claude Code
365
+ * `/v1/messages` target today, whereas the plan worker's `claude-opus-5`
366
+ * default is.
367
+ *
368
+ * `originalModelId` is the optional pre-resolution request id; when supplied it
369
+ * is checked for Claude-likeness alongside the resolved id so an alias that
370
+ * resolves to a non-Claude-looking id can't slip past.
371
+ */
372
+ function classifyMessagesRoute(modelId, model, originalModelId) {
373
+ if (!modelId) return "claude-passthrough";
374
+ if (isClaudeModel(modelId, model, originalModelId)) return "claude-passthrough";
375
+ if (!model) return "claude-passthrough";
376
+ const endpoint = pickEndpoint(model);
377
+ if (endpoint === "responses") return "responses-shim";
378
+ if (endpoint === "chat") return "chat-shim";
379
+ return "claude-passthrough";
380
+ }
165
381
  //#endregion
166
382
  //#region src/lib/port.ts
167
383
  const DEFAULT_PORT = 8787;
@@ -208,11 +424,22 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
208
424
  * This helper detects the catalog state at launch and only opts in
209
425
  * when the backend can actually serve 1M.
210
426
  *
211
- * Sonnet/Haiku families are intentionally NOT given `[1m]` defaults
212
- * because Copilot has no 1M backend for them (and Anthropic-side
213
- * `modelSupports1M` doesn't list haiku at all). See
214
- * `src/lib/server-setup.ts:getClaudeCodeEnvVars` for the
215
- * `ANTHROPIC_DEFAULT_{SONNET,HAIKU,OPUS}_MODEL` tier defaults.
427
+ * This helper answers the question only for the OPUS families, because a
428
+ * family is what it is asked about (`-m 4.7` names no slug). Every other lead
429
+ * slug `-m fast`, a full slug a power user pins, the implicit budget lead —
430
+ * goes through `withOneMSuffixForLead` (`./one-m-context`) instead, which
431
+ * resolves the slug first and then reads the resolved entry's advertised
432
+ * window. The two agree wherever both can be asked: a family that resolves to a
433
+ * 1M backend is 1M by either route.
434
+ *
435
+ * A previous revision of this comment claimed Sonnet and Haiku were left bare
436
+ * because "Copilot has no 1M backend for them". That was true when it was
437
+ * written and is now false for Sonnet: the live catalog advertises
438
+ * `max_context_window_tokens: 1_000_000` on both `claude-sonnet-5` and
439
+ * `claude-sonnet-4.6` (Haiku 4.5 really is 200K, and is left bare by the same
440
+ * catalog check rather than by a hardcoded family rule). Nothing here is
441
+ * family-gated any more — the catalog decides per model, so the next family
442
+ * that ships 1M is picked up without an edit.
216
443
  *
217
444
  * Must be called AFTER `cacheModels()` has populated `state.models`.
218
445
  * Returns the bare slug if the catalog isn't populated (resolveModel
@@ -220,6 +447,88 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
220
447
  * variant" — defaulting safe-side preserves the pre-change behavior).
221
448
  */
222
449
  const DEFAULT_OPUS_FAMILY = "5";
450
+ /** The lead `-m fast` selects. Anthropic-published dashed slug that is also the
451
+ * Copilot catalog id verbatim, so `resolveModel` matches it exactly. */
452
+ const BUDGET_LEAD_MODEL = "claude-sonnet-5";
453
+ /** Small/fast tier for a budget lead, in the two forms this codebase needs.
454
+ *
455
+ * `SLUG` is the Anthropic-published DASHED form and is what goes into
456
+ * `ANTHROPIC_SMALL_FAST_MODEL` / `ANTHROPIC_DEFAULT_HAIKU_MODEL`: Claude Code's
457
+ * `/model` registry is keyed on Anthropic slugs, and seeding Copilot's dotted
458
+ * id there reproduces the documented `claude-opus-5` failure where the picker
459
+ * silently falls back to an older model. `CATALOG_ID` is Copilot's DOTTED id
460
+ * and is what the presence probe must test, because that is the id the catalog
461
+ * actually carries. `resolveModel` bridges the two at request time. */
462
+ const BUDGET_SMALL_FAST_SLUG = "claude-haiku-4-5";
463
+ const BUDGET_SMALL_FAST_CATALOG_ID = "claude-haiku-4.5";
464
+ /**
465
+ * Resolve the `-m` argument to the lead slug to launch with.
466
+ *
467
+ * - `fast` → `BUDGET_LEAD_MODEL` (budget mode)
468
+ * - `N.M` → the best variant of that Opus family, via `pickClaudeDefault`
469
+ * - a full slug → unchanged, including Copilot slugs a power user pins
470
+ * - absent → the ordinary default
471
+ *
472
+ * Every branch is `[1m]`-decorated against the live catalog, by
473
+ * `pickClaudeDefault` on the two Opus-family branches and by
474
+ * `withOneMSuffixForLead` on the other two. Pinning a MODEL is not a request to
475
+ * give up four fifths of its context window, which is what leaving the other
476
+ * two bare amounted to: `claude-sonnet-5` advertises a 1M window, so `-m fast`
477
+ * and `-m claude-sonnet-5` were both budgeted locally at Claude Code's 200K
478
+ * default and auto-compacted at roughly a fifth of the real window. The
479
+ * decoration is catalog-gated per model, so a genuinely 200K model
480
+ * (`claude-haiku-4.5`) still comes back bare.
481
+ *
482
+ * `fast` resolves to an ordinary slug rather than setting a mode flag, because
483
+ * budget mode is keyed off the RESOLVED lead everywhere it matters (the advisor
484
+ * escalation, the delegation prose, the small/fast tier). `-m fast` and
485
+ * `-m claude-sonnet-5` must therefore produce identical sessions, which a flag
486
+ * only one of the two set would break. The shared decoration is part of that
487
+ * identity: decorating one branch and not the other would reintroduce the
488
+ * divergence through the context budget instead of through a flag.
489
+ *
490
+ * Callers must keep treating any explicit `-m` as explicit: the
491
+ * `DEFAULT_CLAUDE_MODEL_FALLBACKS` walk applies to the implicit-default path
492
+ * only, so it cannot override a requested family (or `fast`) with an older Opus.
493
+ */
494
+ function resolveLeadSlugArg(modelArg) {
495
+ const arg = modelArg?.trim();
496
+ if (!arg) return pickClaudeDefault();
497
+ if (arg.toLowerCase() === "fast") return withOneMSuffixForLead(BUDGET_LEAD_MODEL);
498
+ const opusFamilyShorthand = arg.match(/^(\d+\.\d+)$/)?.[1];
499
+ if (opusFamilyShorthand) return pickClaudeDefault(opusFamilyShorthand);
500
+ return withOneMSuffixForLead(arg);
501
+ }
502
+ /**
503
+ * True when `slug` names a Claude model that is NOT an Opus tier — the
504
+ * "budget lead" condition.
505
+ *
506
+ * Selecting sonnet or haiku as the lead is a decision to spend less while
507
+ * holding quality as far as possible, and three surfaces key off it: the
508
+ * advisor escalates to the Anthropic frontier (`resolveAdvisorModel`), the
509
+ * injected delegation prose puts the cheap agent tiers first
510
+ * (`buildNativeReachClauses`), and the small/fast tier drops to Haiku
511
+ * (`getClaudeCodeEnvVars`). One definition here so those three cannot disagree
512
+ * about what counts as a budget lead.
513
+ *
514
+ * Resolves before the family test so the Anthropic dashed form, Copilot's
515
+ * dotted form, and `pickClaudeDefault`'s literal `[1m]` suffix all classify
516
+ * alike. A non-Claude lead is not a budget lead: the concept is about picking a
517
+ * lighter tier WITHIN the Claude family, and the gpt/gemini shim models have
518
+ * their own cost profile that this switch says nothing about.
519
+ *
520
+ * CONTRACT: `slug` is an already-resolved LEAD SLUG, never a raw `-m` argument.
521
+ * `"fast"` and the `N.M` shorthand are not Claude slugs and would classify
522
+ * false here; run them through `resolveLeadSlugArg` first, which is what every
523
+ * caller does. Resolving internally instead would drag `pickClaudeDefault`'s
524
+ * catalog dependency into a pure predicate and make the same input answer
525
+ * differently before and after the catalog loads.
526
+ */
527
+ function isBudgetClaudeLead(slug) {
528
+ if (!slug) return false;
529
+ if (!isClaudeModel(slug)) return false;
530
+ return !/opus/i.test(resolveModel(slug));
531
+ }
223
532
  function pickClaudeDefault(opusFamily = DEFAULT_OPUS_FAMILY) {
224
533
  const dotted = opusFamily.replace(/-/g, ".");
225
534
  const bareSlug = `claude-opus-${dotted.replace(/\./g, "-")}`;
@@ -11888,6 +12197,123 @@ function objectProp(description, properties, required) {
11888
12197
  function anyProp(description) {
11889
12198
  return { description };
11890
12199
  }
12200
+ //#endregion
12201
+ //#region src/lib/tree-sitter-assets/files.ts
12202
+ /** WASM assets code search loads from tree-sitter-wasms. */
12203
+ const TREE_SITTER_GRAMMAR_FILES = {
12204
+ typescript: "tree-sitter-typescript.wasm",
12205
+ tsx: "tree-sitter-tsx.wasm",
12206
+ javascript: "tree-sitter-javascript.wasm",
12207
+ python: "tree-sitter-python.wasm",
12208
+ go: "tree-sitter-go.wasm",
12209
+ rust: "tree-sitter-rust.wasm",
12210
+ java: "tree-sitter-java.wasm",
12211
+ c: "tree-sitter-c.wasm",
12212
+ cpp: "tree-sitter-cpp.wasm"
12213
+ };
12214
+ const TREE_SITTER_RUNTIME_FILE = "tree-sitter.wasm";
12215
+ //#endregion
12216
+ //#region src/lib/tree-sitter-assets/provision.ts
12217
+ const PUBLISH_ATTEMPTS = 3;
12218
+ let _provisioned$1 = false;
12219
+ let _inFlight$1;
12220
+ /**
12221
+ * Materialize every required WASM asset. Single-flight and success-cached;
12222
+ * transient failures remain retryable. Never throws to the launcher.
12223
+ */
12224
+ function provisionTreeSitterAssets() {
12225
+ if (_provisioned$1) return Promise.resolve();
12226
+ if (_inFlight$1) return _inFlight$1;
12227
+ _inFlight$1 = Promise.resolve().then(() => provisionImpl()).then((complete) => {
12228
+ if (complete) _provisioned$1 = true;
12229
+ }).finally(() => {
12230
+ _inFlight$1 = void 0;
12231
+ });
12232
+ return _inFlight$1;
12233
+ }
12234
+ function provisionImpl() {
12235
+ try {
12236
+ const require = createRequire(import.meta.url);
12237
+ const grammarPackage = require.resolve("tree-sitter-wasms/package.json");
12238
+ const grammarRoot = path.join(path.dirname(grammarPackage), "out");
12239
+ const runtime = require.resolve("web-tree-sitter/tree-sitter.wasm");
12240
+ const destination = PATHS.TREE_SITTER_ASSETS_DIR;
12241
+ mkdirSync(destination, { recursive: true });
12242
+ let complete = publishAsset(runtime, path.join(destination, TREE_SITTER_RUNTIME_FILE));
12243
+ for (const filename of Object.values(TREE_SITTER_GRAMMAR_FILES)) complete = publishAsset(path.join(grammarRoot, filename), path.join(destination, filename)) && complete;
12244
+ return complete;
12245
+ } catch (err) {
12246
+ consola.debug("[tree-sitter-assets] provisioning skipped:", err);
12247
+ return false;
12248
+ }
12249
+ }
12250
+ /**
12251
+ * Publish through a unique same-directory temporary file. Never remove the
12252
+ * destination first: a concurrent parser must see either the prior complete
12253
+ * file or the new complete file, never a missing/partial path. A losing racer
12254
+ * accepts the winner when its bytes match the source.
12255
+ */
12256
+ function publishAsset(source, destination) {
12257
+ let bytes;
12258
+ try {
12259
+ bytes = readFileSync(source);
12260
+ } catch {
12261
+ return false;
12262
+ }
12263
+ if (matchesContent(destination, bytes)) return true;
12264
+ for (let attempt = 0; attempt < PUBLISH_ATTEMPTS; attempt++) {
12265
+ const tmp = `${destination}.${process.pid}-${attempt}.tmp`;
12266
+ try {
12267
+ writeFileSync(tmp, bytes);
12268
+ renameSync(tmp, destination);
12269
+ return true;
12270
+ } catch {
12271
+ try {
12272
+ rmSync(tmp, { force: true });
12273
+ } catch {}
12274
+ if (matchesContent(destination, bytes)) return true;
12275
+ }
12276
+ }
12277
+ return false;
12278
+ }
12279
+ /**
12280
+ * Whether the published file is byte-identical to the source.
12281
+ *
12282
+ * Content, not size: a grammar upgrade that happens to keep the same byte
12283
+ * length would otherwise never propagate, and the stale copy would shadow the
12284
+ * package's newer file permanently — a silent, self-perpetuating wrong answer.
12285
+ * The size check first keeps the common case to one `stat`.
12286
+ */
12287
+ function matchesContent(file, bytes) {
12288
+ try {
12289
+ const stat = statSync(file);
12290
+ if (!stat.isFile() || stat.size !== bytes.byteLength) return false;
12291
+ return readFileSync(file).equals(bytes);
12292
+ } catch {
12293
+ return false;
12294
+ }
12295
+ }
12296
+ /**
12297
+ * Whether the stable directory holds the COMPLETE asset set: the runtime plus
12298
+ * every grammar.
12299
+ *
12300
+ * Callers must gate on the whole set, never on the one file they are about to
12301
+ * read. web-tree-sitter enforces a language-ABI version between the runtime and
12302
+ * the grammars, so adopting a stable runtime while grammars still come from the
12303
+ * package tree (or the reverse) can pair mismatched builds. That surfaces as a
12304
+ * caught `Language.load` failure, which silently disables structural ranking —
12305
+ * the exact silent degradation this whole change is meant to end.
12306
+ */
12307
+ function stableTreeSitterAssetsComplete() {
12308
+ const dir = PATHS.TREE_SITTER_ASSETS_DIR;
12309
+ return [TREE_SITTER_RUNTIME_FILE, ...Object.values(TREE_SITTER_GRAMMAR_FILES)].every((name) => {
12310
+ try {
12311
+ return statSync(path.join(dir, name)).isFile();
12312
+ } catch {
12313
+ return false;
12314
+ }
12315
+ });
12316
+ }
11891
12317
  /**
11892
12318
  * Extension → grammar key. Grammars not in this map skip structural
11893
12319
  * parsing (the hit falls back to the regex SYMBOL_REGEX heuristic for
@@ -11915,22 +12341,11 @@ const EXTENSION_TO_LANG = {
11915
12341
  ".hxx": "cpp"
11916
12342
  };
11917
12343
  /**
11918
- * Grammar key → wasm filename under `node_modules/tree-sitter-wasms/out/`.
11919
- * Resolved at runtime from `node_modules`; the file paths are stable
11920
- * because `tree-sitter-wasms` ships prebuilt binaries (no per-install
11921
- * codegen).
12344
+ * Grammar key → wasm filename. The stable APP_DIR copy is preferred; the
12345
+ * original `node_modules/tree-sitter-wasms/out/` directory remains the
12346
+ * first-run fallback while background provisioning completes.
11922
12347
  */
11923
- const GRAMMAR_FILES = {
11924
- typescript: "tree-sitter-typescript.wasm",
11925
- tsx: "tree-sitter-tsx.wasm",
11926
- javascript: "tree-sitter-javascript.wasm",
11927
- python: "tree-sitter-python.wasm",
11928
- go: "tree-sitter-go.wasm",
11929
- rust: "tree-sitter-rust.wasm",
11930
- java: "tree-sitter-java.wasm",
11931
- c: "tree-sitter-c.wasm",
11932
- cpp: "tree-sitter-cpp.wasm"
11933
- };
12348
+ const GRAMMAR_FILES = TREE_SITTER_GRAMMAR_FILES;
11934
12349
  /**
11935
12350
  * Per-language definition-shape node types. When a matched identifier
11936
12351
  * sits inside one of these nodes AND is at the node's "name" position,
@@ -12064,12 +12479,16 @@ function getLanguageKeyForPath(filePath) {
12064
12479
  }
12065
12480
  let _grammarBundle;
12066
12481
  /**
12067
- * Resolve the `tree-sitter-wasms/out/` directory at the package root.
12068
- * `require.resolve` is used through a try/catch — the bundled-only
12069
- * fallback runs in environments where node_modules has been pruned to
12070
- * just runtime deps.
12482
+ * Resolve the grammar directory. The stable APP_DIR copy wins only when the
12483
+ * COMPLETE set is present; otherwise `require.resolve` supplies the
12484
+ * package-tree fallback for first launch or a best-effort provisioning failure.
12485
+ *
12486
+ * All-or-nothing on purpose — see `stableTreeSitterAssetsComplete()`: mixing a
12487
+ * stable runtime with package-tree grammars can pair mismatched ABI builds and
12488
+ * silently disable structural ranking.
12071
12489
  */
12072
12490
  function resolveGrammarRoot() {
12491
+ if (stableTreeSitterAssetsComplete()) return PATHS.TREE_SITTER_ASSETS_DIR;
12073
12492
  try {
12074
12493
  const pkgPath = __require.resolve("tree-sitter-wasms/package.json");
12075
12494
  return path$1.join(path$1.dirname(pkgPath), "out");
@@ -12078,6 +12497,15 @@ function resolveGrammarRoot() {
12078
12497
  }
12079
12498
  }
12080
12499
  /**
12500
+ * web-tree-sitter normally resolves this sidecar relative to its JS module.
12501
+ * Prefer the durable copy, but only under the same all-present gate the
12502
+ * grammars use, so the runtime and the grammars always come from one install.
12503
+ */
12504
+ function parserInitOptions() {
12505
+ if (!stableTreeSitterAssetsComplete()) return void 0;
12506
+ return { locateFile: () => path$1.join(PATHS.TREE_SITTER_ASSETS_DIR, TREE_SITTER_RUNTIME_FILE) };
12507
+ }
12508
+ /**
12081
12509
  * Pre-load all grammars at module-init time so the first search
12082
12510
  * doesn't pay a ~500ms cold-start cost. The Promise is captured at
12083
12511
  * import time and awaited per-call; per-grammar failures are caught
@@ -12088,7 +12516,7 @@ function getGrammarBundle() {
12088
12516
  _grammarBundle = { ready: (async () => {
12089
12517
  const out = /* @__PURE__ */ new Map();
12090
12518
  try {
12091
- await Parser.init();
12519
+ await Parser.init(parserInitOptions());
12092
12520
  } catch (err) {
12093
12521
  consola.warn(`[code_search] tree-sitter Parser.init failed; structural ranking disabled: ${err.message}`);
12094
12522
  return out;
@@ -13271,16 +13699,19 @@ const STRUCTURAL_CACHE_MAX = 64;
13271
13699
  const SYMBOL_REGEX = /^(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:public\s+|private\s+|protected\s+|static\s+|abstract\s+|readonly\s+)*(?:function|class|interface|type|enum|def|fn|trait|impl|module|namespace|const|let|var)\s+[A-Za-z_$]/;
13272
13700
  let _rgResolution;
13273
13701
  /**
13274
- * Tri-tier resolution. Memoized. Mirrors cc-backup
13702
+ * Four-tier resolution. Memoized. Mirrors cc-backup
13275
13703
  * `src/utils/ripgrep.ts:31-65`.
13276
13704
  *
13277
13705
  * 1. System rg on PATH — use the literal command name `"rg"` (NOT
13278
13706
  * the absolute path). This leverages NoDefaultCurrentDirectory-
13279
13707
  * InExePath on Windows, preventing PATH-hijacking via a
13280
13708
  * malicious ./rg.exe in the proxy's cwd.
13281
- * 2. Bundled via `@vscode/ripgrep` falls back to the per-platform
13709
+ * 2. Router-owned toolbelt copy under APP_DIR. Its absolute path is
13710
+ * safe because it cannot resolve to a planted binary in the cwd,
13711
+ * and it survives a temp-hosted bunx package tree being reaped.
13712
+ * 3. Bundled via `@vscode/ripgrep` — falls back to the per-platform
13282
13713
  * binary that `optionalDependencies` installed.
13283
- * 3. Throw — surfaced to the caller as an MCP isError response.
13714
+ * 4. Throw — surfaced to the caller as an MCP isError response.
13284
13715
  */
13285
13716
  function resolveRipgrep() {
13286
13717
  if (_rgResolution) return _rgResolution;
@@ -13291,6 +13722,14 @@ function resolveRipgrep() {
13291
13722
  };
13292
13723
  return _rgResolution;
13293
13724
  }
13725
+ const toolbeltPath = path$1.join(PATHS.TOOLBELT_BIN_DIR, process.platform === "win32" ? "rg.exe" : "rg");
13726
+ if (existsSync(toolbeltPath)) {
13727
+ _rgResolution = {
13728
+ rgPath: toolbeltPath,
13729
+ source: "toolbelt"
13730
+ };
13731
+ return _rgResolution;
13732
+ }
13294
13733
  try {
13295
13734
  const mod = __require("@vscode/ripgrep");
13296
13735
  if (mod.rgPath && existsSync(mod.rgPath)) {
@@ -17206,14 +17645,19 @@ function findPackageRoot(startDir, maxHops = 10) {
17206
17645
  }
17207
17646
  }
17208
17647
  /**
17209
- * Resolve the github-router package root. Uses two sources in order:
17210
- * 1. process.argv[1] the entrypoint script, walks up from there.
17211
- * 2. import.meta.url of THIS module, walks up from there.
17212
- * 3. process.cwd() as last resort.
17648
+ * Resolve the github-router package root. Uses sources in order:
17649
+ * 1. The explicit root baked into the relocated hook launcher's argv. From
17650
+ * `<APP_DIR>/hooks/`, neither entrypoint walk can find the package and cwd
17651
+ * is the user's workspace, so this source must win when present.
17652
+ * 2. process.argv[1] — the entrypoint script, walks up from there.
17653
+ * 3. import.meta.url of THIS module, walks up from there.
17654
+ * 4. process.cwd() as last resort.
17213
17655
  *
17214
- * Robust across bun (src/main.ts) and node (dist/main.js) launch paths.
17656
+ * Robust across relocated hooks, bun (src/main.ts), and node (dist/main.js).
17215
17657
  */
17216
17658
  function packageRoot() {
17659
+ const explicit = explicitPackageRoot();
17660
+ if (explicit) return explicit;
17217
17661
  const entryPath = typeof process$1?.argv?.[1] === "string" ? process$1.argv[1] : void 0;
17218
17662
  if (entryPath) {
17219
17663
  const fromEntry = findPackageRoot(path.dirname(entryPath));
@@ -18470,7 +18914,7 @@ function logAudit$1(record) {
18470
18914
  try {
18471
18915
  const fs = await import("node:fs/promises");
18472
18916
  const path = await import("node:path");
18473
- const { PATHS } = await import("./paths-DN3Nio42.js");
18917
+ const { PATHS } = await import("./paths-CV9K7Xqm.js");
18474
18918
  const dir = path.join(PATHS.APP_DIR, "browser-mcp");
18475
18919
  await fs.mkdir(dir, { recursive: true });
18476
18920
  const line = JSON.stringify({
@@ -19849,70 +20293,6 @@ function detectAgentCall(input) {
19849
20293
  });
19850
20294
  }
19851
20295
  //#endregion
19852
- //#region src/services/copilot/endpoint.ts
19853
- /**
19854
- * Catalog spellings that mean each of our two clients. Copilot is not
19855
- * self-consistent about the `/v1` prefix — the live catalog advertises
19856
- * `/v1/messages` prefixed but `/chat/completions` bare, and this repo's own
19857
- * fixtures carry both forms — so an exact-match on the bare spelling alone
19858
- * silently misses a real shape. `src/lib/model-validation.ts` already
19859
- * normalizes the same way (`ENDPOINT_ALIASES`); this keeps the two agreeing.
19860
- *
19861
- * Matching is EXACT against this set, never a suffix/`includes` test: a
19862
- * `ws:/responses` (websocket transport) entry is NOT the `/responses` HTTP
19863
- * client and must keep resolving to "serves neither".
19864
- */
19865
- const CHAT_ENDPOINTS = /* @__PURE__ */ new Set(["/chat/completions", "/v1/chat/completions"]);
19866
- const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
19867
- /**
19868
- * Decide which endpoint to call for a model from its catalog
19869
- * `supported_endpoints`. Prefers `/chat/completions` when available (the
19870
- * simpler, more widely-supported shape) and falls back to `/responses` for
19871
- * models that ONLY serve the Responses API — the gpt-5.x family except
19872
- * `gpt-5-mini` / `gpt-5.4` (e.g. `gpt-5.4-mini`, `gpt-5.5`, the
19873
- * `*-codex` models). Returns undefined when the model serves neither, so a
19874
- * caller can skip it rather than 400 on `unsupported_api_for_model`.
19875
- *
19876
- * A model that OMITS `supported_endpoints` is treated as chat-eligible: the
19877
- * catalog historically omits the field for chat-default models, and
19878
- * excluding those would be a worse regression than the gap this guards.
19879
- */
19880
- function pickEndpoint(model) {
19881
- const eps = model.supported_endpoints;
19882
- if (!eps || eps.length === 0) return "chat";
19883
- if (eps.some((e) => CHAT_ENDPOINTS.has(e))) return "chat";
19884
- if (eps.some((e) => RESPONSES_ENDPOINTS.has(e))) return "responses";
19885
- }
19886
- /**
19887
- * `pickEndpoint` by model id against the live catalog, WITHOUT collapsing
19888
- * "absent from the catalog" into "serves neither of our endpoints".
19889
- *
19890
- * This function deliberately has no default. The predecessor
19891
- * (`endpointForModelId`) returned `pickEndpoint(found) ?? "chat"`, which
19892
- * coerced both cases to "chat" — defensible for an unknown id, silently wrong
19893
- * for a catalog model serving only, say, `/v1/messages`: the caller would drive
19894
- * it through the chat client and get an opaque upstream 400 with no local
19895
- * signal about the real cause. `src/lib/browser-mcp/compressor.ts` already
19896
- * treats that case correctly (`if (!endpoint) continue`); this makes the same
19897
- * distinction available to callers that resolve by id.
19898
- *
19899
- * Callers that legitimately want the chat default for an unknown id can still
19900
- * have it — they just have to write it, per case, on purpose.
19901
- */
19902
- function resolveEndpointForModelId(id) {
19903
- const found = state.models?.data?.find((m) => m.id === id);
19904
- if (!found) return { kind: "unknown-model" };
19905
- const endpoint = pickEndpoint(found);
19906
- if (endpoint) return {
19907
- kind: "endpoint",
19908
- endpoint
19909
- };
19910
- return {
19911
- kind: "unreachable",
19912
- endpoints: found.supported_endpoints ?? []
19913
- };
19914
- }
19915
- //#endregion
19916
20296
  //#region src/lib/browser-mcp/compressor.ts
19917
20297
  /**
19918
20298
  * Static fallback chain for the inner compressor. Order is preference:
@@ -22715,7 +23095,7 @@ const EMPTY_USAGE = {
22715
23095
  total: 0
22716
23096
  }
22717
23097
  };
22718
- const DEFAULT_MODEL$1 = {
23098
+ const DEFAULT_MODEL = {
22719
23099
  id: "unknown",
22720
23100
  name: "unknown",
22721
23101
  api: "unknown",
@@ -22737,7 +23117,7 @@ function createMutableAgentState(initialState) {
22737
23117
  let messages = initialState?.messages?.slice() ?? [];
22738
23118
  return {
22739
23119
  systemPrompt: initialState?.systemPrompt ?? "",
22740
- model: initialState?.model ?? DEFAULT_MODEL$1,
23120
+ model: initialState?.model ?? DEFAULT_MODEL,
22741
23121
  thinkingLevel: initialState?.thinkingLevel ?? "off",
22742
23122
  get tools() {
22743
23123
  return tools;
@@ -23462,18 +23842,174 @@ function resolveModelAndThinking(opts) {
23462
23842
  if (!clamp) clamp = allowed[0];
23463
23843
  return mkOk(clamp);
23464
23844
  }
23845
+ const CATALOG_PRICE_SCALE = 1e9;
23846
+ const TOKENS_PER_MILLION = 1e6;
23847
+ /**
23848
+ * Last-resort per-1M-token prices, recorded from the live catalog on
23849
+ * 2026-08-12. The LIVE catalog always wins; this only fills in when
23850
+ * `state.models` is unpopulated, which in practice means the startup catalog
23851
+ * fetch failed. Without it the injected roster degrades to bare names and the
23852
+ * model loses the cost signal entirely for that session.
23853
+ *
23854
+ * A hardcoded copy of a value that HAS a live source is a second source of
23855
+ * truth, and this one has already been observed to drift: two figures written
23856
+ * from memory into a commit message (`gpt-5.3-codex` 400/1600, `gpt-5.5`
23857
+ * 500/2000) were both wrong against the live catalog (175/1400 and 500/3000).
23858
+ * That is exactly the silent-misroute failure this table risks, so
23859
+ * `warnOnTokenPriceDrift()` compares it against the live catalog once at
23860
+ * startup and logs any disagreement rather than letting a stale number sit
23861
+ * here indefinitely.
23862
+ *
23863
+ * This is NOT the same trade as `INDICATIVE_TOKENS_PER_SECOND`: throughput
23864
+ * cannot be derived from the catalog at all, so hardcoding is the only option
23865
+ * there. Price can, so hardcoding is strictly a degraded fallback.
23866
+ */
23867
+ const FALLBACK_TOKEN_PRICES = Object.freeze({
23868
+ "gpt-5.6-luna": {
23869
+ in: 20,
23870
+ out: 120
23871
+ },
23872
+ "gpt-5.6-terra": {
23873
+ in: 200,
23874
+ out: 1200
23875
+ },
23876
+ "gpt-5.4-mini": {
23877
+ in: 75,
23878
+ out: 450
23879
+ },
23880
+ "claude-sonnet-5": {
23881
+ in: 200,
23882
+ out: 1e3
23883
+ },
23884
+ "gpt-5.3-codex": {
23885
+ in: 175,
23886
+ out: 1400
23887
+ },
23888
+ "claude-haiku-4.5": {
23889
+ in: 100,
23890
+ out: 500
23891
+ },
23892
+ "claude-opus-5": {
23893
+ in: 500,
23894
+ out: 2500
23895
+ },
23896
+ "gpt-5.6-sol": {
23897
+ in: 500,
23898
+ out: 3e3
23899
+ },
23900
+ "grok-4.5": {
23901
+ in: 200,
23902
+ out: 600
23903
+ },
23904
+ "gpt-5.5": {
23905
+ in: 500,
23906
+ out: 3e3
23907
+ },
23908
+ "gemini-3.6-flash": {
23909
+ in: 150,
23910
+ out: 750
23911
+ },
23912
+ "gemini-3.5-flash": {
23913
+ in: 150,
23914
+ out: 900
23915
+ },
23916
+ "gemini-3.1-pro-preview": {
23917
+ in: 200,
23918
+ out: 1200
23919
+ }
23920
+ });
23921
+ /**
23922
+ * Compare every `FALLBACK_TOKEN_PRICES` entry against the live catalog and warn
23923
+ * on disagreement. Call once after the catalog is populated. Makes fallback
23924
+ * staleness VISIBLE instead of silent: a stale entry only ever surfaces on the
23925
+ * degraded path, where nobody is looking, so without this it could be wrong for
23926
+ * months. Warn-only by design — a price mismatch must never block a launch.
23927
+ */
23928
+ function warnOnTokenPriceDrift() {
23929
+ for (const [id, hardcoded] of Object.entries(FALLBACK_TOKEN_PRICES)) {
23930
+ const live = livePricesFor(id);
23931
+ if (!live) continue;
23932
+ if (live.in !== hardcoded.in || live.out !== hardcoded.out) consola.warn(`[model-resolve] FALLBACK_TOKEN_PRICES is stale for ${id}: hardcoded ${hardcoded.in}/${hardcoded.out}, live catalog ${live.in}/${live.out}. Update the table in src/lib/worker-agent/model-resolve.ts.`);
23933
+ }
23934
+ }
23935
+ /** Live-catalog price lookup with no fallback. Split out so the drift check can
23936
+ * compare against the catalog without the fallback masking a disagreement. */
23937
+ function livePricesFor(modelId) {
23938
+ const prices = state.models?.data.find((model) => model.id === modelId)?.billing?.token_prices;
23939
+ if (!prices || typeof prices.batch_size !== "number" || !Number.isSafeInteger(prices.batch_size) || prices.batch_size <= 0 || typeof prices.input_price !== "number" || !Number.isFinite(prices.input_price) || prices.input_price < 0 || typeof prices.output_price !== "number" || !Number.isFinite(prices.output_price) || prices.output_price < 0) return;
23940
+ const toPerMillion = (price) => price / CATALOG_PRICE_SCALE * TOKENS_PER_MILLION / prices.batch_size;
23941
+ return {
23942
+ in: toPerMillion(prices.input_price),
23943
+ out: toPerMillion(prices.output_price)
23944
+ };
23945
+ }
23946
+ /**
23947
+ * A model's per-1M-token prices: live catalog first, then the dated fallback
23948
+ * table. Still returns undefined for a model in neither, so a caller never
23949
+ * mistakes a guess for a fact — the fallback covers models we have actually
23950
+ * recorded, not every id.
23951
+ */
23952
+ function catalogTokenPrices(modelId) {
23953
+ return livePricesFor(modelId) ?? FALLBACK_TOKEN_PRICES[modelId];
23954
+ }
23955
+ /**
23956
+ * Approximate output tokens/sec, median of n=3 per model, measured 2026-08-12
23957
+ * through this proxy. Reproduce with `bun scripts/bench-model-speed.ts` — the
23958
+ * harness is committed precisely so these numbers can be re-derived and
23959
+ * challenged instead of being trusted. Rounded coarsely on purpose: run-to-run
23960
+ * variance is large (`gpt-5.6-sol` measured 22 in an early n=1 pass and 74 at
23961
+ * n=3), so any digit beyond the leading one or two would be false precision.
23962
+ *
23963
+ * Wall clock includes time-to-first-token, which is why an early n=1 pass put
23964
+ * `gemini-3.1-pro-preview` at 9: that response emitted only 66 tokens, so TTFT
23965
+ * dominated. Reasoning tokens are timed but may not appear in `output_tokens`,
23966
+ * so heavy-reasoning models are penalised here.
23967
+ *
23968
+ * This is a deliberately hardcoded, coarse speed hint, indicative and never a
23969
+ * per-call benchmark: a recoverable speed retry is safer than a quality score
23970
+ * that silently misroutes.
23971
+ *
23972
+ * NOT the whole picture for agent work. The benchmark also measures p50 latency
23973
+ * to a trivial tool call, which is the workload an agent model actually spends
23974
+ * its turns on, and the ordering differs from raw generation: `gpt-5.6-sol`
23975
+ * generates at 75 but takes ~4.3s to reach a tool call, while `gpt-5.6-luna`
23976
+ * takes ~0.9s. That figure is deliberately NOT surfaced to the model, because a
23977
+ * second speed axis invites optimising a routing choice that policy already
23978
+ * settles (see the decorrelation note below).
23979
+ */
23980
+ const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
23981
+ "gpt-5.6-luna": 120,
23982
+ "gpt-5.6-terra": 100,
23983
+ "gpt-5.4-mini": 100,
23984
+ "claude-sonnet-5": 100,
23985
+ "gpt-5.3-codex": 85,
23986
+ "claude-haiku-4.5": 85,
23987
+ "claude-opus-5": 80,
23988
+ "gpt-5.6-sol": 75,
23989
+ "grok-4.5": 70,
23990
+ "gpt-5.5": 65,
23991
+ "gemini-3.6-flash": 45,
23992
+ "gemini-3.5-flash": 40,
23993
+ "gemini-3.1-pro-preview": 25
23994
+ });
23995
+ /** Returns the approximate, indicative output speed when it was measured. */
23996
+ function indicativeTokensPerSecond(modelId) {
23997
+ return INDICATIVE_TOKENS_PER_SECOND[modelId];
23998
+ }
23465
23999
  /** Worker-usable models need a big enough window to be worth delegating to. */
23466
24000
  const CATALOG_MIN_CONTEXT = 2e5;
23467
24001
  /**
23468
24002
  * Derived view of the live catalog: every model a worker could actually be
23469
24003
  * pointed at, with the metadata needed to choose between them.
23470
24004
  *
23471
- * DERIVED ONLY, and that is the whole design. A one-liner like "strong
24005
+ * Derived facts only, with one explicitly-labelled exception:
24006
+ * `INDICATIVE_TOKENS_PER_SECOND` is a dated, coarse measurement whose speed
24007
+ * signal is recoverable by retrying a slow selection. A one-liner like "strong
23472
24008
  * reasoning, weak long-context recall" cannot be computed from catalog
23473
24009
  * metadata — it is editorial, it goes stale silently as vendors ship, and the
23474
24010
  * asymmetry is brutal: a MISSING characterization costs one suboptimal pick
23475
24011
  * the model recovers from, while a WRONG one misroutes invisibly at the call
23476
- * site. So this ships facts and lets the caller judge.
24012
+ * site. So this ships facts, the recoverable speed hint, and no quality score.
23477
24013
  *
23478
24014
  * It exists because the hardcoded chains cannot discover anything. Models are
23479
24015
  * live in the catalog that appear nowhere in `src/` — nobody evaluated them
@@ -23499,13 +24035,16 @@ function buildCatalogView() {
23499
24035
  if (ctx < CATALOG_MIN_CONTEXT) continue;
23500
24036
  const efforts = (supports.reasoning_effort ?? []).filter((effort) => WORKER_THINKING_LEVELS.includes(effort));
23501
24037
  if (efforts.length === 0) continue;
24038
+ const prices = catalogTokenPrices(model.id);
24039
+ const tps = indicativeTokensPerSecond(model.id);
23502
24040
  rows.push({
23503
24041
  id: model.id,
23504
24042
  vendor: model.vendor,
23505
24043
  ctx,
23506
24044
  ...limits?.max_output_tokens ? { maxOut: limits.max_output_tokens } : {},
23507
24045
  efforts,
23508
- ...model.model_picker_price_category ? { cost: model.model_picker_price_category } : {}
24046
+ ...prices ?? {},
24047
+ ...tps === void 0 ? {} : { tps }
23509
24048
  });
23510
24049
  }
23511
24050
  return rows.sort((a, b) => a.id.localeCompare(b.id));
@@ -26511,13 +27050,12 @@ function geminiAvailable(source = state) {
26511
27050
  * one walk instead of hand-copying it. Ids are matched EXACTLY against
26512
27051
  * `catalog.id` — no slug translation, matching the pre-existing behavior.
26513
27052
  *
26514
- * `minContextTokens` is OPT-IN because the two constraints are genuinely
26515
- * per-agent: the `generic*` chains promise 1M end to end, while `scoutModel`
26516
- * deliberately keeps a 400K last resort for its wider availability. Enforcing
26517
- * the floor here rather than by comment is what stops a chain silently
26518
- * degrading when an id's advertised window shrinks upstream `withOneMSuffix`
26519
- * would then just omit the `[1m]` bracket, and the agent would be budgeted at
26520
- * Claude Code's 200K default with no signal that anything changed.
27053
+ * `minContextTokens` is OPT-IN because the constraint is genuinely per-agent:
27054
+ * the conditional cheaper-tier agents promise 1M end to end. Enforcing the
27055
+ * floor here rather than by comment is what stops a chain silently degrading
27056
+ * when an id's advertised window shrinks upstream `withOneMSuffix` would then
27057
+ * just omit the `[1m]` bracket, and the agent would be budgeted at Claude Code's
27058
+ * 200K default with no signal that anything changed.
26521
27059
  */
26522
27060
  function firstPresentInCatalog(chain, opts) {
26523
27061
  const models = state.models?.data;
@@ -26607,61 +27145,47 @@ function scribeModel() {
26607
27145
  * (same behavior as before `scout` existed) rather than to an expensive
26608
27146
  * impostor wearing the cheap agent's name.
26609
27147
  *
26610
- * `gpt-5.6-luna` sits between the two originals because the old chain fell
26611
- * straight from a 1M model to 400K `gpt-5.4-mini`, which loses the `[1m]`
26612
- * bracket and drops Claude Code's accounting to its 200K default. Luna is
26613
- * cheaper than mini, keeps 1M, and is cross-vendor from the primary, so it
26614
- * covers a Gemini-side outage that a same-vendor entry would not.
27148
+ * `gpt-5.6-luna` leads because it is the cheapest 1M-context model in the
27149
+ * catalog; `gemini-3.6-flash` remains the cross-vendor fallback so an OpenAI-side
27150
+ * outage does not remove the scout. Both entries must continue advertising at
27151
+ * least 1M context so Claude Code's `[1m]` accounting remains honest if an
27152
+ * upstream catalog entry shrinks.
26615
27153
  *
26616
- * Deliberately NO `minContextTokens` floor, unlike the `generic*` resolvers:
26617
- * `gpt-5.4-mini` is retained as the last resort precisely BECAUSE it is the
26618
- * widest-availability id here (its `restricted_to` includes `individual_trial`
26619
- * and `edu`, which neither flash nor luna does). On a thin non-enterprise
26620
- * catalog a 400K scout beats no scout.
27154
+ * This chain deliberately uses literal ids rather than `EXPLORE_DEFAULT_MODEL`:
27155
+ * the explore worker default and scout's cross-vendor fallback are independent
27156
+ * policies, so retuning one must not silently collapse the other. There is no
27157
+ * 400K last resort. On a catalog carrying neither chain member, `scout` is
27158
+ * dropped rather than inheriting the lead or presenting a narrower-context agent.
26621
27159
  */
27160
+ const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.6-flash"]);
26622
27161
  function scoutModel() {
26623
- return firstPresentInCatalog([
26624
- EXPLORE_DEFAULT_MODEL,
26625
- "gpt-5.6-luna",
26626
- DEFAULT_MODEL
26627
- ], { requireToolCalls: true });
26628
- }
26629
- /** Model for `generic` — the mid-tier catch-all. Absent → the agent is dropped.
26630
- *
26631
- * `gpt-5.6-sol` is deliberately NOT in this chain: the OpenAI frontier coder is
26632
- * already `implementer`'s job, and a catch-all that quietly costs frontier
26633
- * rates is the opposite of what this agent is for. Both entries are 1M+ and
26634
- * mid-to-high capability, which is the most the description may claim. */
26635
- function genericModel() {
26636
- return firstPresentInCatalog(["gpt-5.6-terra", REVIEW_DEFAULT_MODEL], {
27162
+ return firstPresentInCatalog(SCOUT_MODEL_CHAIN, {
26637
27163
  requireToolCalls: true,
26638
27164
  minContextTokens: ONE_M_TOKENS
26639
27165
  });
26640
27166
  }
26641
- /** Model for `generic-fast` — the Gemini flash tier. Absent → dropped.
27167
+ /** Model for `implementer-fast` — the cheaper implementation tier. Absent →
27168
+ * the agent is dropped.
26642
27169
  *
26643
- * Both entries are the same vendor, context, price point and `minimal..high`
26644
- * effort ladder, so the agent's identity survives the fallback intact. That is
26645
- * why the fallback is `gemini-3.5-flash` and not `gpt-5.6-luna`, which would
26646
- * otherwise be the natural cross-vendor choice: luna is `genericCheapModel`'s
26647
- * only entry, and using it in both places would collapse two roster entries
26648
- * onto one model in the degraded case. */
26649
- function genericFastModel() {
26650
- return firstPresentInCatalog([EXPLORE_DEFAULT_MODEL, "gemini-3.5-flash"], {
27170
+ * `gpt-5.6-sol` is deliberately NOT in this chain: changes needing frontier
27171
+ * judgment already belong to `implementer`, while this agent handles
27172
+ * well-specified, mechanical changes at a lower tier. Both entries are 1M+;
27173
+ * their different speed and effort properties stay out of shared claims. */
27174
+ function implementerFastModel() {
27175
+ return firstPresentInCatalog(["gpt-5.6-terra", REVIEW_DEFAULT_MODEL], {
26651
27176
  requireToolCalls: true,
26652
27177
  minContextTokens: ONE_M_TOKENS
26653
27178
  });
26654
27179
  }
26655
- /** Model for `generic-cheap` — the cheapest catch-all. Absent → dropped.
27180
+ /** Model for `general-purpose-fast` — the fast, cheapest catch-all. Absent →
27181
+ * dropped.
26656
27182
  *
26657
- * Single-entry by design (see `genericFastModel` for why luna is not shared).
26658
- * `gpt-5.6-luna` is the cheapest id in the catalog and undercuts even the 400K
26659
- * `gpt-5.4-mini`, while carrying 1.05M context and the full `none..max` effort
26660
- * ladder so unlike the flash tier it propagates a CLI effort pick above
26661
- * `high` rather than clamping it. No `-mini`/`-lite`/`-haiku` model in the
26662
- * catalog serves 1M, which is why the cheap catch-all is a `gpt-5.6-*` slug
26663
- * rather than a mini one. */
26664
- function genericCheapModel() {
27183
+ * Single-entry by design. `gpt-5.6-luna` is the cheapest model in the live
27184
+ * catalog and measured fastest among the catch-all candidates, while carrying
27185
+ * 1.05M context and the full `none..max` effort ladder. No
27186
+ * `-mini`/`-lite`/`-haiku` model in the catalog serves 1M, which is why this
27187
+ * catch-all uses a `gpt-5.6-*` slug rather than a mini one. */
27188
+ function generalPurposeFastModel() {
26665
27189
  return firstPresentInCatalog(["gpt-5.6-luna"], {
26666
27190
  requireToolCalls: true,
26667
27191
  minContextTokens: ONE_M_TOKENS
@@ -26671,15 +27195,14 @@ function genericCheapModel() {
26671
27195
  * Gate for the worker tools (`explore`, `review`, `implement`).
26672
27196
  *
26673
27197
  * Returns true iff BOTH:
26674
- * 1. Copilot's live catalog (`state.models?.data`) contains the
26675
- * worker default model (`gpt-5.4-mini`, used by explore)
26676
- * AND that entry advertises `capabilities.supports.tool_calls ===
26677
- * true`. The worker loop is function-calling; a model that can't
26678
- * emit tool_calls is unusable, so dormant-register (omit from
26679
- * `tools/list`) keeps the surface honest. (The implement default
26680
- * `gpt-5.6-sol` is NOT gated here — if it's absent, implement calls
26681
- * surface a clean resolve error rather than disabling all worker
26682
- * tools, since explore/review still work.)
27198
+ * 1. Copilot's live catalog (`state.models?.data`) contains any model in the
27199
+ * ordered worker gate chain (`gpt-5.6-luna` → `gpt-5.4-mini`) and that
27200
+ * entry advertises `capabilities.supports.tool_calls === true`. Luna leads
27201
+ * on qualifying tiers; mini preserves the worker surface on individual
27202
+ * trial and education catalogs. The catalog is the entitlement signal.
27203
+ * The worker loop is function-calling, so a model without tool calls is
27204
+ * unusable. Per-mode defaults are NOT gated here — an absent mode default
27205
+ * surfaces a clean resolve error rather than disabling all worker tools.
26683
27206
  * 2. The operator hasn't set `GH_ROUTER_DISABLE_WORKER_TOOLS=1`
26684
27207
  * (opt-out — workers ship enabled by default per plan).
26685
27208
  *
@@ -26688,17 +27211,12 @@ function genericCheapModel() {
26688
27211
  * validation in the engine, which surfaces a clean `isError`
26689
27212
  * envelope with the catalog's eligible model ids on mismatch.
26690
27213
  *
26691
- * `WORKER_DEFAULT_MODEL` is imported (aliased from `DEFAULT_MODEL`)
26692
- * from `src/lib/worker-agent` so the engine owns the single source
26693
- * of truth.
27214
+ * `WORKER_DEFAULT_MODEL_CHAIN` is imported from `src/lib/worker-agent` so the
27215
+ * engine owns the single source of truth for both gating and fallback order.
26694
27216
  */
26695
27217
  function workerToolsEnabled() {
26696
27218
  if (process.env.GH_ROUTER_DISABLE_WORKER_TOOLS === "1") return false;
26697
- const models = state.models?.data;
26698
- if (!models) return false;
26699
- const found = models.find((m) => m.id === DEFAULT_MODEL);
26700
- if (!found) return false;
26701
- return found.capabilities?.supports?.tool_calls === true;
27219
+ return firstPresentInCatalog(DEFAULT_MODEL_CHAIN, { requireToolCalls: true }) != null;
26702
27220
  }
26703
27221
  /**
26704
27222
  * Gate for the compound L2 browser tools (`browser_act`, `browser_observe`,
@@ -26805,10 +27323,10 @@ function artifactToolsEnabled() {
26805
27323
  * browser is on disk. The browse agent drives the SAME Chrome/Edge
26806
27324
  * bridge as the raw `browser_*` tools, so it can't be useful without
26807
27325
  * that surface enabled.
26808
- * 2. The browse default model (`BROWSE_DEFAULT_MODEL`, `gpt-5.4-mini`)
27326
+ * 2. The browse default model (`BROWSE_DEFAULT_MODEL`, `gpt-5.6-luna`)
26809
27327
  * is in Copilot's live catalog AND `pickEndpoint()` resolves a
26810
27328
  * reachable endpoint for it. Unlike `workerToolsEnabled()` (which
26811
- * checks `tool_calls` on the gemini default), the browse default is
27329
+ * checks `tool_calls` on the shared gate sentinel), the browse default is
26812
27330
  * a `/responses`-only gpt-5.x model — `pickEndpoint` is the right
26813
27331
  * reachability probe (it returns undefined only when the model
26814
27332
  * serves neither chat nor responses).
@@ -27353,8 +27871,26 @@ function logTelemetry(t) {
27353
27871
  function toolAcceptsWorkspace(tool) {
27354
27872
  return tool.capability === "worker" || tool.toolNameHttp === "code" || tool.toolNameHttp === "run_workflow";
27355
27873
  }
27874
+ /**
27875
+ * Fold the per-session `X-GH-Workspace` header into `args.workspace` when the
27876
+ * caller left it empty, and REPORT which of the two the tool ended up with.
27877
+ *
27878
+ * The return value is the load-bearing part. This function mutates `args`, so
27879
+ * once it has run a header-derived workspace is byte-indistinguishable from one
27880
+ * the caller chose — and those two cases warrant very different treatment. A
27881
+ * caller that named a directory has told us where it is; a header is a
27882
+ * connection-level default that may be stale (it is computed by a helper Claude
27883
+ * Code runs, and the calling agent may since have moved into a git worktree).
27884
+ * `runWorkerToolCall` uses the distinction to decide what to tell the caller
27885
+ * about the tree it actually ran in, so the provenance must survive the merge.
27886
+ */
27356
27887
  function applySessionWorkspace(args, sessionWorkspace, tool) {
27357
- if ((!tool || toolAcceptsWorkspace(tool)) && typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace) && (args.workspace === void 0 || args.workspace === "")) args.workspace = sessionWorkspace;
27888
+ if (args.workspace !== void 0 && args.workspace !== "") return "argument";
27889
+ if ((!tool || toolAcceptsWorkspace(tool)) && typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace)) {
27890
+ args.workspace = sessionWorkspace;
27891
+ return "session";
27892
+ }
27893
+ return "absent";
27358
27894
  }
27359
27895
  async function handleToolsCall(body, scope, sessionWorkspace) {
27360
27896
  const params = body.params ?? {};
@@ -27388,7 +27924,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
27388
27924
  personaContext = typeof args.context === "string" ? args.context : void 0;
27389
27925
  if (args.imagePaths !== void 0) {
27390
27926
  if (!Array.isArray(args.imagePaths) || args.imagePaths.some((v) => typeof v !== "string")) return rpcError(body.id, RPC_INVALID_PARAMS, "tools/call: arguments.imagePaths must be an array of strings");
27391
- const loaded = await loadPeerImages(args.imagePaths, process.cwd());
27927
+ const imageRoot = typeof sessionWorkspace === "string" && sessionWorkspace.length > 0 && path.isAbsolute(sessionWorkspace) ? sessionWorkspace : process.cwd();
27928
+ const loaded = await loadPeerImages(args.imagePaths, imageRoot);
27392
27929
  if (!loaded.ok) return rpcError(body.id, RPC_INVALID_PARAMS, `tools/call: ${loaded.error}`);
27393
27930
  personaImages = loaded.images;
27394
27931
  }
@@ -27426,8 +27963,8 @@ async function handleToolsCall(body, scope, sessionWorkspace) {
27426
27963
  const telemetryName = persona ? persona.agentName : nonPersonaTool.toolNameHttp;
27427
27964
  const telemetryModel = persona ? persona.model : "(non-persona)";
27428
27965
  try {
27429
- if (nonPersonaTool) applySessionWorkspace(args, sessionWorkspace, nonPersonaTool);
27430
- const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal);
27966
+ const workspaceSource = nonPersonaTool ? applySessionWorkspace(args, sessionWorkspace, nonPersonaTool) : "absent";
27967
+ const result = persona ? await callPersona(persona, personaPrompt, personaContext, personaEffort, aborter?.signal, personaImages) : await nonPersonaTool.handler(args, aborter?.signal, { workspaceSource });
27431
27968
  logTelemetry({
27432
27969
  name: telemetryName,
27433
27970
  model: telemetryModel,
@@ -27753,6 +28290,98 @@ function handleMcpDelete(c) {
27753
28290
  return c.body(null, 200);
27754
28291
  }
27755
28292
  //#endregion
28293
+ //#region src/lib/reasoning-effort.ts
28294
+ /**
28295
+ * Reasoning-effort bucketing + clamping shared by the `/v1/messages` handler
28296
+ * (adaptive-thinking translation) and the Anthropic-translation shim
28297
+ * (thinking-budget → Responses `reasoning.effort`).
28298
+ *
28299
+ * Extracted from `src/routes/messages/handler.ts` so a `~/lib/*` module can
28300
+ * depend on it without importing route code (and without forming a
28301
+ * handler → shim → handler import cycle). `handler.ts` re-exports these for
28302
+ * backward compatibility with existing imports/tests.
28303
+ */
28304
+ /**
28305
+ * Copilot's reasoning-effort tiers, lowest to highest.
28306
+ *
28307
+ * Both ends were added after the fact and both are load-bearing:
28308
+ *
28309
+ * `none` is advertised by every gpt-5.x entry in the live catalog. While it was
28310
+ * missing here it was treated as an UNRECOGNIZED value, so a client asking for
28311
+ * the MINIMUM on a model that does not offer it (gemini advertises only
28312
+ * low/medium/high) was anchored at the unknown-value tier and clamped to
28313
+ * `high` — the maximum. Listing it makes the clamp resolve to `low` instead,
28314
+ * which is what "nearest supported tier" should always have meant.
28315
+ *
28316
+ * `max` is advertised by `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`,
28317
+ * `claude-opus-5` and the 4.7/4.8 Opus lines, and Claude Code's effort picker
28318
+ * offers it for any model whose entry allows it. Listing it lets an explicit
28319
+ * selection pass through, and lets `clampEffort` land on it for a model that
28320
+ * advertises nothing lower.
28321
+ *
28322
+ * `bucketEffort` deliberately reaches neither end — see below.
28323
+ */
28324
+ const EFFORT_ORDER = [
28325
+ "none",
28326
+ "low",
28327
+ "medium",
28328
+ "high",
28329
+ "xhigh",
28330
+ "max"
28331
+ ];
28332
+ /** Anchor for an effort value that is not a recognized tier at all.
28333
+ *
28334
+ * Deliberately NOT the top of `EFFORT_ORDER`. Callers reach this only when the
28335
+ * incoming value is unrecognized, which is a guess — and resolving a guess to
28336
+ * the most expensive tier a model advertises would silently spend more than the
28337
+ * caller could have meant. Anchoring here and clamping DOWN keeps the behavior
28338
+ * identical to before `max` joined the ladder, while `max` stays reachable by
28339
+ * explicit, valid selection. */
28340
+ const UNKNOWN_EFFORT_ANCHOR = "xhigh";
28341
+ /**
28342
+ * Bucket a thinking budget into a Copilot reasoning-effort string.
28343
+ * `<2000`→low, `<8000`→medium, `<24000`→high, else→xhigh.
28344
+ * Defaults missing/non-numeric budgets to 8000 ("high").
28345
+ *
28346
+ * The ceiling stays at `xhigh` even though `max` exists: Anthropic's
28347
+ * `budget_tokens` is unbounded above, so any threshold chosen for a `max`
28348
+ * bucket would silently re-tier existing callers whose budgets already map to
28349
+ * `xhigh`. `max` is reachable only by explicit selection
28350
+ * (`output_config.effort`), which is an unambiguous request rather than an
28351
+ * inference from a token count.
28352
+ */
28353
+ function bucketEffort(budget) {
28354
+ const n = typeof budget === "number" && Number.isFinite(budget) ? budget : 8e3;
28355
+ if (n < 2e3) return "low";
28356
+ if (n < 8e3) return "medium";
28357
+ if (n < 24e3) return "high";
28358
+ return "xhigh";
28359
+ }
28360
+ /**
28361
+ * Clamp a bucketed effort to the closest value in `supported`. Ties resolve to
28362
+ * the lower-tier option (per EFFORT_ORDER).
28363
+ *
28364
+ * Iterates EFFORT_ORDER (canonical low→xhigh) so the first match on a given
28365
+ * distance is always the lower-tier value, regardless of input order in
28366
+ * `supported`.
28367
+ */
28368
+ function clampEffort(bucketed, supported) {
28369
+ if (supported.includes(bucketed)) return bucketed;
28370
+ const targetIdx = EFFORT_ORDER.indexOf(bucketed);
28371
+ let best;
28372
+ let bestDist = Infinity;
28373
+ for (let i = 0; i < EFFORT_ORDER.length; i++) {
28374
+ const value = EFFORT_ORDER[i];
28375
+ if (!supported.includes(value)) continue;
28376
+ const dist = Math.abs(i - targetIdx);
28377
+ if (dist < bestDist) {
28378
+ bestDist = dist;
28379
+ best = value;
28380
+ }
28381
+ }
28382
+ return best ?? bucketed;
28383
+ }
28384
+ //#endregion
27756
28385
  //#region src/lib/stream-relay.ts
27757
28386
  const ENCODER$1 = new TextEncoder();
27758
28387
  /**
@@ -28240,10 +28869,15 @@ function rememberThinkingHistoryRepair(fingerprint) {
28240
28869
  * re-call Copilot for the next turn — stream onto the SAME
28241
28870
  * SSE connection (no new message_start; the original one is
28242
28871
  * still open). Loop up to ADVISOR_MAX_TURNS times.
28243
- * 4. Cross-lab default: route the advisor call to a different model
28244
- * family than the main loop (gpt-5.6-sol by default) so the user gets
28245
- * a true "second set of eyes" instead of Opus reviewing Opus
28246
- * (gemini-critic finding).
28872
+ * 4. Lead-aware model choice: route the advisor call to a different model
28873
+ * family than the main loop (gpt-5.6-sol) so the user gets a true "second
28874
+ * set of eyes" instead of Opus reviewing Opus (gemini-critic finding). When
28875
+ * the LEAD is a lighter Claude tier the choice inverts and the advisor
28876
+ * escalates to `ADVISOR_ESCALATION_MODEL` instead — see that constant for
28877
+ * why trading the cross-lab property is the right call on that path.
28878
+ * 5. Effort follows the Claude Code effort picker (`resolveAdvisorEffort`)
28879
+ * rather than a hardcoded constant, floored so a low picker cannot render
28880
+ * the consultation useless.
28247
28881
  *
28248
28882
  * The translate-loop is bounded to a single user request — no
28249
28883
  * persistent state across requests is needed (unlike Phase G's
@@ -28267,6 +28901,185 @@ const ADVISOR_CLIENT_TOOL_NAME = "advisor";
28267
28901
  * model — Opus 4.6/Sonnet 4.6 typically). */
28268
28902
  const ADVISOR_DEFAULT_MODEL = "gpt-5.6-sol";
28269
28903
  const ADVISOR_DEFAULT_EFFORT = "xhigh";
28904
+ /** The Anthropic frontier model the advisor escalates to when the LEAD is a
28905
+ * lighter Claude tier (sonnet, haiku).
28906
+ *
28907
+ * Selecting a lighter lead is a decision to work on a budget while holding
28908
+ * quality: the lead does the legwork and escalates for direction. Without this,
28909
+ * a budget lead has no transcript-aware path to the strongest Anthropic
28910
+ * reasoner at all — `opus_critic` is stateless and sees one artifact, and the
28911
+ * `plan` worker is read-only and never sees the transcript.
28912
+ *
28913
+ * This deliberately trades the advisor's cross-lab property on that path. The
28914
+ * advisor is not this repo's review instrument: it catches drift and momentum
28915
+ * and inherits the lead's framing by design, while the fresh-context critics
28916
+ * (`codex_critic`, `gemini_critic`, `codex_reviewer`, `gemini_reviewer`) are
28917
+ * the decorrelation instrument and are untouched. `GH_ROUTER_ADVISOR_MODEL`
28918
+ * keeps a cross-lab advisor one env var away for anyone who wants it back. */
28919
+ const ADVISOR_ESCALATION_MODEL = "claude-opus-5";
28920
+ /** Floor for the advisor's reasoning effort.
28921
+ *
28922
+ * The advisor follows the Claude Code effort picker (see `resolveAdvisorEffort`)
28923
+ * so dialing the picker down makes it cheaper, but it does NOT follow it all the
28924
+ * way to the bottom of `EFFORT_ORDER`. The advisor fires a handful of times per
28925
+ * session (`ADVISOR_MAX_TURNS`, typically 1-3), so it is not the budget line —
28926
+ * the lead's own turns are — while an advisor reasoning at `none`/`low` cannot
28927
+ * do the job the consultation exists for. The picker therefore governs the
28928
+ * `high..max` range. */
28929
+ const ADVISOR_MIN_EFFORT = "high";
28930
+ /** Output cap for the Anthropic-branch advisor call when the catalog carries no
28931
+ * limits for the resolved model. The value the branch used unconditionally
28932
+ * before it became reachable, kept so a catalog-less path is no worse off. */
28933
+ const ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS = 4096;
28934
+ /** Catalog spellings that mean the Responses API. Copilot is inconsistent about
28935
+ * the `/v1` prefix, so both are matched — mirroring `CHAT_ENDPOINTS` /
28936
+ * `RESPONSES_ENDPOINTS` in `src/services/copilot/endpoint.ts`. */
28937
+ const ADVISOR_RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
28938
+ /**
28939
+ * Which transport the advisor dispatches on: `/responses` (with
28940
+ * `reasoning.effort`) or `/v1/messages`.
28941
+ *
28942
+ * Catalog-first, name-regex second, and BOTH tests run against the bare id as
28943
+ * well as the given one. `pickEndpoint` is deliberately not reused: it answers
28944
+ * "chat or responses" for the two tool-calling clients and would send
28945
+ * `claude-opus-5` — which advertises `/v1/messages` AND `/chat/completions` — to
28946
+ * chat. The advisor's question is narrower: does this model serve `/responses`?
28947
+ *
28948
+ * The bare-id fallback is what makes `GH_ROUTER_ADVISOR_MODEL` safe. That pin is
28949
+ * accepted verbatim, so an operator can write a vendor-namespaced value like
28950
+ * `openai/gpt-5.6-sol`. Such an id is in no catalog and fails the start-anchored
28951
+ * name regex, so a catalog-only fix still posted it to `/v1/messages` and 400'd
28952
+ * — exported and directly tested for that exact input, because an earlier
28953
+ * version of this function claimed to handle it and did not.
28954
+ */
28955
+ function advisorUsesResponses(resolvedAdvisorModel) {
28956
+ const bare = resolvedAdvisorModel.slice(resolvedAdvisorModel.lastIndexOf("/") + 1);
28957
+ const endpoints = (state.models?.data?.find((m) => m.id === resolvedAdvisorModel || m.id === bare))?.supported_endpoints;
28958
+ if (endpoints && endpoints.length > 0) return endpoints.some((e) => ADVISOR_RESPONSES_ENDPOINTS.has(e));
28959
+ return /^(gpt-|o\d|.*codex)/i.test(bare);
28960
+ }
28961
+ /** True when the model advertises a usable reasoning-effort ladder. */
28962
+ function advertisedEffortLadder(resolvedAdvisorModel) {
28963
+ const supported = state.models?.data?.find((m) => m.id === resolvedAdvisorModel)?.capabilities?.supports?.reasoning_effort;
28964
+ return Array.isArray(supported) && supported.length > 0 ? supported : void 0;
28965
+ }
28966
+ /** True when the advisor should escalate to `ADVISOR_ESCALATION_MODEL` for this
28967
+ * lead: a Claude lead that is NOT already an Opus tier, on a catalog that
28968
+ * actually carries the escalation model.
28969
+ *
28970
+ * The catalog probe mirrors `standInToolEnabled`'s: never name a model the
28971
+ * account cannot reach. A non-Claude lead never gets here in practice (the
28972
+ * advisor tool is stripped for those before the request reaches this module),
28973
+ * but the check is explicit rather than assumed.
28974
+ *
28975
+ * The probe compares the BARE constant rather than `resolveModel`-ing it first,
28976
+ * which is deliberate and not an oversight: `claude-opus-5` is a single-segment
28977
+ * slug whose dashed and dotted spellings are identical, so resolution is a
28978
+ * no-op, and `resolveModel` WARNS on an id it cannot find — routing this probe
28979
+ * through it would emit that warning on every advisor request for anyone whose
28980
+ * catalog lacks opus-5, which is exactly the tier this returns false for.
28981
+ * `standInToolEnabled` compares the same id the same way. */
28982
+ function shouldEscalateAdvisor(leadModel) {
28983
+ if (!isBudgetClaudeLead(leadModel)) return false;
28984
+ return state.models?.data?.some((m) => m.id === "claude-opus-5") ?? false;
28985
+ }
28986
+ /**
28987
+ * Pick the advisor model for one request from the LEAD model that request is
28988
+ * running on.
28989
+ *
28990
+ * Resolved per request rather than at launch because the lead changes
28991
+ * mid-session via the `/model` picker; launch-time env plumbing would pin the
28992
+ * advisor to whatever was selected at spawn.
28993
+ *
28994
+ * Precedence:
28995
+ * 1. `GH_ROUTER_ADVISOR_MODEL` (trimmed) — the operator pin, checked first so
28996
+ * it works on every lead.
28997
+ * 2. A lighter Claude lead with the escalation model in the catalog.
28998
+ * 3. `ADVISOR_DEFAULT_MODEL`.
28999
+ *
29000
+ * Step 3 returns the LITERAL constant rather than walking the OpenAI frontier
29001
+ * chain. An Opus lead must resolve to exactly what it resolves to today, and a
29002
+ * frontier walk could yield `gpt-5.5` on a catalog missing `gpt-5.6-sol` —
29003
+ * a silent change to the one path that is required not to move.
29004
+ */
29005
+ /**
29006
+ * Map an operator pin onto the id the catalog actually carries.
29007
+ *
29008
+ * `GH_ROUTER_ADVISOR_MODEL` is free-form, and the natural thing to write is a
29009
+ * vendor-namespaced id like `openai/gpt-5.6-sol`. Copilot's catalog carries the
29010
+ * bare `gpt-5.6-sol`, so forwarding the namespaced form verbatim gets a 400
29011
+ * `model_not_supported` and the advisor silently degrades to its
29012
+ * "[Advisor unavailable: ...]" fallback — measured, not theorised: choosing the
29013
+ * transport correctly was NOT sufficient, because the id itself was still
29014
+ * wrong on the wire.
29015
+ *
29016
+ * An exact catalog hit wins first, so a real id containing a slash could never
29017
+ * be mangled. Only when the pin is absent from the catalog do we try its last
29018
+ * path segment, and only when THAT is present do we rewrite. A pin that matches
29019
+ * nothing is passed through untouched: the catalog may simply not be loaded
29020
+ * yet, and inventing an id would be worse than letting upstream reject it.
29021
+ */
29022
+ function normalizeAdvisorPin(pinned) {
29023
+ const models = state.models?.data;
29024
+ if (!models) return pinned;
29025
+ if (models.some((m) => m.id === pinned)) return pinned;
29026
+ const bare = pinned.slice(pinned.lastIndexOf("/") + 1);
29027
+ return bare !== pinned && models.some((m) => m.id === bare) ? bare : pinned;
29028
+ }
29029
+ function resolveAdvisorModel(leadModel) {
29030
+ const pinned = process.env.GH_ROUTER_ADVISOR_MODEL?.trim();
29031
+ if (pinned) return {
29032
+ model: normalizeAdvisorPin(pinned),
29033
+ escalated: false
29034
+ };
29035
+ if (leadModel && shouldEscalateAdvisor(leadModel)) return {
29036
+ model: ADVISOR_ESCALATION_MODEL,
29037
+ escalated: true
29038
+ };
29039
+ return {
29040
+ model: ADVISOR_DEFAULT_MODEL,
29041
+ escalated: false
29042
+ };
29043
+ }
29044
+ /**
29045
+ * Resolve the advisor's reasoning effort from the ORIGINAL request body, so the
29046
+ * advisor thinks at the level selected in the Claude Code effort picker instead
29047
+ * of a hardcoded constant.
29048
+ *
29049
+ * The source is the RAW pre-`resolveModelInBody` body, deliberately. By the time
29050
+ * the handler holds a parsed body, `translateThinking` has already bucketed
29051
+ * `thinking.budget_tokens` into `output_config.effort` AND clamped it to the
29052
+ * LEAD model's allowlist — so that value encodes "what the lead could do", not
29053
+ * "what the user picked". Re-clamping it against the advisor cannot recover the
29054
+ * difference: a `max` pick on a lead whose ceiling is `high` would reach an
29055
+ * xhigh-capable advisor as `high`.
29056
+ *
29057
+ * Precedence mirrors the repo-wide rule that an explicit client effort wins:
29058
+ * 1. `output_config.effort`
29059
+ * 2. `bucketEffort(thinking.budget_tokens)`
29060
+ * 3. `ADVISOR_DEFAULT_EFFORT` — a request expressing no preference behaves
29061
+ * exactly as it did before the picker was honored at all.
29062
+ *
29063
+ * Then floor, THEN clamp. The order is load-bearing: a model whose ceiling sits
29064
+ * below `ADVISOR_MIN_EFFORT` must still receive a value it accepts, so the clamp
29065
+ * is allowed to pull back under the floor. Flipping the two would forward an
29066
+ * effort upstream rejects.
29067
+ */
29068
+ function resolveAdvisorEffort(rawRequestBody, advisorModel) {
29069
+ let requested = ADVISOR_DEFAULT_EFFORT;
29070
+ if (rawRequestBody) try {
29071
+ const body = JSON.parse(rawRequestBody);
29072
+ const oc = body.output_config;
29073
+ const explicit = oc && typeof oc === "object" ? oc.effort : void 0;
29074
+ const thinking = body.thinking;
29075
+ if (typeof explicit === "string" && EFFORT_ORDER.includes(explicit)) requested = explicit;
29076
+ else if (thinking && typeof thinking === "object" && thinking.type === "enabled") requested = bucketEffort(thinking.budget_tokens);
29077
+ } catch {}
29078
+ const floored = EFFORT_ORDER.indexOf(requested) < EFFORT_ORDER.indexOf(ADVISOR_MIN_EFFORT) ? ADVISOR_MIN_EFFORT : requested;
29079
+ const supported = state.models?.data?.find((m) => m.id === resolveModel(advisorModel))?.capabilities?.supports?.reasoning_effort;
29080
+ if (!Array.isArray(supported) || supported.length === 0) return floored;
29081
+ return clampEffort(floored, supported);
29082
+ }
28270
29083
  /** ADVISOR_TOOL_INSTRUCTIONS verbatim from cc-backup
28271
29084
  * src/utils/advisor.ts — describes when the model should invoke
28272
29085
  * the advisor. Long-form prose; see source for justification. */
@@ -28362,8 +29175,12 @@ const ADVISOR_FALLBACK_MAX_TOKENS = 24e4;
28362
29175
  * our o200k count and Copilot's full-payload count. The transcript token
28363
29176
  * budget is `max_prompt_tokens - reserve`. Generous on purpose: a 400
28364
29177
  * `model_max_prompt_tokens_exceeded` degrades to a silent advisor
28365
- * fallback, and the marginal window we give up is irrelevant next to
28366
- * gpt-5.6-sol's ~1M. */
29178
+ * fallback, and the window given up is marginal against either advisor
29179
+ * model's real prompt window (`claude-opus-5` 936k, `gpt-5.6-sol` ~1M off
29180
+ * the live catalog). Sized as a fraction of the smaller of the two, not as
29181
+ * "irrelevant next to ~1M" — that framing assumed the advisor was always
29182
+ * the cheap side of the pair, which stopped being true once a budget lead
29183
+ * escalates to Opus. */
28367
29184
  const ADVISOR_PROMPT_TOKEN_RESERVE = 8e3;
28368
29185
  /**
28369
29186
  * Derive the TOKEN budget for the rendered transcript from the advisor
@@ -28474,9 +29291,9 @@ function truncateTailToUnits(text, maxUnits, measure) {
28474
29291
  * Anthropic's own ADVISOR ("see the whole task + every tool call +
28475
29292
  * every result").
28476
29293
  */
28477
- async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
29294
+ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, advisorEscalated = false) {
28478
29295
  if (signal?.aborted) throw new Error("advisor call aborted before dispatch");
28479
- const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance.";
29296
+ const advisorSystem = "You are an expert advisor reviewing an in-progress Claude Code session. The transcript below is the work-in-progress (turns numbered, with tool calls and results inlined). Read carefully and provide concrete, actionable advice on the next step or course-correction. Be specific — cite the parts of the transcript you're responding to. If the assistant is on the right track, say so explicitly. If they're stuck or off-track, name the specific assumption or step to revisit. Aim for 2-5 paragraphs of substantive guidance." + (advisorEscalated ? " The requesting agent is running a lighter, faster model than you. Give a directive recommendation and commit to the decision rather than laying out options for it to weigh." : "");
28480
29297
  const resolvedAdvisorModel = resolveModel(advisorModel);
28481
29298
  let measure;
28482
29299
  let maxUnits;
@@ -28491,7 +29308,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
28491
29308
  maxUnits = ADVISOR_MAX_CONVERSATION_CHARS;
28492
29309
  }
28493
29310
  const conversationText = renderConversationAsText(conversation, maxUnits, measure);
28494
- if (/^(gpt-|o\d|.*codex)/i.test(resolvedAdvisorModel)) {
29311
+ if (advisorUsesResponses(resolvedAdvisorModel)) {
28495
29312
  const payload = {
28496
29313
  model: resolvedAdvisorModel,
28497
29314
  instructions: advisorSystem,
@@ -28526,15 +29343,22 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal) {
28526
29343
  if (!text) throw new Error(`Advisor model ${resolvedAdvisorModel} returned empty assistant output`);
28527
29344
  return text;
28528
29345
  }
29346
+ const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
29347
+ const limits = advisorEntry?.capabilities?.limits;
29348
+ const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
28529
29349
  const advisorBody = JSON.stringify({
28530
29350
  model: resolvedAdvisorModel,
28531
- max_tokens: 4096,
29351
+ max_tokens: maxTokens,
28532
29352
  system: advisorSystem,
28533
29353
  messages: [{
28534
29354
  role: "user",
28535
29355
  content: conversationText
28536
29356
  }],
28537
- stream: false
29357
+ stream: false,
29358
+ ...advisorEntry?.capabilities?.supports?.adaptive_thinking ? {
29359
+ thinking: { type: "adaptive" },
29360
+ ...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
29361
+ } : {}
28538
29362
  });
28539
29363
  const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
28540
29364
  signal,
@@ -28589,6 +29413,7 @@ function sseEvent(type, data) {
28589
29413
  function buildAdvisorStream(opts) {
28590
29414
  const advisorModel = opts.advisorModel ?? "gpt-5.6-sol";
28591
29415
  const advisorEffort = opts.advisorEffort ?? "xhigh";
29416
+ const advisorEscalated = opts.advisorEscalated ?? false;
28592
29417
  const aborter = opts.externalAborter ?? new AbortController();
28593
29418
  let conversation = [...opts.initialConversation];
28594
29419
  return new ReadableStream({
@@ -28838,7 +29663,7 @@ function buildAdvisorStream(opts) {
28838
29663
  const advisorConversation = conversation;
28839
29664
  const advisorTexts = await Promise.all(advisorToolUses.map(async () => {
28840
29665
  try {
28841
- return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal);
29666
+ return await runAdvisor(advisorConversation, advisorModel, advisorEffort, aborter.signal, advisorEscalated);
28842
29667
  } catch (err) {
28843
29668
  if (aborter.signal.aborted) throw err;
28844
29669
  const msg = err instanceof Error ? err.message : String(err);
@@ -31561,32 +32386,33 @@ async function saveOverflowPatch(dir) {
31561
32386
  */
31562
32387
  const WORKTREE_REGISTRY = new WorktreeRegistry();
31563
32388
  registerExitHandlers$1(WORKTREE_REGISTRY);
31564
- /** Worker-availability GATE sentinel + final fallback. `gpt-5.4-mini` cheap,
31565
- * broadly-available, tool-call-capable, 400k-context, with tight
31566
- * function-calling-loop discipline (earlier gemini-flash cheap defaults
31567
- * early-stopped with empty turns on the function-calling loop; gpt-5.4-mini
31568
- * does not). Exported and aliased as `WORKER_DEFAULT_MODEL`:
31569
- * `workerToolsEnabled()` gates the ENTIRE worker surface on this id being
31570
- * present with `tool_calls`. It is no longer `explore`'s default (see
31571
- * `EXPLORE_DEFAULT_MODEL`) it stays the gate sentinel because it's the
31572
- * cheapest broadly-present tool-caller, and the fallback for any unmatched
31573
- * mode. */
31574
- const DEFAULT_MODEL = "gpt-5.4-mini";
32389
+ /** Worker-availability gate + unmatched-mode fallback chain. Luna leads where
32390
+ * the live catalog grants access: it is cheaper, faster, and has a larger context
32391
+ * window than mini. Mini remains the broad-tier fallback because individual-trial
32392
+ * and education catalogs may omit Luna. The catalog itself is the entitlement
32393
+ * signal; `billing.restricted_to` describes model policy, not the user's tier.
32394
+ *
32395
+ * `workerToolsEnabled()` admits the worker surface when either entry is present
32396
+ * with `tool_calls`. `resolveDefaultModel()` picks the first usable live entry for
32397
+ * an unmatched worker mode. Per-mode defaults remain independent of the gate. */
32398
+ const DEFAULT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gpt-5.4-mini"]);
31575
32399
  const DEFAULT_THINKING = "xhigh";
31576
- /** Default model for the READ-ONLY `explore` mode. `gemini-3.6-flash` at `high`
31577
- * (via `EXPLORE_DEFAULT_THINKING`; flash advertises no xhigh) — a fast, cheap,
31578
- * 1M-context tool-caller for read-only repo research. Routes over
31579
- * `/chat/completions` via the translation shim (the same proven path the
31580
- * `review` worker uses for gemini). Like `implement`'s gpt-5.6-sol this is NOT a
31581
- * `workerToolsEnabled` gate input if absent (e.g. a non-enterprise tier)
31582
- * `explore` errors helpfully at call time rather than vanishing the whole worker
31583
- * surface. The caller (the main model) overrides BOTH the model and the reasoning
31584
- * per call via the `model` / `thinking` args see the tier ladder (gpt-5.6-sol
31585
- * heavy / gpt-5.6-terra moderate / gemini-3.6-flash light) in the MCP tool desc. */
31586
- const EXPLORE_DEFAULT_MODEL = "gemini-3.6-flash";
31587
- /** Default thinking for `explore`. `high` (flash has no xhigh); explicit rather
31588
- * than inherited from `DEFAULT_THINKING` so the explore effort can't drift if the
31589
- * shared fallback changes. */
32400
+ function resolveDefaultModel() {
32401
+ const models = state.models?.data ?? [];
32402
+ return DEFAULT_MODEL_CHAIN.find((id) => models.some((model) => model.id === id && model.capabilities?.supports?.tool_calls === true)) ?? DEFAULT_MODEL_CHAIN[0];
32403
+ }
32404
+ /** Default model for the READ-ONLY `explore` mode. `gpt-5.6-luna` at `high`
32405
+ * (via `EXPLORE_DEFAULT_THINKING`) is the measured strict improvement over the
32406
+ * former Gemini Flash default: lower token cost, faster generation and tool-call
32407
+ * latency, a larger context window, and the full reasoning-effort ladder. `high`
32408
+ * is therefore a real selected tier rather than a clamp. Like `implement`'s
32409
+ * gpt-5.6-sol this per-mode default is NOT a `workerToolsEnabled` gate input — if
32410
+ * absent on a thin catalog, `explore` errors helpfully at call time rather than
32411
+ * vanishing the whole worker surface. The caller can override model and thinking
32412
+ * per call via the `model` / `thinking` args. */
32413
+ const EXPLORE_DEFAULT_MODEL = "gpt-5.6-luna";
32414
+ /** Default thinking for `explore`. Explicit rather than inherited from
32415
+ * `DEFAULT_THINKING` so the explore effort cannot drift with the fallback. */
31590
32416
  const EXPLORE_DEFAULT_THINKING = "high";
31591
32417
  /** Default model + thinking for the READ-ONLY `review` mode.
31592
32418
  * `gemini-3.1-pro-preview` at `xhigh` (clamped to `high` at call time — gemini
@@ -31603,41 +32429,44 @@ const EXPLORE_DEFAULT_THINKING = "high";
31603
32429
  const REVIEW_DEFAULT_MODEL = "gemini-3.1-pro-preview";
31604
32430
  const REVIEW_DEFAULT_THINKING = "xhigh";
31605
32431
  /** Default model + thinking for the READ+WRITE `implement` mode. `gpt-5.6-sol`
31606
- * at `xhigh` — the strongest reasoning tier in the catalog, 1M+ context,
31607
- * routed through `/responses` by the stream-fn endpoint split. Coding edits
31608
- * benefit from maximum reasoning; the higher per-call cost is justified for
31609
- * autonomous implementation. An explicit `opts.model` still wins. */
32432
+ * at `high` — a time-to-outcome default for the 1M+ context model, routed
32433
+ * through `/responses` by the stream-fn endpoint split. Any caller can restore
32434
+ * a higher tier per call via `thinking` or per session via `worker_defaults`;
32435
+ * precedence is per-call > session > built-in. */
31610
32436
  const IMPLEMENT_DEFAULT_MODEL = "gpt-5.6-sol";
31611
- const IMPLEMENT_DEFAULT_THINKING = "xhigh";
31612
- /** `test` starts with the same built-in pair as `implement`, but remains an
31613
- * independent mode so either can be overridden without affecting the other. */
32437
+ const IMPLEMENT_DEFAULT_THINKING = "high";
32438
+ /** `test` starts with the same time-to-outcome built-in pair as `implement`, but
32439
+ * remains independent so either mode can restore a higher tier per call via
32440
+ * `thinking` or per session via `worker_defaults`. Resolution precedence is
32441
+ * per-call > session > built-in. */
31614
32442
  const TEST_DEFAULT_MODEL = "gpt-5.6-sol";
31615
- const TEST_DEFAULT_THINKING = "xhigh";
31616
- /** Default model for `browse` mode. `gpt-5.4-mini` the Gate-B-winning
31617
- * browse model (small + fast enough to drive a tab at human pace, with
31618
- * enough tool-calling discipline to terminate). This is DISTINCT from the
31619
- * gemini worker `DEFAULT_MODEL`: browse is a different workload (drive a
31620
- * page, not read a repo) and was tuned separately. May be retuned after
31621
- * the flash-vs-mini eval settles. Routed through `/responses` by the
31622
- * stream-fn's endpoint split (it's a gpt-5.x model). Caller can override
32443
+ const TEST_DEFAULT_THINKING = "high";
32444
+ /** Default model for `browse` mode. `gpt-5.6-luna` has the same measured image
32445
+ * ceiling as `gpt-5.4-mini`, so screenshot-heavy sessions retain their input
32446
+ * capacity, and endpoint routing is derived from the live catalog. Luna's full
32447
+ * reasoning-effort ladder preserves the independent `high` browse default.
32448
+ * This is deliberately a per-mode default even though it also leads
32449
+ * `DEFAULT_MODEL_CHAIN`; the general worker gate remains tier-adaptive while
32450
+ * browse still applies its independent reachability gate. Caller can override
31623
32451
  * per call via the `model` arg.
31624
32452
  *
31625
32453
  * Exported so the MCP browse handler reads the same constant — drift
31626
32454
  * between the two would ship a tool whose docs disagree with its runtime
31627
32455
  * default. */
31628
- const BROWSE_DEFAULT_MODEL = "gpt-5.4-mini";
32456
+ const BROWSE_DEFAULT_MODEL = "gpt-5.6-luna";
31629
32457
  /** Default thinking for `browse`. Higher than the page-driving workload
31630
32458
  * strictly needs, but the termination discipline benefits from it. */
31631
32459
  const BROWSE_DEFAULT_THINKING = "high";
31632
- /** Default model + thinking for the read-only `plan` mode. `claude-opus-4.8`
31633
- * at `xhigh` planning is the highest-leverage read-only step (the plan
31634
- * shapes everything downstream), so it gets the strongest reasoning model
31635
- * rather than the lightweight `gemini-3.6-flash` explore default. Uses the DOTTED
31636
- * Copilot catalog id (the worker resolver exact-matches `catalog.id`, it does
31637
- * NOT translate the Anthropic dashed slug; `claude-opus-5` is a single-segment
31638
- * slug so dotted == dashed). Falls back to a helpful unknown-model error at call
31639
- * time if opus-5 isn't in the catalog (e.g. a non-enterprise tier), exactly like
31640
- * `implement`'s `gpt-5.6-sol`. Caller's `model` arg still wins. */
32460
+ /** Default model + thinking for the read-only `plan` mode. `claude-opus-5`
32461
+ * at `high` favours time-to-outcome while retaining the strongest planning
32462
+ * model rather than the lightweight `gpt-5.6-luna` explore default. Any
32463
+ * caller can restore a higher tier per call via `thinking` or per session via
32464
+ * `worker_defaults`; precedence is per-call > session > built-in. Uses the
32465
+ * DOTTED Copilot catalog id (the worker resolver exact-matches `catalog.id`, it
32466
+ * does NOT translate the Anthropic dashed slug; `claude-opus-5` is a
32467
+ * single-segment slug so dotted == dashed). Falls back to a helpful unknown-model
32468
+ * error at call time if opus-5 isn't in the catalog (e.g. a non-enterprise tier),
32469
+ * exactly like `implement`'s `gpt-5.6-sol`. */
31641
32470
  const PLAN_DEFAULT_MODEL = "claude-opus-5";
31642
32471
  const BUILT_IN_MODE_DEFAULTS = Object.freeze({
31643
32472
  explore: {
@@ -31650,7 +32479,7 @@ const BUILT_IN_MODE_DEFAULTS = Object.freeze({
31650
32479
  },
31651
32480
  plan: {
31652
32481
  model: PLAN_DEFAULT_MODEL,
31653
- thinking: "xhigh"
32482
+ thinking: "high"
31654
32483
  },
31655
32484
  implement: {
31656
32485
  model: IMPLEMENT_DEFAULT_MODEL,
@@ -31665,10 +32494,10 @@ const BUILT_IN_MODE_DEFAULTS = Object.freeze({
31665
32494
  thinking: BROWSE_DEFAULT_THINKING
31666
32495
  }
31667
32496
  });
31668
- /** Resolve the effective mode ladder without changing the gate sentinel. */
32497
+ /** Resolve the effective mode ladder without changing the worker gate. */
31669
32498
  function resolveModeDefaults(mode, ignoreSessionDefaults = false) {
31670
32499
  const builtIn = BUILT_IN_MODE_DEFAULTS[mode] ?? {
31671
- model: "gpt-5.4-mini",
32500
+ model: resolveDefaultModel(),
31672
32501
  thinking: DEFAULT_THINKING
31673
32502
  };
31674
32503
  const override = ignoreSessionDefaults ? {} : getWorkerSessionDefault(mode);
@@ -33720,7 +34549,7 @@ const PERSONAS_READ = Object.freeze([
33720
34549
  toolNameHttp: "codex_reviewer",
33721
34550
  model: "gpt-5.3-codex",
33722
34551
  endpoint: "/v1/responses",
33723
- description: "Line-level code reviewer backed by gpt-5.3-codex (OpenAI, ≈272K-token input window), a code-specialist reviewer that is fastest around high effort (~16s at high effort). It reviews concrete diffs, files, or function bodies and returns findings with severity, file:line locations, issue impact, and a minimal suggested fix. Use when the artifact is actual code and the goal is bug, edge-case, security, concurrency, resource, or idiom review. Not for architecture or tradeoff review, use codex_critic or gemini_critic; pass the diff or file content verbatim.",
34552
+ description: "Line-level code reviewer backed by gpt-5.3-codex (OpenAI, ≈272K-token input window), a code specialist for line-level review. It reviews concrete diffs, files, or function bodies and returns findings with severity, file:line locations, issue impact, and a minimal suggested fix. Use when the artifact is actual code and the goal is bug, edge-case, security, concurrency, resource, or idiom review. Not for architecture or tradeoff review, use codex_critic or gemini_critic; pass the diff or file content verbatim.",
33724
34553
  baseInstructions: REVIEWER_BASE,
33725
34554
  agentPrompt: "",
33726
34555
  writeCapable: false,
@@ -33912,13 +34741,8 @@ function buildPeerAwarenessSnippet(opts) {
33912
34741
  const codexCliClause = opts.codexCli ? " `mcp__codex-cli__codex` dispatches to `codex-implementer` (gpt-5.3-codex with workspace-write) for end-to-end coding tasks." : "";
33913
34742
  const para2Parts = [`\`mcp__${searchKey}__code\` is the one-stop code search (no extra model call). Its DEFAULT mode (or \`mode:"semantic"\`) ranks by MEANING via ColBERT over a per-workspace index, the first thing to reach for on intent/concept questions ("where is retry/backoff handled", "how does auth work"); when that index isn't ready it transparently falls back to lexical (the response \`source\` says which engine ran). Forced modes cover the rest: \`lexical\` (BM25F-ranked + tree-sitter, best for exact symbols), \`exact\`, \`regex\`, \`complete\` (exhaustive set), \`ast_pattern\`+\`ast_lang\` for multi-line AST shapes, \`scan\` for a whole-workspace symbol outline, \`multiline\` for cross-line regex. Multiple queries can run in a single turn. The index covers code-shaped files; for unstructured files (logs, \`.csv\`, \`.env*\`, config-only wiring), \`grep\`/\`glob\` still apply.`];
33914
34743
  if (opts.workerToolsAvailable) para2Parts.push(`\`worker-*\` are background Agent subagents (subagent_type) that run the matching worker in its own context and deliver the result as a completion notification, so a long run never blocks the turn: \`worker-explore\` (read-only research), \`worker-review\` (reads the code to verify a change or claim), \`worker-plan\` (ordered implementation plan), \`worker-implement\` (edit/write/bash; ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file; for in-place edits use the \`implementer\` subagent), \`worker-test\` (independent test author; also always worktree-isolated)${opts.browseAvailable ? ", `worker-browse` (autonomous browser agent driving a real browser)" : ""}. The raw \`mcp__${workersKey}__*\` tools they call are guarded (a direct main-thread call is redirected to the matching agent); Workers themselves have \`code_search\`.`);
33915
- const catchAllNames = [
33916
- opts.genericAvailable === false ? void 0 : "`generic`",
33917
- opts.genericFastAvailable === false ? void 0 : "`generic-fast`",
33918
- opts.genericCheapAvailable === false ? void 0 : "`generic-cheap`"
33919
- ].filter((n) => n != null);
33920
- const catchAllClause = catchAllNames.length > 0 ? ` Catch-alls on non-lead models, for work no specialist fits, cheapest last: ${catchAllNames.join(", ")}.` : "";
33921
- para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (you know what to build), \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
34744
+ const catchAllClause = opts.generalPurposeFastAvailable === false ? "" : " Catch-all on a fast, economical non-lead model for work no specialist fits: `general-purpose-fast`.";
34745
+ para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
33922
34746
  if (opts.workerToolsAvailable) para2Parts.push(`For a bounded, well-scoped implementation, prefer the \`implementer\` subagent over \`worker-implement\`; reach for \`worker-implement\` only when you specifically need git-worktree isolation, parallel variants, or a throwaway experiment.`);
33923
34747
  if (opts.workerToolsAvailable) para2Parts.push(`\`mcp__${orchestrateKey}__decompose\` composes an open-ended ask into a typed, VERIFIED workflow IR (a strong driver decorrelated by a cross-lab critic, so the decompose step isn't a single point of failure), and \`mcp__${orchestrateKey}__run_workflow\` executes that IR through a frozen kernel delivering max(orchestrated, baseline) over a sealed executable gate, so it never ships worse than a plain single-model run. \`mcp__${orchestrateKey}__verify_workflow\` checks an IR's floor invariants before you run it, and \`mcp__${orchestrateKey}__attest_step\` audits that a finished run's producers were each checked by a different lab. They suit non-trivial, role-separated asks; a trivial ask does not need them.`);
33924
34748
  else para2Parts.push(`\`mcp__${orchestrateKey}__verify_workflow\` statically checks a workflow IR's floor invariants and \`mcp__${orchestrateKey}__attest_step\` audits a run's cross-lab lineage (the \`decompose\`/\`run_workflow\` composer + kernel need the worker backend, unavailable here).`);
@@ -33951,15 +34775,27 @@ function buildPeerAwarenessSnippet(opts) {
33951
34775
  */
33952
34776
  function buildPeerAwarenessSummary(opts) {
33953
34777
  const key = (g) => opts.groupKeys?.[g] ?? GROUP_META[g].preferredKey;
33954
- const summaryCatchAlls = [
33955
- opts.genericAvailable === false ? void 0 : "`generic`",
33956
- opts.genericFastAvailable === false ? void 0 : "`generic-fast`",
33957
- opts.genericCheapAvailable === false ? void 0 : "`generic-cheap`"
34778
+ const renderNative = (name) => {
34779
+ const modelId = opts.nativeAgentModels?.[name];
34780
+ if (!modelId) return `\`${name}\``;
34781
+ const prices = catalogTokenPrices(modelId);
34782
+ const tps = indicativeTokensPerSecond(modelId);
34783
+ if (!prices || tps == null) return `\`${name}\``;
34784
+ return `\`${name}\` ${prices.in}/${prices.out} ~${tps}t/s`;
34785
+ };
34786
+ const summaryNativeNames = [
34787
+ renderNative("implementer"),
34788
+ opts.implementerFastAvailable === false ? void 0 : renderNative("implementer-fast"),
34789
+ renderNative("reviewer"),
34790
+ renderNative("brainstorm"),
34791
+ opts.scoutAvailable === false ? void 0 : renderNative("scout"),
34792
+ renderNative("scribe"),
34793
+ opts.generalPurposeFastAvailable === false ? void 0 : renderNative("general-purpose-fast")
33958
34794
  ].filter((n) => n != null);
33959
34795
  const lines = [
33960
34796
  "## Injected capabilities (summary)",
33961
34797
  "",
33962
- `Native subagents (Task), each in its own context: \`implementer\` (you know what to build), \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${summaryCatchAlls.length > 0 ? ` Catch-alls on non-lead models for work no specialist fits, cheapest last: ${summaryCatchAlls.join(", ")}.` : ""} They read the repo and can run things; the peer critics below cannot, so reach for \`reviewer\` when an assessment needs execution or repo context and for a critic when you already hold the artifact.`,
34798
+ `${summaryNativeNames.some((n) => /\d/.test(n)) ? "Native subagents (Task), own context. Cost is per 1M tokens in/out, tok/s approximate:" : "Native subagents (Task), each in its own context:"} ${summaryNativeNames.join(", ")}. Each agent's own description states when it applies. They read the repo and can run things; the peer critics below cannot, so reach for \`reviewer\` when an assessment needs execution or repo context and for a critic when you already hold the artifact.`,
33963
34799
  `A layer of MCP tools, background workers, and skills is injected into this session. Cross-lab peer critics under \`mcp__${key("peers")}__*\` (plus the \`peer-review-coordinator\` subagent) review plans and diffs adversarially, and Claude Code's built-in \`advisor\` catches approach drift. \`mcp__${key("search")}__code\` is meaning-first code search and \`mcp__${key("search")}__web\` returns citable web sources.`
33964
34800
  ];
33965
34801
  if (opts.workerToolsAvailable) lines.push(`Background \`worker-*\` agents (explore, review, plan, implement, test${opts.browseAvailable ? ", browse" : ""}) run delegated work in their own context without blocking your turn, and \`mcp__${key("orchestrate")}__*\` composes, verifies, and runs floor-raising workflows.`);
@@ -34262,7 +35098,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34262
35098
  toolNameHttp: "explore",
34263
35099
  group: "workers",
34264
35100
  capability: "worker",
34265
- description: "Runs as the background `worker-explore` agent. Dispatch via the Agent tool (subagent_type: worker-explore) so the turn is never blocked; the result arrives as a completion notification. Read-only investigation by an autonomous worker (Pi runtime; default model `gemini-3.6-flash` at high reasoning, override via the `model` arg with any Copilot-catalog model that advertises `tool_calls`). It has read, glob, grep, semantic-first code search, web search, fetch_url, advisor, update_plan, and read-only toolbelt tools, and it returns a single text answer. Use for bounded research, repo discovery, dependency investigation, or multi-file reading that would otherwise consume the lead context window. Not for implementation, test authoring, or verification of a concrete diff; use implement, test, or review for those scopes. Brief the investigation goal and constraints, not step-by-step tool semantics. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35101
+ description: "Runs as the background `worker-explore` agent. Dispatch via the Agent tool (subagent_type: worker-explore) so the turn is never blocked; the result arrives as a completion notification. Read-only investigation by an autonomous worker (Pi runtime; default model `gpt-5.6-luna` at high reasoning, override via the `model` arg with any Copilot-catalog model that advertises `tool_calls`). It has read, glob, grep, semantic-first code search, web search, fetch_url, advisor, update_plan, and read-only toolbelt tools, and it returns a single text answer. Use for bounded research, repo discovery, dependency investigation, or multi-file reading that would otherwise consume the lead context window. Not for implementation, test authoring, or verification of a concrete diff; use implement, test, or review for those scopes. Brief the investigation goal and constraints, not step-by-step tool semantics. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34266
35102
  inputSchema: {
34267
35103
  type: "object",
34268
35104
  required: ["prompt"],
@@ -34274,7 +35110,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34274
35110
  },
34275
35111
  model: {
34276
35112
  type: "string",
34277
- description: "Optional Copilot catalog model id (defaults to gemini-3.6-flash). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gemini-3.6-flash` (light/cheap) — all 1M context; pair with thinking:'high'."
35113
+ description: "Optional Copilot catalog model id (defaults to gpt-5.6-luna). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
34278
35114
  },
34279
35115
  thinking: {
34280
35116
  type: "string",
@@ -34283,7 +35119,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34283
35119
  },
34284
35120
  workspace: {
34285
35121
  type: "string",
34286
- description: "Optional absolute path to the workspace the worker operates in. Defaults to the proxy's launch cwd. Use this when the parent agent has multiple workspaces open and the worker must operate in a specific one. Must be absolute (relative paths rejected)."
35122
+ description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
34287
35123
  },
34288
35124
  maxWallClockMs: {
34289
35125
  type: "integer",
@@ -34291,11 +35127,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34291
35127
  }
34292
35128
  }
34293
35129
  },
34294
- async handler(args, signal) {
35130
+ async handler(args, signal, ctx) {
34295
35131
  return runWorkerToolCall({
34296
35132
  mode: "explore",
34297
35133
  args,
34298
- signal
35134
+ signal,
35135
+ ctx
34299
35136
  });
34300
35137
  }
34301
35138
  },
@@ -34303,7 +35140,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34303
35140
  toolNameHttp: "implement",
34304
35141
  group: "workers",
34305
35142
  capability: "worker",
34306
- description: "Runs as the background `worker-implement` agent. Dispatch via the Agent tool (subagent_type: worker-implement) so the turn is never blocked; the result arrives as a completion notification. Delegates a scoped coding task to an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at xhigh reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the explore read-only tools plus edit, write, bash, and codex_review, and it returns its final text with any changed files or worktree diff. Use for bounded implementation work that may take a while or benefits from isolated worker context. Not for pure research, planning, review, or independent test authoring; use explore, plan, review, or test for those scopes. ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place edits, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35143
+ description: "Runs as the background `worker-implement` agent. Dispatch via the Agent tool (subagent_type: worker-implement) so the turn is never blocked; the result arrives as a completion notification. Delegates a scoped coding task to an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the explore read-only tools plus edit, write, bash, and codex_review, and it returns its final text with any changed files or worktree diff. Use for bounded implementation work that may take a while or benefits from isolated worker context. Not for pure research, planning, review, or independent test authoring; use explore, plan, review, or test for those scopes. ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place edits, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34307
35144
  inputSchema: {
34308
35145
  type: "object",
34309
35146
  required: ["prompt"],
@@ -34319,7 +35156,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34319
35156
  },
34320
35157
  model: {
34321
35158
  type: "string",
34322
- description: "Optional Copilot catalog model id (defaults to gpt-5.6-sol). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gemini-3.6-flash` (light/cheap) — all 1M context; pair with thinking:'high'."
35159
+ description: "Optional Copilot catalog model id (defaults to gpt-5.6-sol). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
34323
35160
  },
34324
35161
  thinking: {
34325
35162
  type: "string",
@@ -34328,7 +35165,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34328
35165
  },
34329
35166
  workspace: {
34330
35167
  type: "string",
34331
- description: "Optional absolute path to the workspace the worker operates in. Defaults to the proxy's launch cwd. Use this when the parent agent has multiple workspaces open and the worker must operate in a specific one. Must be absolute (relative paths rejected). Must be inside a git repo (implement always runs in a worktree)."
35168
+ description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (implement always runs in a worktree)."
34332
35169
  },
34333
35170
  maxWallClockMs: {
34334
35171
  type: "integer",
@@ -34336,11 +35173,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34336
35173
  }
34337
35174
  }
34338
35175
  },
34339
- async handler(args, signal) {
35176
+ async handler(args, signal, ctx) {
34340
35177
  return runWorkerToolCall({
34341
35178
  mode: "implement",
34342
35179
  args,
34343
- signal
35180
+ signal,
35181
+ ctx
34344
35182
  });
34345
35183
  }
34346
35184
  },
@@ -34348,7 +35186,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34348
35186
  toolNameHttp: "review",
34349
35187
  group: "workers",
34350
35188
  capability: "worker",
34351
- description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification. Read-only code review by an autonomous worker (Pi runtime; default model `gemini-3.1-pro-preview`, default thinking xhigh clamped to high for that model, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read-only toolset as explore and verifies claims against surrounding repository context before returning severity-ranked findings with `file:line` citations. Use for reviewing a change, diff, or correctness claim when the reviewer should read the code itself rather than trusting a pasted artifact. Not for architecture critique, implementation, or test authoring; use codex_critic or gemini_critic for design review, implement for edits, and test for independent test creation. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35189
+ description: "Runs as the background `worker-review` agent. Dispatch via the Agent tool (subagent_type: worker-review) so the turn is never blocked; the result arrives as a completion notification. Code review by an autonomous worker (Pi runtime; default model `gemini-3.1-pro-preview`, default thinking xhigh clamped to high for that model, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has explore's read-and-search tools PLUS `bash`, so it verifies claims by running them reproducing a failure or running the build or suite — rather than only reading, and returns severity-ranked findings with `file:line` citations. It gets no edit/write tools unless you pass `worktree: true`, but `bash` runs real commands in the workspace, so a build or test it invokes can touch the tree. Use for reviewing a change, diff, or correctness claim when the reviewer should read the code itself rather than trusting a pasted artifact. Not for architecture critique, implementation, or test authoring; use codex_critic or gemini_critic for design review, implement for edits, and test for independent test creation. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34352
35190
  inputSchema: {
34353
35191
  type: "object",
34354
35192
  required: ["prompt"],
@@ -34360,16 +35198,20 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34360
35198
  },
34361
35199
  model: {
34362
35200
  type: "string",
34363
- description: "Optional Copilot catalog model id (defaults to gemini-3.1-pro-preview). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gemini-3.6-flash` (light/cheap) — all 1M context; pair with thinking:'high'."
35201
+ description: "Optional Copilot catalog model id (defaults to gemini-3.1-pro-preview). Must advertise tool_calls support; the engine emits an isError envelope listing the eligible catalog models on mismatch. Override by task weight: `gpt-5.6-sol` (heavy/deep), `gpt-5.6-terra` (moderate), `gpt-5.6-luna` (light/cheap) — all 1M context; pair with thinking:'high'."
34364
35202
  },
34365
35203
  thinking: {
34366
35204
  type: "string",
34367
35205
  enum: WORKER_THINKING_LEVELS,
34368
35206
  description: "Optional reasoning depth (defaults to xhigh, clamped to high for the default review model). Silently clamped to the model's allowed range; \"off\" drops the parameter entirely."
34369
35207
  },
35208
+ worktree: {
35209
+ type: "boolean",
35210
+ description: "Optional. When true, the review runs in an isolated git worktree replaying the workspace's working tree (dirty tracked changes and untracked-not-ignored files), which additionally grants `edit`/`write` so the reviewer can author a throwaway probe test to prove a claim. Default false: the reviewer reads and runs commands in the workspace itself. Prefer the default when verifying needs the build to work — a fresh worktree does not carry IGNORED files, so installed dependencies are absent. Requires a git repository."
35211
+ },
34370
35212
  workspace: {
34371
35213
  type: "string",
34372
- description: "Optional absolute path to the workspace the worker operates in. Defaults to the proxy's launch cwd. Use this when the parent agent has multiple workspaces open and the worker must operate in a specific one. Must be absolute (relative paths rejected)."
35214
+ description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
34373
35215
  },
34374
35216
  maxWallClockMs: {
34375
35217
  type: "integer",
@@ -34377,11 +35219,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34377
35219
  }
34378
35220
  }
34379
35221
  },
34380
- async handler(args, signal) {
35222
+ async handler(args, signal, ctx) {
34381
35223
  return runWorkerToolCall({
34382
35224
  mode: "review",
34383
35225
  args,
34384
- signal
35226
+ signal,
35227
+ ctx
34385
35228
  });
34386
35229
  }
34387
35230
  },
@@ -34389,7 +35232,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34389
35232
  toolNameHttp: "plan",
34390
35233
  group: "workers",
34391
35234
  capability: "worker",
34392
- description: "Runs as the background `worker-plan` agent. Dispatch via the Agent tool (subagent_type: worker-plan) so the turn is never blocked; the result arrives as a completion notification. Read-only implementation planning by an autonomous worker (Pi runtime; default model `claude-opus-5` at xhigh reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read-only toolset as explore and returns a concrete, ordered implementation plan covering files, approach, risks, and how acceptance criteria will be verified. Use before coding when the task needs repo-grounded sequencing or acceptance criteria translated into implementation steps. Not for editing files, running an implementation, writing tests, or adversarial review; use implement, test, or review for those scopes. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35235
+ description: "Runs as the background `worker-plan` agent. Dispatch via the Agent tool (subagent_type: worker-plan) so the turn is never blocked; the result arrives as a completion notification. Read-only implementation planning by an autonomous worker (Pi runtime; default model `claude-opus-5` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read-only toolset as explore and returns a concrete, ordered implementation plan covering files, approach, risks, and how acceptance criteria will be verified. Use before coding when the task needs repo-grounded sequencing or acceptance criteria translated into implementation steps. Not for editing files, running an implementation, writing tests, or adversarial review; use implement, test, or review for those scopes. Strictly read-only: it changes nothing on disk, and its returned text is the only artifact it produces — route to `implement` or `test` when a file actually needs to change. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34393
35236
  inputSchema: {
34394
35237
  type: "object",
34395
35238
  required: ["prompt"],
@@ -34410,7 +35253,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34410
35253
  },
34411
35254
  workspace: {
34412
35255
  type: "string",
34413
- description: "Optional absolute path to the workspace the worker operates in. Defaults to the proxy's launch cwd. Use this when the parent agent has multiple workspaces open and the worker must operate in a specific one. Must be absolute (relative paths rejected)."
35256
+ description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected)."
34414
35257
  },
34415
35258
  maxWallClockMs: {
34416
35259
  type: "integer",
@@ -34418,11 +35261,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34418
35261
  }
34419
35262
  }
34420
35263
  },
34421
- async handler(args, signal) {
35264
+ async handler(args, signal, ctx) {
34422
35265
  return runWorkerToolCall({
34423
35266
  mode: "plan",
34424
35267
  args,
34425
- signal
35268
+ signal,
35269
+ ctx
34426
35270
  });
34427
35271
  }
34428
35272
  },
@@ -34430,7 +35274,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34430
35274
  toolNameHttp: "test",
34431
35275
  group: "workers",
34432
35276
  capability: "worker",
34433
- description: "Runs as the background `worker-test` agent. Dispatch via the Agent tool (subagent_type: worker-test) so the turn is never blocked; the result arrives as a completion notification. Independent adversarial test authoring by an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at xhigh reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read/write toolset as implement and writes tests that try to break the implementation through edge cases, error paths, and acceptance criteria, then runs them and reports pass/fail. Use when a separate test author should challenge an implementation without modifying the production code to make tests pass. Not for implementing fixes, broad research, or code review; use implement, explore, or review for those scopes. ALWAYS runs in an isolated git worktree and returns the test diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place test authoring, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35277
+ description: "Runs as the background `worker-test` agent. Dispatch via the Agent tool (subagent_type: worker-test) so the turn is never blocked; the result arrives as a completion notification. Independent adversarial test authoring by an autonomous worker (Pi runtime; default model `gpt-5.6-sol` at high reasoning, override via `model` with any Copilot-catalog model that advertises `tool_calls`). It has the same read/write toolset as implement and writes tests that try to break the implementation through edge cases, error paths, and acceptance criteria, then runs them and reports pass/fail. Use when a separate test author should challenge an implementation without modifying the production code to make tests pass. Not for implementing fixes, broad research, or code review; use implement, explore, or review for those scopes. ALWAYS runs in an isolated git worktree and returns the test diff via a saved patch file (a `--stat` summary + a bounded preview + the patch path; a small diff is inlined in full) — it never edits your working tree, and it HARD-ERRORS if the workspace is not a git repository. For in-place test authoring, use the native `implementer` subagent. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34434
35278
  inputSchema: {
34435
35279
  type: "object",
34436
35280
  required: ["prompt"],
@@ -34455,7 +35299,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34455
35299
  },
34456
35300
  workspace: {
34457
35301
  type: "string",
34458
- description: "Optional absolute path to the workspace the worker operates in. Defaults to the proxy's launch cwd. Use this when the parent agent has multiple workspaces open and the worker must operate in a specific one. Must be absolute (relative paths rejected). Must be inside a git repo (test always runs in a worktree)."
35302
+ description: "Absolute path of the directory the worker operates in. Pass YOUR OWN current working directory unless you were told to target a different one — the proxy is a separate, long-lived process and cannot see where you are, so it falls back to your session's directory, which is not necessarily yours (a sub-agent or a git worktree moves you without moving the session). Omitting it when the connection supplies nothing is an error rather than a guess. Must be absolute (relative paths rejected). Must be inside a git repo (test always runs in a worktree)."
34459
35303
  },
34460
35304
  maxWallClockMs: {
34461
35305
  type: "integer",
@@ -34463,11 +35307,12 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34463
35307
  }
34464
35308
  }
34465
35309
  },
34466
- async handler(args, signal) {
35310
+ async handler(args, signal, ctx) {
34467
35311
  return runWorkerToolCall({
34468
35312
  mode: "test",
34469
35313
  args,
34470
- signal
35314
+ signal,
35315
+ ctx
34471
35316
  });
34472
35317
  }
34473
35318
  },
@@ -34677,7 +35522,7 @@ const NON_PERSONA_MCP_TOOLS = Object.freeze([
34677
35522
  toolNameHttp: "browse",
34678
35523
  group: "workers",
34679
35524
  capability: "browse_agent",
34680
- description: "Runs as the background `worker-browse` agent. Dispatch via the Agent tool (subagent_type: worker-browse) so the turn is never blocked; the result arrives as a completion notification. A Pi-driven autonomous browser worker (default model `gpt-5.4-mini`) drives a real browser to accomplish `task`, keeps raw DOM and page snapshots inside its own context, and returns a single text result. Use for delegated multi-step web tasks such as comparing prices, logging into a dashboard, or summarizing pages when the lead does not need to steer each click. Not for direct in-context browser control, screenshots, or precise element interactions; use the `browser` MCP tools for those. Pass `sessionId` to continue a prior browse session, or omit it for a fresh isolated session; multiple calls run as parallel sessions on the shared browser. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
35525
+ description: "Runs as the background `worker-browse` agent. Dispatch via the Agent tool (subagent_type: worker-browse) so the turn is never blocked; the result arrives as a completion notification. A Pi-driven autonomous browser worker (default model `gpt-5.6-luna`) drives a real browser to accomplish `task`, keeps raw DOM and page snapshots inside its own context, and returns a single text result. Use for delegated multi-step web tasks such as comparing prices, logging into a dashboard, or summarizing pages when the lead does not need to steer each click. Not for direct in-context browser control, screenshots, or precise element interactions; use the `browser` MCP tools for those. Pass `sessionId` to continue a prior browse session, or omit it for a fresh isolated session; multiple calls run as parallel sessions on the shared browser. If the result is too large to relay it is truncated to a preview and the FULL text is written to a file, whose absolute path is in the returned text; read that file rather than re-running the worker.",
34681
35526
  inputSchema: {
34682
35527
  type: "object",
34683
35528
  required: ["task"],
@@ -34790,10 +35635,12 @@ function assertMcpToolSurfaceConsistent() {
34790
35635
  /**
34791
35636
  * Shared closure body for the two worker MCP tools. Validates the
34792
35637
  * minimal arg shape (prompt required + optional knobs typed), then
34793
- * forwards to `runWorkerAgent`. Outside serve mode, `workspace` defaults
34794
- * to the proxy's launch cwd; serve mode requires an explicit/header-derived
34795
- * workspace. Callers can override via the optional `workspace` arg
34796
- * (absolute paths only enforced here). The engine performs every
35638
+ * forwards to `runWorkerAgent`. `workspace` comes from the caller's
35639
+ * argument or, failing that, the per-connection session header the
35640
+ * boundary folds in; with neither, the call is REFUSED rather than
35641
+ * defaulted to the proxy's launch cwd (see the resolution block below
35642
+ * for why that default was a bug, and `GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE`
35643
+ * for the raw-client escape hatch). The engine performs every
34797
35644
  * deeper validation (model existence, thinking clamp, worktree
34798
35645
  * provisioning, semaphore acquisition, workspace realpath +
34799
35646
  * accessibility) and never throws — its `{text, isError?}` envelope
@@ -34806,7 +35653,7 @@ function assertMcpToolSurfaceConsistent() {
34806
35653
  * client that ignores the schema.
34807
35654
  */
34808
35655
  async function runWorkerToolCall(call) {
34809
- const { mode, args, signal } = call;
35656
+ const { mode, args, signal, ctx } = call;
34810
35657
  const prompt = typeof args.prompt === "string" ? args.prompt : "";
34811
35658
  if (!prompt) return {
34812
35659
  content: [{
@@ -34838,7 +35685,7 @@ async function runWorkerToolCall(call) {
34838
35685
  }
34839
35686
  let worktree;
34840
35687
  let worktreeNote = "";
34841
- if (mode === "implement" || mode === "test") {
35688
+ if (mode === "implement" || mode === "test" || mode === "review") {
34842
35689
  if (args.worktree !== void 0 && typeof args.worktree !== "boolean") return {
34843
35690
  content: [{
34844
35691
  type: "text",
@@ -34846,11 +35693,16 @@ async function runWorkerToolCall(call) {
34846
35693
  }],
34847
35694
  isError: true
34848
35695
  };
34849
- worktree = true;
34850
- if (args.worktree === false) worktreeNote = `[note: worker_${mode} always runs in an isolated git worktree; the requested worktree:false was overridden. For in-place edits, use the \`implementer\` subagent.]
35696
+ if (mode === "review") worktree = args.worktree === true ? true : void 0;
35697
+ else {
35698
+ worktree = true;
35699
+ if (args.worktree === false) worktreeNote = `[note: worker_${mode} always runs in an isolated git worktree; the requested worktree:false was overridden. For in-place edits, use the \`implementer\` subagent.]
34851
35700
 
34852
35701
  `;
35702
+ }
34853
35703
  }
35704
+ const allowProxyCwd = process.env.GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE === "1";
35705
+ const callerGaveWorkspace = args.workspace !== void 0;
34854
35706
  let workspace;
34855
35707
  if (args.workspace !== void 0) {
34856
35708
  if (typeof args.workspace !== "string" || args.workspace.length === 0) return {
@@ -34875,7 +35727,16 @@ async function runWorkerToolCall(call) {
34875
35727
  }],
34876
35728
  isError: true
34877
35729
  };
34878
- else workspace = process.cwd();
35730
+ else if (allowProxyCwd) workspace = process.cwd();
35731
+ else return {
35732
+ content: [{
35733
+ type: "text",
35734
+ text: `worker_${mode}: a workspace is required. Nothing in this call said which directory to run in, and the proxy's own launch directory is not a safe guess — it is where the proxy was started, which may be a different checkout or git worktree than the one you are working in. Re-issue this call with \`workspace\` set to the absolute path of your current working directory. (Operators running a raw MCP client that cannot send one can restore the old launch-cwd default with GH_ROUTER_ALLOW_PROXY_CWD_WORKSPACE=1.)`
35735
+ }],
35736
+ isError: true
35737
+ };
35738
+ const effectiveSource = ctx?.workspaceSource ?? (callerGaveWorkspace ? "argument" : "absent");
35739
+ const workspaceNote = effectiveSource === "argument" ? "" : `[workspace: ${workspace} (${effectiveSource === "session" ? "from your session's working directory; pass `workspace` explicitly if you are running somewhere else, such as a git worktree" : "the proxy's launch directory"})]\n\n`;
34879
35740
  let maxWallClockMs;
34880
35741
  let clampNote = "";
34881
35742
  if (args.maxWallClockMs !== void 0) {
@@ -34903,7 +35764,7 @@ async function runWorkerToolCall(call) {
34903
35764
  maxWallClockMs,
34904
35765
  signal
34905
35766
  });
34906
- const notePrefix = `${clampNote}${worktreeNote}`;
35767
+ const notePrefix = `${workspaceNote}${clampNote}${worktreeNote}`;
34907
35768
  return {
34908
35769
  content: [{
34909
35770
  type: "text",
@@ -35136,6 +35997,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
35136
35997
  return [...new Set(names)];
35137
35998
  }
35138
35999
  //#endregion
35139
- export { browserToolsEnabled as $, searchWeb as A, CONDENSED_OPERATING_SEQUENCE as At, buildAnthropicErrorEvent as B, UPSTREAM_INACTIVITY_TIMEOUT_MS as Bt, buildToolbeltAwareness as C, provisionBrowserAssets as Ct, TOOLBELT_TOOLS$1 as D, extractTarGzMember as Dt, vscodeRipgrepPath as E, provisionAndIndexColbert as Et, isAdvisorRequested as F, DEFAULT_CLAUDE_MODEL_FALLBACKS as Ft, relayAnthropicStream as G, withOneMSuffix as Gt, isControllerClosedError as H, pickClaudeDefault as Ht, formatThinkingRepairDecline as I, DEFAULT_CODEX_MODEL as It, agentToolsEnabled as J, handleMcpDelete as K, withInstallLock as Kt, rememberThinkingHistoryRepair as L, DEFAULT_CODEX_MODEL_FALLBACKS as Lt, ADVISOR_TOOL_INSTRUCTIONS as M, shouldUseInsecureTls as Mt, buildAdvisorStream as N, collapsePathKeys as Nt, assetFor as O, extractZipMember as Ot, injectAdvisorTool as P, toolbeltPathOverride as Pt, browserCompoundToolsEnabled as Q, repairKnownThinkingHistory as R, DEFAULT_PORT as Rt, availableToolCommands as S, parseJsonOrDiagnose as St, toolbeltSkipSet as T, colbertDegradedWarning as Tt, logStreamError as U, upstreamAllowH2 as Ut, buildOpenAIErrorEvent as V, generateRandomPort as Vt, readIteratorWithTimeout as W, upstreamMaxConnections as Wt, brainstormModel as X, artifactToolsEnabled as Y, browseAgentEnabled as Z, appendPlanReminder as _, pickEndpoint as _t, buildPeerAwarenessSnippet as a, nativeSubagentModel as at, runWorkerAgent as b, MAX_RESPONSE_BODY_BYTES as bt, personasFor as c, scribeModel as ct, EXPLORE_DEFAULT_MODEL as d, shimDefaultsToXhigh as dt, fleetToolsEnabled as et, EXPLORE_DEFAULT_THINKING as f, countTokens as ft, TEST_DEFAULT_MODEL as g, resolveMcpToolTimeoutMs as gt, REVIEW_DEFAULT_MODEL as h, assembleResponsesPayload as ht, buildAgentPrompt as i, genericModel as it, ADVISOR_INTERNAL_TOOL_NAME as j, DEFINITION_OF_GREATNESS as jt, satisfiesMinVersion as k, warmTreeSitterPool as kt, BROWSE_DEFAULT_MODEL as l, standInToolEnabled as lt, PLAN_DEFAULT_MODEL as m, getTokenCount as mt, MCP_GROUPS as n, genericCheapModel as nt, buildPeerAwarenessSummary as o, reviewerModel as ot, IMPLEMENT_DEFAULT_MODEL as p, createMessages as pt, handleMcpPost as q, assertMcpToolSurfaceConsistent as r, genericFastModel as rt, enumerateInjectedMcpToolNames as s, scoutModel as st, GROUP_META as t, geminiAvailable as tt, DEFAULT_MODEL as u, workerToolsEnabled as ut, resolveModeDefaults as v, createResponses as vt, toolbeltEnabled as w, hasSupportedBrowserInstalled as wt, buildEnv as x, readResponseBodyCapped as xt, resolveWorkerRunOpts as y, createChatCompletions as yt, repairRejectedThinkingHistory as z, UPSTREAM_FETCH_TIMEOUT_MS as zt };
36000
+ export { handleMcpDelete as $, resolveLeadSlugArg as $t, satisfiesMinVersion as A, hasSupportedBrowserInstalled as At, rememberThinkingHistoryRepair as B, collapsePathKeys as Bt, availableToolCommands as C, resolveMcpToolTimeoutMs as Ct, vscodeRipgrepPath as D, readResponseBodyCapped as Dt, toolbeltSkipSet as E, MAX_RESPONSE_BODY_BYTES as Et, injectAdvisorTool as F, warmTreeSitterPool as Ft, isControllerClosedError as G, DEFAULT_CODEX_MODEL as Gt, repairRejectedThinkingHistory as H, BUDGET_SMALL_FAST_CATALOG_ID as Ht, isAdvisorRequested as I, provisionTreeSitterAssets as It, relayAnthropicStream as J, UPSTREAM_FETCH_TIMEOUT_MS as Jt, logStreamError as K, DEFAULT_CODEX_MODEL_FALLBACKS as Kt, resolveAdvisorEffort as L, CONDENSED_OPERATING_SEQUENCE as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, provisionAndIndexColbert as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, extractTarGzMember as Nt, TOOLBELT_TOOLS$1 as O, parseJsonOrDiagnose as Ot, buildAdvisorStream as P, extractZipMember as Pt, clampEffort as Q, pickClaudeDefault as Qt, resolveAdvisorModel as R, DEFINITION_OF_GREATNESS as Rt, buildEnv as S, warnOnTokenPriceDrift as St, toolbeltEnabled as T, createChatCompletions as Tt, buildAnthropicErrorEvent as U, BUDGET_SMALL_FAST_SLUG as Ut, repairKnownThinkingHistory as V, toolbeltPathOverride as Vt, buildOpenAIErrorEvent as W, DEFAULT_CLAUDE_MODEL_FALLBACKS as Wt, UNKNOWN_EFFORT_ANCHOR as X, generateRandomPort as Xt, EFFORT_ORDER as Y, UPSTREAM_INACTIVITY_TIMEOUT_MS as Yt, bucketEffort as Z, isBudgetClaudeLead as Zt, appendPlanReminder as _, shimDefaultsToXhigh as _t, buildPeerAwarenessSnippet as a, withInstallLock as an, browserCompoundToolsEnabled as at, resolveWorkerRunOpts as b, getTokenCount as bt, personasFor as c, geminiAvailable as ct, EXPLORE_DEFAULT_MODEL as d, nativeSubagentModel as dt, upstreamAllowH2 as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, reviewerModel as ft, TEST_DEFAULT_MODEL as g, workerToolsEnabled as gt, REVIEW_DEFAULT_MODEL as h, standInToolEnabled as ht, buildAgentPrompt as i, withOneMSuffixForLead as in, browseAgentEnabled as it, searchWeb as j, colbertDegradedWarning as jt, assetFor as k, provisionBrowserAssets as kt, BROWSE_DEFAULT_MODEL as l, generalPurposeFastModel as lt, PLAN_DEFAULT_MODEL as m, scribeModel as mt, MCP_GROUPS as n, classifyMessagesRoute as nn, artifactToolsEnabled as nt, buildPeerAwarenessSummary as o, browserToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, scoutModel as pt, readIteratorWithTimeout as q, DEFAULT_PORT as qt, assertMcpToolSurfaceConsistent as r, withOneMSuffix as rn, brainstormModel as rt, enumerateInjectedMcpToolNames as s, fleetToolsEnabled as st, GROUP_META as t, upstreamMaxConnections as tn, agentToolsEnabled as tt, DEFAULT_MODEL_CHAIN as u, implementerFastModel as ut, resolveDefaultModel as v, countTokens as vt, buildToolbeltAwareness as w, createResponses as wt, runWorkerAgent as x, assembleResponsesPayload as xt, resolveModeDefaults as y, createMessages as yt, formatThinkingRepairDecline as z, shouldUseInsecureTls as zt };
35140
36001
 
35141
- //# sourceMappingURL=peer-mcp-personas-DRL_xT4h.js.map
36002
+ //# sourceMappingURL=peer-mcp-personas-V6stFvpq.js.map