github-router 0.3.292 → 0.3.295

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/{attribution-settings-CpLUCi8R.js → attribution-settings-pkIop1NU.js} +69 -30
  2. package/dist/attribution-settings-pkIop1NU.js.map +1 -0
  3. package/dist/{auth-BwUHopJz.js → auth-CluG5e-l.js} +3 -3
  4. package/dist/{auth-BwUHopJz.js.map → auth-CluG5e-l.js.map} +1 -1
  5. package/dist/browser-ext/manifest.json +1 -1
  6. package/dist/{check-usage-BTda5753.js → check-usage-iOzfnKJe.js} +4 -4
  7. package/dist/{check-usage-BTda5753.js.map → check-usage-iOzfnKJe.js.map} +1 -1
  8. package/dist/{claude-_DYGKCw8.js → claude-DNmxk-Rm.js} +162 -42
  9. package/dist/claude-DNmxk-Rm.js.map +1 -0
  10. package/dist/{codex-rTJ8jW5G.js → codex-Dl4gpHn-.js} +5 -5
  11. package/dist/{codex-rTJ8jW5G.js.map → codex-Dl4gpHn-.js.map} +1 -1
  12. package/dist/{debug-B3UrZTHQ.js → debug-DsEwleGg.js} +2 -2
  13. package/dist/{debug-B3UrZTHQ.js.map → debug-DsEwleGg.js.map} +1 -1
  14. package/dist/engine-XnluJUs6.js +2 -0
  15. package/dist/{gate-discovery-Bjar5dgv.js → gate-discovery-WcduofL4.js} +5 -5
  16. package/dist/{gate-discovery-Bjar5dgv.js.map → gate-discovery-WcduofL4.js.map} +1 -1
  17. package/dist/{get-copilot-usage-CRrf1ZSC.js → get-copilot-usage-B7Z4XVWj.js} +2 -2
  18. package/dist/{get-copilot-usage-CRrf1ZSC.js.map → get-copilot-usage-B7Z4XVWj.js.map} +1 -1
  19. package/dist/hooks.mjs +180 -3
  20. package/dist/hooks.sha256 +1 -1
  21. package/dist/{internal-artifact-open-BsEQRvDi.js → internal-artifact-open-D6tKMg8v.js} +2 -2
  22. package/dist/{internal-artifact-open-BsEQRvDi.js.map → internal-artifact-open-D6tKMg8v.js.map} +1 -1
  23. package/dist/internal-fast-dispatch-guard-CPlCrTlU.js +2 -0
  24. package/dist/internal-fast-dispatch-guard-Xyu17k9X.js +187 -0
  25. package/dist/internal-fast-dispatch-guard-Xyu17k9X.js.map +1 -0
  26. package/dist/{internal-first-mate-guard-CVFFJriA.js → internal-first-mate-guard-C9BV7J4U.js} +3 -3
  27. package/dist/{internal-first-mate-guard-CVFFJriA.js.map → internal-first-mate-guard-C9BV7J4U.js.map} +1 -1
  28. package/dist/{internal-first-mate-guard-DJdX6t48.js → internal-first-mate-guard-WJApyc8j.js} +1 -1
  29. package/dist/{internal-plan-review-8TKDLco6.js → internal-plan-review-t3PuNfuL.js} +3 -3
  30. package/dist/{internal-plan-review-8TKDLco6.js.map → internal-plan-review-t3PuNfuL.js.map} +1 -1
  31. package/dist/{internal-prompt-submit-Da7pxqua.js → internal-prompt-submit-C3KovHQG.js} +4 -4
  32. package/dist/{internal-prompt-submit-Da7pxqua.js.map → internal-prompt-submit-C3KovHQG.js.map} +1 -1
  33. package/dist/{internal-session-bind-BOFytA1f.js → internal-session-bind-d2uoja8I.js} +2 -2
  34. package/dist/{internal-session-bind-BOFytA1f.js.map → internal-session-bind-d2uoja8I.js.map} +1 -1
  35. package/dist/{internal-stop-hook-DrX2xlj0.js → internal-stop-hook-DJFtGCsE.js} +5 -5
  36. package/dist/{internal-stop-hook-DrX2xlj0.js.map → internal-stop-hook-DJFtGCsE.js.map} +1 -1
  37. package/dist/{internal-stop-review-CdouacHL.js → internal-stop-review-YtDUGXpX.js} +2 -2
  38. package/dist/{internal-stop-review-CdouacHL.js.map → internal-stop-review-YtDUGXpX.js.map} +1 -1
  39. package/dist/{internal-worker-guard-Bx-itoP8.js → internal-worker-guard-CdkeS9PN.js} +2 -2
  40. package/dist/{internal-worker-guard-Bx-itoP8.js.map → internal-worker-guard-CdkeS9PN.js.map} +1 -1
  41. package/dist/{internal-workspace-header-8WT0iB5K.js → internal-workspace-header-tuBXJV22.js} +2 -2
  42. package/dist/{internal-workspace-header-8WT0iB5K.js.map → internal-workspace-header-tuBXJV22.js.map} +1 -1
  43. package/dist/{lifecycle-nuOHfwgj.js → lifecycle-Bg6doY3-.js} +2 -2
  44. package/dist/{lifecycle-nuOHfwgj.js.map → lifecycle-Bg6doY3-.js.map} +1 -1
  45. package/dist/lifecycle-CGLt1cJQ.js +2 -0
  46. package/dist/{lifecycle-LeSfa7wH.js → lifecycle-CM9eTzvk.js} +2 -2
  47. package/dist/{lifecycle-LeSfa7wH.js.map → lifecycle-CM9eTzvk.js.map} +1 -1
  48. package/dist/lifecycle-DoUwpVDB.js +2 -0
  49. package/dist/main.js +19 -18
  50. package/dist/main.js.map +1 -1
  51. package/dist/{mcp-workspace-header-q34H_4wL.js → mcp-workspace-header-dERl2YTT.js} +2 -2
  52. package/dist/{mcp-workspace-header-q34H_4wL.js.map → mcp-workspace-header-dERl2YTT.js.map} +1 -1
  53. package/dist/{models-hhJcrZhr.js → models-DZ4hYsR7.js} +3 -3
  54. package/dist/{models-hhJcrZhr.js.map → models-DZ4hYsR7.js.map} +1 -1
  55. package/dist/{orchestration-pzbrKkgD.js → orchestration-BiGCEwbH.js} +2 -2
  56. package/dist/{orchestration-pzbrKkgD.js.map → orchestration-BiGCEwbH.js.map} +1 -1
  57. package/dist/{paths-BH4J7slC.js → paths-Ci485qSJ.js} +4 -4
  58. package/dist/{paths-BH4J7slC.js.map → paths-Ci485qSJ.js.map} +1 -1
  59. package/dist/paths-GD7bgGGy.js +2 -0
  60. package/dist/{peer-mcp-personas-CHbl6MwM.js → peer-mcp-personas-BwNtC8jt.js} +411 -84
  61. package/dist/peer-mcp-personas-BwNtC8jt.js.map +1 -0
  62. package/dist/{plan-review-hook-CfcanA7_.js → plan-review-hook-DuW0pSoM.js} +3 -3
  63. package/dist/{plan-review-hook-CfcanA7_.js.map → plan-review-hook-DuW0pSoM.js.map} +1 -1
  64. package/dist/{prompt-submit-hook-Bqf9ORgb.js → prompt-submit-hook-IMl7ssfr.js} +3 -3
  65. package/dist/{prompt-submit-hook-Bqf9ORgb.js.map → prompt-submit-hook-IMl7ssfr.js.map} +1 -1
  66. package/dist/{provision-BYFd9nPK.js → provision-DeNqzvSM.js} +4 -4
  67. package/dist/{provision-BYFd9nPK.js.map → provision-DeNqzvSM.js.map} +1 -1
  68. package/dist/{self-invocation-DhO1Z8iD.js → self-invocation-DAB_od0C.js} +2 -2
  69. package/dist/{self-invocation-DhO1Z8iD.js.map → self-invocation-DAB_od0C.js.map} +1 -1
  70. package/dist/{serve-BWMxDLnD.js → serve-BS5EmCpM.js} +12 -12
  71. package/dist/{serve-BWMxDLnD.js.map → serve-BS5EmCpM.js.map} +1 -1
  72. package/dist/{server-setup-CqlaZukJ.js → server-setup-DsGJnJ_O.js} +398 -283
  73. package/dist/server-setup-DsGJnJ_O.js.map +1 -0
  74. package/dist/{start-5MgGT4IF.js → start-vJdt2MiM.js} +3 -3
  75. package/dist/{start-5MgGT4IF.js.map → start-vJdt2MiM.js.map} +1 -1
  76. package/dist/{stop-gate-hook-BiBp5aGm.js → stop-gate-hook-DQnh_KfI.js} +3 -3
  77. package/dist/{stop-gate-hook-BiBp5aGm.js.map → stop-gate-hook-DQnh_KfI.js.map} +1 -1
  78. package/dist/{stop-gate-policy-BGd6b5hR.js → stop-gate-policy-DMr3KVmw.js} +2 -2
  79. package/dist/{stop-gate-policy-BGd6b5hR.js.map → stop-gate-policy-DMr3KVmw.js.map} +1 -1
  80. package/dist/{token-8drORhXg.js → token-BtJhjXXu.js} +54 -7
  81. package/dist/token-BtJhjXXu.js.map +1 -0
  82. package/dist/{worker-dispatch-D5fGroNr.js → worker-dispatch-BEHkTbnH.js} +2 -2
  83. package/dist/{worker-dispatch-D5fGroNr.js.map → worker-dispatch-BEHkTbnH.js.map} +1 -1
  84. package/package.json +1 -1
  85. package/dist/attribution-settings-CpLUCi8R.js.map +0 -1
  86. package/dist/claude-_DYGKCw8.js.map +0 -1
  87. package/dist/engine-C9axTIu7.js +0 -2
  88. package/dist/lifecycle-C8fOsQke.js +0 -2
  89. package/dist/lifecycle-D4Yc1aap.js +0 -2
  90. package/dist/paths-DJZoXfAS.js +0 -2
  91. package/dist/peer-mcp-personas-CHbl6MwM.js.map +0 -1
  92. package/dist/server-setup-CqlaZukJ.js.map +0 -1
  93. package/dist/token-8drORhXg.js.map +0 -1
@@ -1,8 +1,8 @@
1
- import { $ as bucketEffort, At as countTokens, B as resolveAdvisorModel, Bt as createChatCompletions, Cn as withOneMSuffixForLead, E as toolbeltEnabled, F as buildAdvisorStream, G as buildAnthropicErrorEvent, H as rememberThinkingHistoryRepair, Ht as readResponseBodyCapped, I as injectAdvisorTool, It as assembleResponsesPayload, J as logStreamError, K as buildOpenAIErrorEvent, L as isAdvisorRequested, Lt as warnOnTokenPriceDrift, M as searchWeb, Mt as getTokenCount, N as ADVISOR_INTERNAL_TOOL_NAME, Nt as findLaunchBySecret, P as ADVISOR_TOOL_INSTRUCTIONS, Q as UNKNOWN_EFFORT_ANCHOR, Qt as provisionTreeSitterAssets, R as isFastProfileLead, Rt as resolveMcpToolTimeoutMs, Sn as withOneMSuffix, U as repairKnownThinkingHistory, Ut as parseJsonOrDiagnose, V as formatThinkingRepairDecline, Vt as MAX_RESPONSE_BODY_BYTES, W as repairRejectedThinkingHistory, Wt as normalizeOpenAIUsage, X as relayAnthropicStream, Y as readIteratorWithTimeout, Z as EFFORT_ORDER, _n as upstreamMaxConnections, an as BUDGET_SMALL_FAST_SLUG, bn as catalogAdvertises1M, dn as UPSTREAM_INACTIVITY_TIMEOUT_MS, et as clampEffort, fn as generateRandomPort, gn as upstreamAllowH2, i as assertMcpToolSurfaceConsistent, jt as createMessages, kt as shimDefaultsToXhigh, nt as handleMcpPost, ot as agentToolsEnabled, pn as isBudgetClaudeLead, q as isControllerClosedError, rn as toolbeltPathOverride, tt as handleMcpDelete, un as UPSTREAM_FETCH_TIMEOUT_MS, vn as classifyMessagesRoute, wn as withInstallLock, xn as oneMContextDisabled, yn as pickEndpoint, z as resolveAdvisorEffort, zt as createResponses } from "./peer-mcp-personas-CHbl6MwM.js";
1
+ import { $ as bucketEffort, $t as assembleResponsesPayload, An as isBudgetClaudeLead, B as resolveAdvisorModel, Bn as stripTrailingOneMSuffix, Dn as UPSTREAM_FETCH_TIMEOUT_MS, E as toolbeltEnabled, F as FAST_ADVISOR_TOOL_INSTRUCTIONS, Fn as classifyMessagesRoute, G as buildAnthropicErrorEvent, Gt as countTokens, H as rememberThinkingHistoryRepair, Ht as resolveModelAlias, I as buildAdvisorStream, In as catalogAdvertises1M, J as logStreamError, Jt as getTokenCount, K as buildOpenAIErrorEvent, Kt as createMessages, L as injectAdvisorTool, Ln as oneMContextDisabled, Lt as LUNA_SONNET_ALIAS_ID, M as searchWeb, Mt as LUNA_DRIVER_ALIAS_ID, N as ADVISOR_INTERNAL_TOOL_NAME, Nn as upstreamAllowH2, Nt as LUNA_HAIKU_ALIAS_ID, On as UPSTREAM_INACTIVITY_TIMEOUT_MS, P as ADVISOR_TOOL_INSTRUCTIONS, Pn as upstreamMaxConnections, Q as UNKNOWN_EFFORT_ANCHOR, R as isAdvisorRequested, Rn as withOneMSuffix, Rt as canonicalizeAliasModel, Sn as BUDGET_SMALL_FAST_SLUG, U as repairKnownThinkingHistory, V as formatThinkingRepairDecline, Vn as withInstallLock, W as repairRejectedThinkingHistory, Wt as shimDefaultsToXhigh, X as relayAnthropicStream, Xt as findLaunchBySecret, Y as readIteratorWithTimeout, Yt as getTokenizerFromModel, Z as EFFORT_ORDER, an as readResponseBodyCapped, bn as toolbeltPathOverride, en as warnOnTokenPriceDrift, et as clampEffort, hn as provisionTreeSitterAssets, i as assertMcpToolSurfaceConsistent, in as MAX_RESPONSE_BODY_BYTES, kn as generateRandomPort, nn as createResponses, nt as handleMcpPost, on as parseJsonOrDiagnose, q as isControllerClosedError, qt as getTextTokenCount, rn as createChatCompletions, sn as normalizeOpenAIUsage, st as agentToolsEnabled, tn as resolveMcpToolTimeoutMs, tt as handleMcpDelete, z as resolveAdvisorEffort, zn as withOneMSuffixForLead } from "./peer-mcp-personas-BwNtC8jt.js";
2
2
  import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
3
- import { i as ensurePaths, t as PATHS } from "./paths-BH4J7slC.js";
4
- import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-8drORhXg.js";
5
- import { t as getCopilotUsage } from "./get-copilot-usage-CRrf1ZSC.js";
3
+ import { i as ensurePaths, t as PATHS } from "./paths-Ci485qSJ.js";
4
+ import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-BtJhjXXu.js";
5
+ import { t as getCopilotUsage } from "./get-copilot-usage-B7Z4XVWj.js";
6
6
  import { a as resolveExecutable, n as killManagedTree, o as runCommandCapture, r as parseBoolEnv, s as runCommandVoid } from "./exec-y8C_MU8A.js";
7
7
  import consola from "consola";
8
8
  import * as fs$2 from "node:fs";
@@ -354,232 +354,6 @@ async function runSelfUpdate(opts) {
354
354
  }
355
355
  }
356
356
  //#endregion
357
- //#region src/lib/launch-profile.ts
358
- const STANDARD_PROFILE = Object.freeze({
359
- id: "standard",
360
- hasCoordinator: true
361
- });
362
- /**
363
- * The `-m fast` roster: exactly four native agents (`scout`, `implementer`,
364
- * `reviewer`, `planner`), the fast-only `oracle` peer tool, no coordinator,
365
- * and only `peers`/`search` plus the ordinary opt-in `browser` group.
366
- * `workers`/`orchestrate`/`decide`/`fleet`/`first-mate` are hard denies even
367
- * when their independent standard-profile gates pass.
368
- */
369
- const FAST_PROFILE = Object.freeze({
370
- id: "fast",
371
- nativeRoster: /* @__PURE__ */ new Set([
372
- "scout",
373
- "implementer",
374
- "reviewer",
375
- "planner"
376
- ]),
377
- personaAllowlist: /* @__PURE__ */ new Set(["oracle"]),
378
- allowedGroups: /* @__PURE__ */ new Set([
379
- "peers",
380
- "search",
381
- "browser"
382
- ]),
383
- hasCoordinator: false
384
- });
385
- function profileDescriptor(id) {
386
- return id === "fast" ? FAST_PROFILE : STANDARD_PROFILE;
387
- }
388
- /**
389
- * Resolve the parsed `-m` argument to a launch profile.
390
- *
391
- * Deliberately keyed on the RAW alias string (trimmed, case-insensitive
392
- * `"fast"`), never on a resolved model id: `resolveLeadSlugArg` maps `fast`
393
- * to `FAST_LEAD_MODEL` (`./port`) before this is of any use to a caller who
394
- * only has the resolved id, so callers that already resolved the lead must
395
- * pass the ORIGINAL `-m` value here, not the resolved one. This is what
396
- * keeps `-m gpt-5.6-luna` (a direct pin of the same underlying model) a
397
- * standard-surface launch — only the literal alias narrows the surface.
398
- */
399
- function resolveLaunchProfile(modelArg) {
400
- return modelArg?.trim().toLowerCase() === "fast" ? "fast" : "standard";
401
- }
402
- /**
403
- * Router-owned alias id for the fast profile's Sonnet-tier row
404
- * (`ANTHROPIC_DEFAULT_SONNET_MODEL`). Never sent upstream — canonicalized to
405
- * `LUNA_REAL_MODEL_ID` by `canonicalizeAliasModel` before the request
406
- * reaches Copilot.
407
- */
408
- const LUNA_DRIVER_ALIAS_ID = "gh-router-luna-driver-max";
409
- /** Fast native-agent alias ids preserve role-specific effort provenance until
410
- * the authenticated request boundary. They both canonicalize to Luna, but the
411
- * scout is fixed high while the implementer is fixed max. */
412
- const LUNA_SCOUT_ALIAS_ID = "gh-router-luna-scout-high";
413
- const LUNA_IMPLEMENTER_ALIAS_ID = "gh-router-luna-implementer-max";
414
- const LUNA_SONNET_ALIAS_ID = "gh-router-luna-sonnet-xhigh";
415
- /**
416
- * Router-owned alias id for the fast profile's Haiku-tier row
417
- * (`ANTHROPIC_DEFAULT_HAIKU_MODEL` / `ANTHROPIC_SMALL_FAST_MODEL`).
418
- */
419
- const LUNA_HAIKU_ALIAS_ID = "gh-router-luna-haiku-high";
420
- /** The real Copilot catalog id every Luna alias (including the driver
421
- * itself) canonicalizes to. */
422
- const LUNA_REAL_MODEL_ID = "gpt-5.6-luna";
423
- /**
424
- * The full alias table, keyed by `aliasId`. A simpler model-id-only table is
425
- * rejected by design: the driver, the Sonnet tier, and the Haiku tier all
426
- * resolve to the SAME Luna catalog id, so after early canonicalization a
427
- * table keyed on the real id could no longer tell which absent-effort
428
- * default applies. Alias provenance — which of the three ids the request
429
- * actually carried — is the minimum discriminator that survives from tier
430
- * selection through to request preprocessing, which is why canonicalization
431
- * must happen LAST (in the `/v1/messages` identity preflight), after the
432
- * effort default has already been read off the alias.
433
- */
434
- const MODEL_ALIAS_TABLE = /* @__PURE__ */ new Map([
435
- [LUNA_DRIVER_ALIAS_ID, {
436
- aliasId: LUNA_DRIVER_ALIAS_ID,
437
- realModel: LUNA_REAL_MODEL_ID,
438
- absentEffortDefault: "max"
439
- }],
440
- [LUNA_SCOUT_ALIAS_ID, {
441
- aliasId: LUNA_SCOUT_ALIAS_ID,
442
- realModel: LUNA_REAL_MODEL_ID,
443
- absentEffortDefault: "high"
444
- }],
445
- [LUNA_IMPLEMENTER_ALIAS_ID, {
446
- aliasId: LUNA_IMPLEMENTER_ALIAS_ID,
447
- realModel: LUNA_REAL_MODEL_ID,
448
- absentEffortDefault: "max"
449
- }],
450
- [LUNA_SONNET_ALIAS_ID, {
451
- aliasId: LUNA_SONNET_ALIAS_ID,
452
- realModel: LUNA_REAL_MODEL_ID,
453
- absentEffortDefault: "xhigh"
454
- }],
455
- [LUNA_HAIKU_ALIAS_ID, {
456
- aliasId: LUNA_HAIKU_ALIAS_ID,
457
- realModel: LUNA_REAL_MODEL_ID,
458
- absentEffortDefault: "high"
459
- }]
460
- ]);
461
- /**
462
- * Look up the alias descriptor for a wire-facing model id (with or without
463
- * a trailing `[1m]` bracket — the bracket is stripped before the table
464
- * lookup and is orthogonal to alias identity). Returns undefined for any
465
- * id that isn't one of the three registered aliases (including the bare
466
- * `claude-*` ids and every other real Copilot catalog id).
467
- */
468
- function resolveModelAlias(id) {
469
- const bare = id.replace(/\[1m\]$/i, "");
470
- return MODEL_ALIAS_TABLE.get(bare);
471
- }
472
- /**
473
- * Strip alias provenance and return the real catalog id to send upstream.
474
- * Idempotent passthrough for any id that isn't a registered alias (a bare
475
- * `claude-*` slug, an already-real Copilot id, or anything else) — this is
476
- * safe to call unconditionally on every `body.model` at the outbound
477
- * boundary. Preserves a trailing `[1m]` bracket: canonicalization only
478
- * erases ALIAS identity, not the 1M-context accounting decoration.
479
- */
480
- function canonicalizeAliasModel(id) {
481
- const bracket = /\[1m\]$/i.test(id) ? "[1m]" : "";
482
- const bare = bracket ? id.slice(0, -bracket.length) : id;
483
- const alias = MODEL_ALIAS_TABLE.get(bare);
484
- return alias ? `${alias.realModel}${bracket}` : id;
485
- }
486
- const FAST_REQUIRED_CONTEXT_TOKENS = 1e6;
487
- function findModel(catalog, id) {
488
- return catalog?.data?.find((m) => m.id === id);
489
- }
490
- function hasToolCalls(model) {
491
- return model?.capabilities?.supports?.tool_calls === true;
492
- }
493
- function hasContextAtLeast(model, tokens) {
494
- return (model?.capabilities?.limits?.max_context_window_tokens ?? 0) >= tokens;
495
- }
496
- function supportsEffort(model, effort) {
497
- const list = model?.capabilities?.supports?.reasoning_effort;
498
- return Array.isArray(list) && list.includes(effort);
499
- }
500
- function supportsEndpoint(model, paths) {
501
- const endpoints = model?.supported_endpoints;
502
- return Array.isArray(endpoints) && endpoints.some((endpoint) => paths.has(endpoint));
503
- }
504
- const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
505
- const MESSAGES_ENDPOINTS = /* @__PURE__ */ new Set(["/messages", "/v1/messages"]);
506
- function hasUsablePromptMetadata(model) {
507
- const prompt = model?.capabilities?.limits?.max_prompt_tokens;
508
- return typeof prompt === "number" && Number.isFinite(prompt) && prompt > 0;
509
- }
510
- /**
511
- * Validate the live Copilot catalog carries every model the fast profile's
512
- * EXACT roster depends on, with the specific capabilities each assignment
513
- * needs. These are capability-availability PREREQUISITES for constructing
514
- * the roster — not an allowlist of models the user may select later in the
515
- * session — so a partial catalog fails the whole `-m fast` launch rather
516
- * than silently substituting or dropping an agent.
517
- *
518
- * Checks, per the fast-launch-profile design:
519
- * - Luna lead/scout/implementer: tool calls, >=1M, high+max, Responses.
520
- * - Sol planner: tool calls, >=1M, high, Responses.
521
- * - Grok reviewer: tool calls, medium, Responses, usable prompt metadata.
522
- * - Gemini Advisor: >=1M, high, chat-completions.
523
- * - Opus Oracle: exact Opus 5, >=1M, adaptive/high, Messages, prompt metadata.
524
- *
525
- * Pure over the passed-in catalog snapshot so it's unit-testable without
526
- * `state` — callers pass `state.models` at call time.
527
- */
528
- function validateFastProfilePrerequisites(catalog) {
529
- const missing = [];
530
- const luna = findModel(catalog, LUNA_REAL_MODEL_ID);
531
- if (!luna) missing.push(`${LUNA_REAL_MODEL_ID}: absent from the live catalog`);
532
- else {
533
- if (!hasToolCalls(luna)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise tool_calls`);
534
- if (!hasContextAtLeast(luna, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push(`${LUNA_REAL_MODEL_ID}: advertised context window is below 1M`);
535
- if (!supportsEffort(luna, "high") || !supportsEffort(luna, "max")) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise both "high" and "max" reasoning effort`);
536
- if (!supportsEndpoint(luna, RESPONSES_ENDPOINTS)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise a supported Responses endpoint`);
537
- }
538
- const sol = findModel(catalog, "gpt-5.6-sol");
539
- if (!sol) missing.push("gpt-5.6-sol: absent from the live catalog");
540
- else {
541
- if (!hasToolCalls(sol)) missing.push("gpt-5.6-sol: does not advertise tool_calls");
542
- if (!hasContextAtLeast(sol, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gpt-5.6-sol: advertised context window is below 1M");
543
- if (!supportsEffort(sol, "high")) missing.push("gpt-5.6-sol: does not advertise a \"high\" reasoning effort");
544
- if (!supportsEndpoint(sol, RESPONSES_ENDPOINTS)) missing.push("gpt-5.6-sol: does not advertise a supported Responses endpoint");
545
- }
546
- const grok = findModel(catalog, "grok-4.6");
547
- if (!grok) missing.push("grok-4.6: absent from the live catalog");
548
- else {
549
- if (!hasToolCalls(grok)) missing.push("grok-4.6: does not advertise tool_calls");
550
- if (!supportsEffort(grok, "medium")) missing.push("grok-4.6: does not advertise a \"medium\" reasoning effort");
551
- if (!hasUsablePromptMetadata(grok)) missing.push("grok-4.6: no usable max_prompt_tokens metadata");
552
- if (!supportsEndpoint(grok, RESPONSES_ENDPOINTS)) missing.push("grok-4.6: does not advertise a supported Responses endpoint");
553
- }
554
- const gemini = findModel(catalog, "gemini-3.7-flash");
555
- if (!gemini) missing.push("gemini-3.7-flash: absent from the live catalog");
556
- else {
557
- if (!hasContextAtLeast(gemini, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gemini-3.7-flash: advertised context window is below 1M");
558
- if (!supportsEffort(gemini, "high")) missing.push("gemini-3.7-flash: does not advertise a \"high\" reasoning effort");
559
- if (pickEndpoint(gemini) !== "chat") missing.push("gemini-3.7-flash: does not advertise a supported chat-completions endpoint");
560
- }
561
- const opus = findModel(catalog, "claude-opus-5");
562
- if (!opus) missing.push("claude-opus-5: absent from the live catalog");
563
- else {
564
- if (!hasContextAtLeast(opus, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("claude-opus-5: advertised context window is below 1M");
565
- if (!supportsEffort(opus, "high")) missing.push("claude-opus-5: does not advertise a \"high\" reasoning effort");
566
- if (opus.capabilities?.supports?.adaptive_thinking !== true) missing.push("claude-opus-5: does not advertise adaptive_thinking");
567
- if (!hasUsablePromptMetadata(opus)) missing.push("claude-opus-5: no usable max_prompt_tokens metadata");
568
- if (!supportsEndpoint(opus, MESSAGES_ENDPOINTS)) missing.push("claude-opus-5: does not advertise a supported Messages endpoint");
569
- }
570
- return {
571
- ok: missing.length === 0,
572
- missing
573
- };
574
- }
575
- /**
576
- * Format `validateFastProfilePrerequisites`'s failure list into the launch
577
- * error message: every missing/invalid model, plus the rollback command.
578
- */
579
- function formatFastPrerequisiteFailure(missing) {
580
- return "github-router claude -m fast requires the following live-catalog capabilities, which this account's catalog does not fully provide:\n" + missing.map((m) => ` - ${m}`).join("\n") + "\n\nFalling back or silently dropping an agent is not supported for the fast profile's exact roster. Run plain `github-router claude` instead.";
581
- }
582
- //#endregion
583
357
  //#region src/lib/file-log-reporter.ts
584
358
  const MAX_LOG_BYTES = 1048576;
585
359
  const DEDUP_MAX = 1e3;
@@ -1306,6 +1080,73 @@ function wireDaemonTeardown(handle, opts = {}) {
1306
1080
  });
1307
1081
  }
1308
1082
  //#endregion
1083
+ //#region src/lib/grok-context.ts
1084
+ /** Claude Code's fixed output-token reserve when computing its own
1085
+ * compaction-window assumption. Grok 4.6's 128K max output stays well
1086
+ * above this, so the reserve — not the model's real output ceiling — is
1087
+ * what caps the second term. */
1088
+ const CLIENT_OUTPUT_RESERVE_TOKENS = 2e4;
1089
+ /**
1090
+ * The client's additional flat reserve between the output-reduced window and
1091
+ * its reactive compaction threshold. Read from the installed Claude Code
1092
+ * 2.1.247 bundle: `xWe(e,t)` computes `n = e - 13000` and returns it when no
1093
+ * percentage override is set, where `e` is already
1094
+ * `window - min(maxOutput, 20_000)`.
1095
+ */
1096
+ const CLIENT_THRESHOLD_RESERVE_TOKENS = 13e3;
1097
+ /**
1098
+ * Bounds the `CLAUDE_CODE_AUTO_COMPACT_WINDOW` env path applies to whatever it
1099
+ * parses: `IWe = 1e5` floor, `ZRt = 1e6` cap. A value below the floor is
1100
+ * silently RAISED to it, so deriving something smaller would be a lie; a value
1101
+ * above the cap is clamped. Both read from the same 2.1.247 bundle.
1102
+ */
1103
+ const ENV_WINDOW_MIN_TOKENS = 1e5;
1104
+ const ENV_WINDOW_MAX_TOKENS = 1e6;
1105
+ /**
1106
+ * Derive the conservative 85%-of-prompt compaction trigger and the client
1107
+ * window that would need to be assumed to make that trigger point natural.
1108
+ *
1109
+ * `target trigger = floor(max_prompt_tokens * 0.85)`
1110
+ * `assumed client window = target trigger + min(max_output_tokens, 20_000)`
1111
+ *
1112
+ * For Grok 4.6's current catalog entry (max_prompt_tokens=372_000,
1113
+ * max_output_tokens=128_000) this yields `{ triggerTokens: 316_200,
1114
+ * assumedClientWindowTokens: 336_200 }` — the exact figures the plan
1115
+ * document records. Pure and side-effect-free; callers are responsible for
1116
+ * sourcing `maxPromptTokens`/`maxOutputTokens` from the live catalog.
1117
+ */
1118
+ function computeConservativeCompactionTrigger(maxPromptTokens, maxOutputTokens) {
1119
+ const triggerTokens = Math.floor(maxPromptTokens * .85);
1120
+ return {
1121
+ triggerTokens,
1122
+ assumedClientWindowTokens: triggerTokens + Math.min(maxOutputTokens, CLIENT_OUTPUT_RESERVE_TOKENS)
1123
+ };
1124
+ }
1125
+ /**
1126
+ * The integer `CLAUDE_CODE_AUTO_COMPACT_WINDOW` value that puts the client's
1127
+ * REACTIVE compaction trigger at 85% of the provider's real prompt ceiling.
1128
+ *
1129
+ * The client computes its trigger as
1130
+ * `window - min(maxOutput, 20_000) - 13_000`, so we invert that:
1131
+ * `window = assumedClientWindowTokens + 13_000`.
1132
+ *
1133
+ * Worked, against the live catalog at time of writing:
1134
+ * Luna 922_000 prompt / 128_000 out -> trigger 783_700, window 816_700
1135
+ * Opus 5 936_000 prompt / 64_000 out -> trigger 795_600, window 828_600
1136
+ *
1137
+ * Returns undefined when the catalog metadata is missing or nonsensical, so
1138
+ * the caller omits the variable entirely rather than exporting a guess. That
1139
+ * degrades to today's behaviour (client budgets against the total window)
1140
+ * instead of to a wrong number, which is the safer direction for a value the
1141
+ * client silently floors.
1142
+ */
1143
+ function deriveAutoCompactWindowTokens(maxPromptTokens, maxOutputTokens) {
1144
+ if (typeof maxPromptTokens !== "number" || !Number.isFinite(maxPromptTokens) || maxPromptTokens <= 0) return;
1145
+ const { assumedClientWindowTokens } = computeConservativeCompactionTrigger(maxPromptTokens, typeof maxOutputTokens === "number" && Number.isFinite(maxOutputTokens) && maxOutputTokens > 0 ? maxOutputTokens : CLIENT_OUTPUT_RESERVE_TOKENS);
1146
+ const window = assumedClientWindowTokens + CLIENT_THRESHOLD_RESERVE_TOKENS;
1147
+ return Math.min(ENV_WINDOW_MAX_TOKENS, Math.max(ENV_WINDOW_MIN_TOKENS, window));
1148
+ }
1149
+ //#endregion
1309
1150
  //#region src/lib/proxy.ts
1310
1151
  let socketCounter = 0;
1311
1152
  /**
@@ -1656,7 +1497,7 @@ function collectToolFieldKeys(body) {
1656
1497
  //#endregion
1657
1498
  //#region package.json
1658
1499
  var name = "github-router";
1659
- var version = "0.3.292";
1500
+ var version = "0.3.295";
1660
1501
  //#endregion
1661
1502
  //#region src/lib/approval.ts
1662
1503
  const awaitApproval = async () => {
@@ -1758,7 +1599,7 @@ const NO_LOOP = {
1758
1599
  tier: null,
1759
1600
  repeats: 0
1760
1601
  };
1761
- function isRecord(value) {
1602
+ function isRecord$1(value) {
1762
1603
  return typeof value === "object" && value !== null;
1763
1604
  }
1764
1605
  function digest(value) {
@@ -1788,7 +1629,7 @@ function normalizeResultBody(content) {
1788
1629
  }
1789
1630
  function normalizeResultBlock(block) {
1790
1631
  if (typeof block === "string") return `s:${boundedText(block)}`;
1791
- if (!isRecord(block)) return `j:${boundedText(safeJson(block))}`;
1632
+ if (!isRecord$1(block)) return `j:${boundedText(safeJson(block))}`;
1792
1633
  const type = typeof block.type === "string" ? block.type : "unknown";
1793
1634
  if (type === "text" && typeof block.text === "string") return `t:${boundedText(block.text)}`;
1794
1635
  return `${type}:${digest(safeJson(block))}`;
@@ -1899,15 +1740,15 @@ function mayContainToolTraffic(rawBody) {
1899
1740
  return rawBody.includes("tool_result") || rawBody.includes("tool_call_id") || rawBody.includes("function_call_output");
1900
1741
  }
1901
1742
  function blockType(block) {
1902
- return isRecord(block) && typeof block.type === "string" ? block.type : void 0;
1743
+ return isRecord$1(block) && typeof block.type === "string" ? block.type : void 0;
1903
1744
  }
1904
1745
  /** Anthropic Messages: assistant `tool_use` blocks ↔ user `tool_result` blocks. */
1905
1746
  function extractAnthropicTurns(body) {
1906
- if (!isRecord(body) || !Array.isArray(body.messages)) return [];
1747
+ if (!isRecord$1(body) || !Array.isArray(body.messages)) return [];
1907
1748
  const turns = [];
1908
1749
  for (let i = 0; i < body.messages.length; i++) {
1909
1750
  const message = body.messages[i];
1910
- if (!isRecord(message) || message.role !== "assistant") continue;
1751
+ if (!isRecord$1(message) || message.role !== "assistant") continue;
1911
1752
  if (!Array.isArray(message.content)) continue;
1912
1753
  const uses = message.content.filter((b) => blockType(b) === "tool_use");
1913
1754
  if (uses.length === 0) {
@@ -1920,17 +1761,17 @@ function extractAnthropicTurns(body) {
1920
1761
  const results = /* @__PURE__ */ new Map();
1921
1762
  for (let j = i + 1; j < body.messages.length; j++) {
1922
1763
  const next = body.messages[j];
1923
- if (!isRecord(next) || next.role !== "user") break;
1764
+ if (!isRecord$1(next) || next.role !== "user") break;
1924
1765
  if (!Array.isArray(next.content)) break;
1925
1766
  for (const block of next.content) {
1926
- if (!isRecord(block) || blockType(block) !== "tool_result") continue;
1767
+ if (!isRecord$1(block) || blockType(block) !== "tool_result") continue;
1927
1768
  const id = block.tool_use_id;
1928
1769
  if (typeof id === "string") results.set(id, block);
1929
1770
  }
1930
1771
  }
1931
1772
  const calls = [];
1932
1773
  for (const use of uses) {
1933
- if (!isRecord(use)) continue;
1774
+ if (!isRecord$1(use)) continue;
1934
1775
  const name = typeof use.name === "string" ? use.name : "unknown";
1935
1776
  const id = typeof use.id === "string" ? use.id : void 0;
1936
1777
  const result = id === void 0 ? void 0 : results.get(id);
@@ -1949,16 +1790,16 @@ function anthropicHasNarration(content) {
1949
1790
  return content.some((block) => {
1950
1791
  const type = blockType(block);
1951
1792
  if (type === "thinking" || type === "redacted_thinking") return true;
1952
- return type === "text" && isRecord(block) && typeof block.text === "string" && block.text.trim() !== "";
1793
+ return type === "text" && isRecord$1(block) && typeof block.text === "string" && block.text.trim() !== "";
1953
1794
  });
1954
1795
  }
1955
1796
  /** OpenAI Chat Completions: assistant `tool_calls[]` ↔ `role:"tool"` messages. */
1956
1797
  function extractChatTurns(body) {
1957
- if (!isRecord(body) || !Array.isArray(body.messages)) return [];
1798
+ if (!isRecord$1(body) || !Array.isArray(body.messages)) return [];
1958
1799
  const turns = [];
1959
1800
  for (let i = 0; i < body.messages.length; i++) {
1960
1801
  const message = body.messages[i];
1961
- if (!isRecord(message) || message.role !== "assistant") continue;
1802
+ if (!isRecord$1(message) || message.role !== "assistant") continue;
1962
1803
  if (!Array.isArray(message.tool_calls) || message.tool_calls.length === 0) {
1963
1804
  turns.push({
1964
1805
  calls: [],
@@ -1969,14 +1810,14 @@ function extractChatTurns(body) {
1969
1810
  const results = /* @__PURE__ */ new Map();
1970
1811
  for (let j = i + 1; j < body.messages.length; j++) {
1971
1812
  const next = body.messages[j];
1972
- if (!isRecord(next) || next.role !== "tool") break;
1813
+ if (!isRecord$1(next) || next.role !== "tool") break;
1973
1814
  const id = next.tool_call_id;
1974
1815
  if (typeof id === "string") results.set(id, next);
1975
1816
  }
1976
1817
  const calls = [];
1977
1818
  for (const call of message.tool_calls) {
1978
- if (!isRecord(call)) continue;
1979
- const fn = isRecord(call.function) ? call.function : void 0;
1819
+ if (!isRecord$1(call)) continue;
1820
+ const fn = isRecord$1(call.function) ? call.function : void 0;
1980
1821
  const name = typeof fn?.name === "string" ? fn.name : "unknown";
1981
1822
  const id = typeof call.id === "string" ? call.id : void 0;
1982
1823
  const result = id === void 0 ? void 0 : results.get(id);
@@ -1999,15 +1840,15 @@ function extractChatTurns(body) {
1999
1840
  function chatHasNarration(content) {
2000
1841
  if (typeof content === "string") return content.trim() !== "";
2001
1842
  if (!Array.isArray(content)) return false;
2002
- return content.some((part) => isRecord(part) && typeof part.text === "string" && part.text.trim() !== "");
1843
+ return content.some((part) => isRecord$1(part) && typeof part.text === "string" && part.text.trim() !== "");
2003
1844
  }
2004
1845
  /** OpenAI Responses: `function_call` items ↔ `function_call_output` items. */
2005
1846
  function extractResponsesTurns(body) {
2006
- if (!isRecord(body) || !Array.isArray(body.input)) return [];
1847
+ if (!isRecord$1(body) || !Array.isArray(body.input)) return [];
2007
1848
  const items = body.input;
2008
1849
  const results = /* @__PURE__ */ new Map();
2009
1850
  for (const item of items) {
2010
- if (!isRecord(item) || item.type !== "function_call_output") continue;
1851
+ if (!isRecord$1(item) || item.type !== "function_call_output") continue;
2011
1852
  const id = item.call_id;
2012
1853
  if (typeof id === "string") results.set(id, item);
2013
1854
  }
@@ -2015,14 +1856,14 @@ function extractResponsesTurns(body) {
2015
1856
  let i = 0;
2016
1857
  while (i < items.length) {
2017
1858
  const item = items[i];
2018
- if (!isRecord(item) || item.type !== "function_call") {
1859
+ if (!isRecord$1(item) || item.type !== "function_call") {
2019
1860
  i++;
2020
1861
  continue;
2021
1862
  }
2022
1863
  const batch = [];
2023
1864
  while (i < items.length) {
2024
1865
  const candidate = items[i];
2025
- if (!isRecord(candidate) || candidate.type !== "function_call") break;
1866
+ if (!isRecord$1(candidate) || candidate.type !== "function_call") break;
2026
1867
  batch.push(candidate);
2027
1868
  i++;
2028
1869
  }
@@ -2050,7 +1891,7 @@ function extractResponsesTurns(body) {
2050
1891
  function responsesHasNarration(items, batchStart) {
2051
1892
  for (let i = batchStart - 1; i >= 0; i--) {
2052
1893
  const item = items[i];
2053
- if (!isRecord(item)) return false;
1894
+ if (!isRecord$1(item)) return false;
2054
1895
  if (item.type === "function_call_output") continue;
2055
1896
  if (item.type === "reasoning") return true;
2056
1897
  if (item.type === "message" || item.role === "assistant") return responsesItemHasText(item);
@@ -2061,13 +1902,13 @@ function responsesHasNarration(items, batchStart) {
2061
1902
  function responsesItemHasText(item) {
2062
1903
  if (typeof item.content === "string") return item.content.trim() !== "";
2063
1904
  if (!Array.isArray(item.content)) return false;
2064
- return item.content.some((block) => isRecord(block) && typeof block.text === "string" && block.text.trim() !== "");
1905
+ return item.content.some((block) => isRecord$1(block) && typeof block.text === "string" && block.text.trim() !== "");
2065
1906
  }
2066
1907
  function injectAnthropicNudge(body, text) {
2067
- if (!isRecord(body) || !Array.isArray(body.messages)) return false;
1908
+ if (!isRecord$1(body) || !Array.isArray(body.messages)) return false;
2068
1909
  const messages = body.messages;
2069
1910
  const last = messages[messages.length - 1];
2070
- if (!isRecord(last) || last.role !== "user") return false;
1911
+ if (!isRecord$1(last) || last.role !== "user") return false;
2071
1912
  if (!Array.isArray(last.content)) return false;
2072
1913
  messages[messages.length - 1] = {
2073
1914
  ...last,
@@ -2079,7 +1920,7 @@ function injectAnthropicNudge(body, text) {
2079
1920
  return true;
2080
1921
  }
2081
1922
  function injectChatNudge(body, text) {
2082
- if (!isRecord(body) || !Array.isArray(body.messages)) return false;
1923
+ if (!isRecord$1(body) || !Array.isArray(body.messages)) return false;
2083
1924
  body.messages = [...body.messages, {
2084
1925
  role: "user",
2085
1926
  content: text
@@ -2087,7 +1928,7 @@ function injectChatNudge(body, text) {
2087
1928
  return true;
2088
1929
  }
2089
1930
  function injectResponsesNudge(body, text) {
2090
- if (!isRecord(body) || !Array.isArray(body.input)) return false;
1931
+ if (!isRecord$1(body) || !Array.isArray(body.input)) return false;
2091
1932
  body.input = [...body.input, {
2092
1933
  role: "user",
2093
1934
  content: [{
@@ -4300,10 +4141,10 @@ function isAsyncIterable(x) {
4300
4141
  * from the catalog (which would also mean re-parsing/re-picking work the
4301
4142
  * caller already did).
4302
4143
  * - `makeShimContinueTurn` (below), which `buildAdvisorStream`
4303
- * (`src/services/advisor/advisor.ts`) injects as its `continueTurn` for
4304
- * the fast Luna-lead profile, so an advisor continuation on a non-Claude
4305
- * lead runs through the SAME translation + SSE-synthesis machinery as
4306
- * the initial turn instead of a parallel, divergent implementation.
4144
+ * (`src/services/advisor/advisor.ts`) injects for any non-Claude model
4145
+ * selected by an authenticated fast primary lead, so its Advisor
4146
+ * continuation runs through the SAME translation + SSE-synthesis machinery
4147
+ * as the initial turn instead of a parallel, divergent implementation.
4307
4148
  */
4308
4149
  async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
4309
4150
  const routePath = opts.routePath ?? "/v1/messages (advisor lead shim)";
@@ -4338,8 +4179,8 @@ async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
4338
4179
  * Build an injectable `continueTurn(body, signal)` for `buildAdvisorStream`
4339
4180
  * (`src/services/advisor/advisor.ts`) that routes a continuation turn
4340
4181
  * through THIS module's non-Claude shim instead of Claude passthrough — used
4341
- * for the fast Luna-lead profile's advisor translate-loop. No `onCancel` is
4342
- * threaded through: the advisor loop's own `aborter` (shared with `signal`
4182
+ * by any non-Claude model selected in an authenticated fast primary lead.
4183
+ * No `onCancel` is threaded through: the advisor loop's own `aborter` (shared with `signal`
4343
4184
  * here) already tears down on consumer cancel via `buildAdvisorStream`'s
4344
4185
  * `cancel()`, so this stream needs no independent teardown hook.
4345
4186
  */
@@ -4551,7 +4392,7 @@ function preprocessFastRequest(rawBody, launch) {
4551
4392
  originalModel,
4552
4393
  modified: false
4553
4394
  };
4554
- const bare = originalModel.replace(/\[1m\]$/i, "");
4395
+ const { base: bare } = stripTrailingOneMSuffix(originalModel);
4555
4396
  let effort;
4556
4397
  if (alias) {
4557
4398
  effort = alias.absentEffortDefault;
@@ -4581,6 +4422,216 @@ function preprocessFastRequest(rawBody, launch) {
4581
4422
  };
4582
4423
  }
4583
4424
  //#endregion
4425
+ //#region src/lib/prompt-window-salvage.ts
4426
+ /**
4427
+ * Keep enough room for Copilot's message framing and for small differences
4428
+ * between counting the JSON text here and counting the decoded request there.
4429
+ */
4430
+ const PROMPT_WINDOW_RESERVE = 2e3;
4431
+ const TOOL_RESULT_STUB = "[earlier tool output elided to fit prompt window]";
4432
+ const MESSAGE_TEXT_STUB = "[earlier message text elided to fit prompt window]";
4433
+ const MARKER_RE = /^\[github-router: elided ~([0-9]+) tokens of older tool output to fit this model's prompt window\]$/;
4434
+ function markerText(tokens) {
4435
+ return `[github-router: elided ~${tokens} tokens of older tool output to fit this model's prompt window]`;
4436
+ }
4437
+ function isRecord(value) {
4438
+ return typeof value === "object" && value !== null;
4439
+ }
4440
+ function textReplacement(block, replacement) {
4441
+ if (block.type !== "text" || typeof block.text !== "string") return void 0;
4442
+ if (block.text === replacement) return void 0;
4443
+ return {
4444
+ original: block.text,
4445
+ replacement,
4446
+ apply: () => {
4447
+ block.text = replacement;
4448
+ }
4449
+ };
4450
+ }
4451
+ function collectToolResultReplacements(messages, lastMessageIndex) {
4452
+ const replacements = [];
4453
+ for (let i = 0; i < lastMessageIndex; i += 1) {
4454
+ const message = messages[i];
4455
+ if (!isRecord(message) || message.role !== "user" || !Array.isArray(message.content)) continue;
4456
+ for (const block of message.content) {
4457
+ if (!isRecord(block) || block.type !== "tool_result") continue;
4458
+ if (typeof block.content === "string") {
4459
+ if (block.content === TOOL_RESULT_STUB) continue;
4460
+ const original = block.content;
4461
+ replacements.push({
4462
+ original,
4463
+ replacement: TOOL_RESULT_STUB,
4464
+ apply: () => {
4465
+ block.content = TOOL_RESULT_STUB;
4466
+ }
4467
+ });
4468
+ continue;
4469
+ }
4470
+ if (!Array.isArray(block.content)) continue;
4471
+ for (const nested of block.content) {
4472
+ if (!isRecord(nested)) continue;
4473
+ const replacement = textReplacement(nested, TOOL_RESULT_STUB);
4474
+ if (replacement) replacements.push(replacement);
4475
+ }
4476
+ }
4477
+ }
4478
+ return replacements;
4479
+ }
4480
+ function collectMessageTextReplacements(messages, lastMessageIndex) {
4481
+ const replacements = [];
4482
+ for (let i = 0; i < lastMessageIndex; i += 1) {
4483
+ const message = messages[i];
4484
+ if (!isRecord(message)) continue;
4485
+ if (message.role !== "assistant" && message.role !== "user") continue;
4486
+ if (typeof message.content === "string") {
4487
+ if (message.content === MESSAGE_TEXT_STUB) continue;
4488
+ const original = message.content;
4489
+ replacements.push({
4490
+ original,
4491
+ replacement: MESSAGE_TEXT_STUB,
4492
+ apply: () => {
4493
+ message.content = MESSAGE_TEXT_STUB;
4494
+ }
4495
+ });
4496
+ continue;
4497
+ }
4498
+ if (!Array.isArray(message.content)) continue;
4499
+ for (const block of message.content) {
4500
+ if (!isRecord(block)) continue;
4501
+ const replacement = textReplacement(block, MESSAGE_TEXT_STUB);
4502
+ if (replacement) replacements.push(replacement);
4503
+ }
4504
+ }
4505
+ return replacements;
4506
+ }
4507
+ function hasSalvageStub(messages) {
4508
+ for (const message of messages.slice(0, -1)) {
4509
+ if (!isRecord(message)) continue;
4510
+ if (message.content === MESSAGE_TEXT_STUB) return true;
4511
+ if (!Array.isArray(message.content)) continue;
4512
+ for (const block of message.content) {
4513
+ if (!isRecord(block)) continue;
4514
+ if (block.type === "text" && block.text === MESSAGE_TEXT_STUB) return true;
4515
+ if (block.type !== "tool_result") continue;
4516
+ if (block.content === TOOL_RESULT_STUB) return true;
4517
+ if (Array.isArray(block.content) && block.content.some((nested) => isRecord(nested) && nested.type === "text" && nested.text === TOOL_RESULT_STUB)) return true;
4518
+ }
4519
+ }
4520
+ return false;
4521
+ }
4522
+ function ensureMarker(messages) {
4523
+ const last = messages.at(-1);
4524
+ if (!isRecord(last) || last.role !== "user") return void 0;
4525
+ if (typeof last.content === "string") last.content = [{
4526
+ type: "text",
4527
+ text: last.content
4528
+ }];
4529
+ if (!Array.isArray(last.content)) return void 0;
4530
+ for (const contentBlock of last.content) {
4531
+ if (!isRecord(contentBlock)) continue;
4532
+ if (contentBlock.type !== "text" || typeof contentBlock.text !== "string") continue;
4533
+ const match = MARKER_RE.exec(contentBlock.text);
4534
+ if (!match) continue;
4535
+ if (!hasSalvageStub(messages)) return void 0;
4536
+ const parsed = Number.parseInt(match[1], 10);
4537
+ if (!Number.isSafeInteger(parsed) || parsed < 0) return void 0;
4538
+ return {
4539
+ block: contentBlock,
4540
+ priorElidedTokens: parsed
4541
+ };
4542
+ }
4543
+ const block = {
4544
+ type: "text",
4545
+ text: markerText(0)
4546
+ };
4547
+ last.content.push(block);
4548
+ return {
4549
+ block,
4550
+ priorElidedTokens: 0
4551
+ };
4552
+ }
4553
+ async function fragmentTokenSavings(replacement, encoding) {
4554
+ const [before, after] = await Promise.all([getTextTokenCount(JSON.stringify(replacement.original), encoding), getTextTokenCount(JSON.stringify(replacement.replacement), encoding)]);
4555
+ return before - after;
4556
+ }
4557
+ /**
4558
+ * Last-resort history salvage for requests that escaped the client's normal
4559
+ * compaction path. Protected content is never rewritten, and the original
4560
+ * string is returned unless a complete, valid-looking salvage fits the live
4561
+ * model budget.
4562
+ */
4563
+ async function salvageOversizedPrompt(rawBody, model) {
4564
+ const unchanged = {
4565
+ body: rawBody,
4566
+ salvaged: false
4567
+ };
4568
+ if (!model) return unchanged;
4569
+ const maxPromptTokens = model.capabilities?.limits?.max_prompt_tokens;
4570
+ if (typeof maxPromptTokens !== "number" || !Number.isFinite(maxPromptTokens) || maxPromptTokens <= 0) return unchanged;
4571
+ const budget = Math.floor(maxPromptTokens) - PROMPT_WINDOW_RESERVE;
4572
+ if (budget <= 0) return unchanged;
4573
+ if (Buffer.byteLength(rawBody, "utf8") <= budget) return unchanged;
4574
+ const encoding = getTokenizerFromModel(model);
4575
+ let initialTokens;
4576
+ try {
4577
+ initialTokens = await getTextTokenCount(rawBody, encoding);
4578
+ } catch (error) {
4579
+ consola.debug("Prompt-window salvage tokenization failed; allowing request:", error);
4580
+ return unchanged;
4581
+ }
4582
+ if (initialTokens <= budget) return unchanged;
4583
+ let body;
4584
+ try {
4585
+ const parsed = JSON.parse(rawBody);
4586
+ if (!isRecord(parsed) || !Array.isArray(parsed.messages) || parsed.messages.length === 0) return unchanged;
4587
+ body = parsed;
4588
+ } catch {
4589
+ return unchanged;
4590
+ }
4591
+ const messages = body.messages;
4592
+ const marker = ensureMarker(messages);
4593
+ if (!marker) return unchanged;
4594
+ const replacements = [...collectToolResultReplacements(messages, messages.length - 1), ...collectMessageTextReplacements(messages, messages.length - 1)];
4595
+ let newlyElidedTokens = 0;
4596
+ let estimatedTokens = initialTokens;
4597
+ for (const replacement of replacements) {
4598
+ let savings;
4599
+ try {
4600
+ savings = await fragmentTokenSavings(replacement, encoding);
4601
+ } catch (error) {
4602
+ consola.debug("Prompt-window salvage tokenization failed; allowing request:", error);
4603
+ return unchanged;
4604
+ }
4605
+ if (savings <= 0) continue;
4606
+ replacement.apply();
4607
+ newlyElidedTokens += savings;
4608
+ estimatedTokens -= savings;
4609
+ const totalElidedTokens = marker.priorElidedTokens + newlyElidedTokens;
4610
+ marker.block.text = markerText(totalElidedTokens);
4611
+ if (estimatedTokens > budget) continue;
4612
+ let serialized;
4613
+ let finalTokens;
4614
+ try {
4615
+ serialized = JSON.stringify(body);
4616
+ finalTokens = await getTextTokenCount(serialized, encoding);
4617
+ } catch (error) {
4618
+ consola.debug("Prompt-window salvage serialization failed; allowing request:", error);
4619
+ return unchanged;
4620
+ }
4621
+ if (finalTokens > budget) {
4622
+ estimatedTokens = finalTokens;
4623
+ continue;
4624
+ }
4625
+ consola.warn(`Prompt-window salvage: model=${model.id} tokens=${initialTokens} budget=${budget} elided=${totalElidedTokens}`);
4626
+ return {
4627
+ body: serialized,
4628
+ salvaged: true,
4629
+ elidedTokens: totalElidedTokens
4630
+ };
4631
+ }
4632
+ return unchanged;
4633
+ }
4634
+ //#endregion
4584
4635
  //#region src/routes/messages/handler.ts
4585
4636
  const MAX_THINKING_REPAIR_ATTEMPTS = 5;
4586
4637
  const isWebSearchTool$1 = (tool) => !!tool && typeof tool === "object" && (typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search");
@@ -4641,13 +4692,13 @@ function stripWebSearchTool(body) {
4641
4692
  }
4642
4693
  /**
4643
4694
  * Strip the injected `__anthropic_advisor` tool (and any Anthropic-native
4644
- * `advisor_*` typed tool) from a request body. Used ONLY on the non-Claude
4645
- * shim path: ADVISOR's server-side translate-loop (buildAdvisorStream) lives
4646
- * on the native /v1/messages route, so a non-Claude model has no handler for
4647
- * the tool and it must be removed before forwarding — otherwise the model
4648
- * could emit a tool_use that nothing fulfils. Mirrors stripWebSearchTool's
4649
- * tool_choice cleanup. Returns the original string (same reference) when
4650
- * nothing was removed.
4695
+ * `advisor_*` typed tool) from a request body. Used on non-Claude shim paths
4696
+ * without an advisor handler and on authenticated fast Task-subagent requests,
4697
+ * where Advisor is intentionally lead-only. Mirrors stripWebSearchTool's
4698
+ * tool_choice cleanup. Returns the original string when nothing was removed.
4699
+ * End-to-end evidence for both stripped tool forms and the resulting 200 lives
4700
+ * in probes `shim_advisor_degrade_gpt55` and
4701
+ * `shim_advisor_degrade_gemini35flash`.
4651
4702
  */
4652
4703
  function stripAdvisorTool(rawBody) {
4653
4704
  let body;
@@ -4780,7 +4831,11 @@ async function handleCompletion(c) {
4780
4831
  if (state.manualApprove) await awaitApproval();
4781
4832
  const betaHeaders = extractBetaHeaders(c);
4782
4833
  const incomingBeta = c.req.header("anthropic-beta");
4783
- const advisorEnabled = isAdvisorRequested(incomingBeta);
4834
+ const advisorRequested = isAdvisorRequested(incomingBeta);
4835
+ const fastProfileRequest = identity.launch?.profileId === "fast";
4836
+ const fastSubagentRequest = fastProfileRequest && Boolean(c.req.header("x-claude-code-agent-id"));
4837
+ const fastLeadAdvisor = fastProfileRequest && !fastSubagentRequest;
4838
+ const advisorEnabled = advisorRequested && !fastSubagentRequest;
4784
4839
  const fastPreprocess = preprocessFastRequest(rawBody, identity.launch);
4785
4840
  if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
4786
4841
  const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
@@ -4794,6 +4849,7 @@ async function handleCompletion(c) {
4794
4849
  }
4795
4850
  let finalBody = await processWebSearch(fastPreprocess.body);
4796
4851
  finalBody = sanitizeAnthropicBody(finalBody);
4852
+ if (fastSubagentRequest) finalBody = stripAdvisorTool(finalBody);
4797
4853
  const loopGuard = guardAnthropicBody(finalBody);
4798
4854
  if (loopGuard.action === "abort") return c.json({
4799
4855
  type: "error",
@@ -4804,7 +4860,7 @@ async function handleCompletion(c) {
4804
4860
  }, 400, { "x-should-retry": "false" });
4805
4861
  if (loopGuard.body !== void 0) finalBody = loopGuard.body;
4806
4862
  if (advisorEnabled) {
4807
- finalBody = injectAdvisorTool(finalBody);
4863
+ finalBody = injectAdvisorTool(finalBody, fastLeadAdvisor ? FAST_ADVISOR_TOOL_INSTRUCTIONS : void 0);
4808
4864
  consola.info("ADVISOR enabled for this request — injecting __anthropic_advisor tool; will translate tool_use → server_tool_use{advisor} on the SSE stream");
4809
4865
  }
4810
4866
  if (finalBody.includes("\"mcp_servers\"")) try {
@@ -4819,15 +4875,16 @@ async function handleCompletion(c) {
4819
4875
  } catch {}
4820
4876
  const { body: resolvedBody, originalModel, resolvedModel, selectedModel } = resolveModelInBody$1(finalBody);
4821
4877
  const modelId = resolvedModel ?? originalModel;
4822
- const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel);
4878
+ const { body: promptWindowBody } = await salvageOversizedPrompt(resolvedBody, selectedModel);
4879
+ const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel, fastProfileRequest);
4823
4880
  if (messagesRoute !== "claude-passthrough") {
4824
4881
  const endpoint = messagesRoute === "chat-shim" ? "chat" : "responses";
4825
4882
  let parsedBase;
4826
4883
  try {
4827
- parsedBase = JSON.parse(resolvedBody);
4884
+ parsedBase = JSON.parse(promptWindowBody);
4828
4885
  } catch {}
4829
4886
  const wantsStream = parsedBase?.stream === true;
4830
- if (advisorEnabled && wantsStream && identity.launch?.profileId === "fast" && isFastProfileLead(modelId)) {
4887
+ if (advisorEnabled && wantsStream && fastLeadAdvisor) {
4831
4888
  const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
4832
4889
  const parsedInitial = parseAnthropicRequest(parsedBase, modelId, selectedModel);
4833
4890
  const fastAdvisorAborter = new AbortController();
@@ -4845,14 +4902,15 @@ async function handleCompletion(c) {
4845
4902
  status: 200,
4846
4903
  streaming: true
4847
4904
  }, selectedModel, startTime);
4848
- const advisorChoice = resolveAdvisorModel(modelId, true);
4905
+ const advisorChoice = resolveAdvisorModel(modelId, fastLeadAdvisor);
4849
4906
  return new Response(buildAdvisorStream({
4850
4907
  firstResponse,
4851
4908
  initialConversation,
4852
4909
  baseBody: parsedBase,
4853
4910
  requestHeaders: {},
4854
4911
  advisorModel: advisorChoice.model,
4855
- advisorEscalated: advisorChoice.escalated || advisorChoice.fastProfile,
4912
+ advisorEscalated: advisorChoice.escalated,
4913
+ advisorFastProfile: fastLeadAdvisor,
4856
4914
  advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model, true),
4857
4915
  externalAborter: fastAdvisorAborter,
4858
4916
  continueTurn: makeShimContinueTurn(endpoint, {
@@ -4869,7 +4927,7 @@ async function handleCompletion(c) {
4869
4927
  }
4870
4928
  });
4871
4929
  }
4872
- const shimBody = stripAdvisorTool(resolvedBody);
4930
+ const shimBody = stripAdvisorTool(promptWindowBody);
4873
4931
  if (advisorEnabled) consola.info("ADVISOR requested with a non-Claude model — stripping the injected __anthropic_advisor tool and proceeding without advisor (Claude-only feature; gracefully degraded).");
4874
4932
  const shimOpts = {
4875
4933
  rawBody: shimBody,
@@ -4887,7 +4945,7 @@ async function handleCompletion(c) {
4887
4945
  ...selectedModel?.requestHeaders,
4888
4946
  ...effectiveBetas
4889
4947
  };
4890
- let nativeBody = resolvedBody;
4948
+ let nativeBody = promptWindowBody;
4891
4949
  const knownThinkingRepair = repairKnownThinkingHistory(nativeBody);
4892
4950
  if (knownThinkingRepair) {
4893
4951
  nativeBody = knownThinkingRepair.body;
@@ -4962,7 +5020,7 @@ async function handleCompletion(c) {
4962
5020
  parsedBase = JSON.parse(nativeBody);
4963
5021
  } catch {}
4964
5022
  const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
4965
- const advisorChoice = resolveAdvisorModel(originalModel);
5023
+ const advisorChoice = resolveAdvisorModel(originalModel, fastLeadAdvisor);
4966
5024
  return new Response(buildAdvisorStream({
4967
5025
  firstResponse: response,
4968
5026
  initialConversation,
@@ -4970,7 +5028,8 @@ async function handleCompletion(c) {
4970
5028
  requestHeaders,
4971
5029
  advisorModel: advisorChoice.model,
4972
5030
  advisorEscalated: advisorChoice.escalated,
4973
- advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model),
5031
+ advisorFastProfile: fastLeadAdvisor,
5032
+ advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model, fastLeadAdvisor),
4974
5033
  externalAborter: advisorAborter
4975
5034
  }), {
4976
5035
  status: response.status,
@@ -6548,10 +6607,66 @@ function getClaudeCodeEnvVars(serverUrl, model, launchProfileId = "standard") {
6548
6607
  if (nativeModels.length > 0) {
6549
6608
  if (seedGatewayModelCache(serverUrl, nativeModels) && process.env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0 && vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0) vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY = "1";
6550
6609
  } else clearGatewayModelCache();
6610
+ applyAutoCompactWindow(vars);
6551
6611
  if (toolbeltEnabled()) Object.assign(vars, toolbeltPathOverride(process.env, PATHS.TOOLBELT_BIN_DIR));
6552
6612
  return vars;
6553
6613
  }
6554
6614
  /**
6615
+ * Every env var whose value can become the ACTIVE main-loop model id, and so
6616
+ * whose prompt ceiling the compaction window has to respect. `ANTHROPIC_MODEL`
6617
+ * is the lead; the tier rows and the custom option each become the active id
6618
+ * verbatim when picked from `/model`.
6619
+ *
6620
+ * `ANTHROPIC_SMALL_FAST_MODEL` is deliberately absent: background ops run
6621
+ * there, but compaction itself runs on the main-loop model (verified in the
6622
+ * 2.1.247 bundle — the summarizer starts from `r.options.mainLoopModel`).
6623
+ */
6624
+ const LEAD_CAPABLE_MODEL_ENV_KEYS = [
6625
+ "ANTHROPIC_MODEL",
6626
+ "ANTHROPIC_DEFAULT_OPUS_MODEL",
6627
+ "ANTHROPIC_DEFAULT_SONNET_MODEL",
6628
+ "ANTHROPIC_DEFAULT_HAIKU_MODEL",
6629
+ "ANTHROPIC_CUSTOM_MODEL_OPTION"
6630
+ ];
6631
+ /**
6632
+ * Resolve one seeded env value to its live catalog entry: strip the `[1m]`
6633
+ * accounting bracket, erase router-owned alias provenance, then translate the
6634
+ * Anthropic-dashed slug onto Copilot's catalog id.
6635
+ */
6636
+ function catalogEntryForSeededModel(value) {
6637
+ const { base } = stripTrailingOneMSuffix(canonicalizeAliasModel(value));
6638
+ const id = resolveModel(base);
6639
+ return state.models?.data?.find((m) => m.id === id);
6640
+ }
6641
+ /**
6642
+ * Set `CLAUDE_CODE_AUTO_COMPACT_WINDOW` to the smallest complete derived
6643
+ * window across every 1M-accounted model this launch can reach. Each candidate
6644
+ * puts the client's reactive trigger at 85% of that model's prompt ceiling;
6645
+ * the minimum is therefore safe after any `/model` switch.
6646
+ *
6647
+ * Only `[1m]`-decorated candidates participate. An undecorated row already
6648
+ * budgets at the client's conservative 200K default, which is below every
6649
+ * prompt ceiling in the lineup, so including it would drag the window down for
6650
+ * no benefit. When nothing is decorated, or no candidate carries usable
6651
+ * catalog limits, the variable is omitted entirely rather than guessed.
6652
+ *
6653
+ * Presence-guarded on the parent env, symmetric with every other guard here.
6654
+ */
6655
+ function applyAutoCompactWindow(vars) {
6656
+ if (process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW !== void 0) return;
6657
+ let tightestWindow;
6658
+ const consider = (value) => {
6659
+ if (!value || !/\[1m\]/i.test(value)) return;
6660
+ const limits = catalogEntryForSeededModel(value)?.capabilities?.limits;
6661
+ const candidateWindow = deriveAutoCompactWindowTokens(limits?.max_prompt_tokens, limits?.max_output_tokens);
6662
+ if (candidateWindow !== void 0 && (tightestWindow === void 0 || candidateWindow < tightestWindow)) tightestWindow = candidateWindow;
6663
+ };
6664
+ for (const key of LEAD_CAPABLE_MODEL_ENV_KEYS) consider(vars[key] ?? process.env[key]);
6665
+ for (const model of nativeSelectableModelsInCatalog()) consider(model.id);
6666
+ if (tightestWindow === void 0) return;
6667
+ vars.CLAUDE_CODE_AUTO_COMPACT_WINDOW = String(tightestWindow);
6668
+ }
6669
+ /**
6555
6670
  * Build environment variables for Codex CLI.
6556
6671
  *
6557
6672
  * Like `getClaudeCodeEnvVars`, the parent env is sanitized of
@@ -6573,6 +6688,6 @@ function getCodexEnvVars(serverUrl) {
6573
6688
  return vars;
6574
6689
  }
6575
6690
  //#endregion
6576
- export { validateFastProfilePrerequisites as _, sharedServerArgs as a, updateClaude as b, stopKeepAwake as c, LUNA_DRIVER_ALIAS_ID as d, LUNA_IMPLEMENTER_ALIAS_ID as f, resolveLaunchProfile as g, profileDescriptor as h, setupAndServe as i, listModelsForEndpoint as l, formatFastPrerequisiteFailure as m, getCodexEnvVars as n, LAUNCH_SECRET_HEADER as o, LUNA_SCOUT_ALIAS_ID as p, parseSharedArgs as r, startKeepAwake as s, getClaudeCodeEnvVars as t, enableFileLogging as u, runSelfUpdate as v, checkClaudeVersion as y };
6691
+ export { sharedServerArgs as a, stopKeepAwake as c, runSelfUpdate as d, checkClaudeVersion as f, setupAndServe as i, listModelsForEndpoint as l, getCodexEnvVars as n, LAUNCH_SECRET_HEADER as o, updateClaude as p, parseSharedArgs as r, startKeepAwake as s, getClaudeCodeEnvVars as t, enableFileLogging as u };
6577
6692
 
6578
- //# sourceMappingURL=server-setup-CqlaZukJ.js.map
6693
+ //# sourceMappingURL=server-setup-DsGJnJ_O.js.map