github-router 0.3.289 → 0.3.293

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/dist/{attribution-settings-Cmz2jt7P.js → attribution-settings-Dx8TBPE8.js} +124 -34
  2. package/dist/attribution-settings-Dx8TBPE8.js.map +1 -0
  3. package/dist/{auth-DG4vh8-F.js → auth-BwUHopJz.js} +3 -3
  4. package/dist/{auth-DG4vh8-F.js.map → auth-BwUHopJz.js.map} +1 -1
  5. package/dist/browser-ext/manifest.json +1 -1
  6. package/dist/{check-usage-BqN7mBYv.js → check-usage-BTda5753.js} +4 -4
  7. package/dist/{check-usage-BqN7mBYv.js.map → check-usage-BTda5753.js.map} +1 -1
  8. package/dist/{claude-C-9xFI4b.js → claude-BoAd_L2W.js} +118 -59
  9. package/dist/claude-BoAd_L2W.js.map +1 -0
  10. package/dist/{codex-DRW0yb8x.js → codex-BdJgGT6Q.js} +5 -5
  11. package/dist/{codex-DRW0yb8x.js.map → codex-BdJgGT6Q.js.map} +1 -1
  12. package/dist/{debug-B5TjPTTH.js → debug-B3UrZTHQ.js} +2 -2
  13. package/dist/{debug-B5TjPTTH.js.map → debug-B3UrZTHQ.js.map} +1 -1
  14. package/dist/engine-CK2b_cTt.js +2 -0
  15. package/dist/{gate-discovery-Cz6kwIVG.js → gate-discovery-C30rfnyo.js} +5 -5
  16. package/dist/{gate-discovery-Cz6kwIVG.js.map → gate-discovery-C30rfnyo.js.map} +1 -1
  17. package/dist/{get-copilot-usage-BjA0nyGR.js → get-copilot-usage-CRrf1ZSC.js} +2 -2
  18. package/dist/{get-copilot-usage-BjA0nyGR.js.map → get-copilot-usage-CRrf1ZSC.js.map} +1 -1
  19. package/dist/{internal-artifact-open-BskmUpnb.js → internal-artifact-open-BsEQRvDi.js} +2 -2
  20. package/dist/{internal-artifact-open-BskmUpnb.js.map → internal-artifact-open-BsEQRvDi.js.map} +1 -1
  21. package/dist/{internal-first-mate-guard-5XEHMaqy.js → internal-first-mate-guard-CVFFJriA.js} +3 -3
  22. package/dist/{internal-first-mate-guard-5XEHMaqy.js.map → internal-first-mate-guard-CVFFJriA.js.map} +1 -1
  23. package/dist/{internal-first-mate-guard-DHDQ6hFz.js → internal-first-mate-guard-DJdX6t48.js} +1 -1
  24. package/dist/{internal-plan-review-BUuMx4ku.js → internal-plan-review-8TKDLco6.js} +3 -3
  25. package/dist/{internal-plan-review-BUuMx4ku.js.map → internal-plan-review-8TKDLco6.js.map} +1 -1
  26. package/dist/{internal-prompt-submit-CQQ15xdO.js → internal-prompt-submit-Da7pxqua.js} +4 -4
  27. package/dist/{internal-prompt-submit-CQQ15xdO.js.map → internal-prompt-submit-Da7pxqua.js.map} +1 -1
  28. package/dist/{internal-session-bind-D04W2yWI.js → internal-session-bind-BOFytA1f.js} +2 -2
  29. package/dist/{internal-session-bind-D04W2yWI.js.map → internal-session-bind-BOFytA1f.js.map} +1 -1
  30. package/dist/{internal-stop-hook-Dvkppwo7.js → internal-stop-hook-OnfK3BxE.js} +5 -5
  31. package/dist/{internal-stop-hook-Dvkppwo7.js.map → internal-stop-hook-OnfK3BxE.js.map} +1 -1
  32. package/dist/{internal-stop-review-CdByyJLc.js → internal-stop-review-CdouacHL.js} +2 -2
  33. package/dist/{internal-stop-review-CdByyJLc.js.map → internal-stop-review-CdouacHL.js.map} +1 -1
  34. package/dist/{internal-worker-guard-BIPN6Rv9.js → internal-worker-guard-Bx-itoP8.js} +2 -2
  35. package/dist/{internal-worker-guard-BIPN6Rv9.js.map → internal-worker-guard-Bx-itoP8.js.map} +1 -1
  36. package/dist/{internal-workspace-header-BKqejstG.js → internal-workspace-header-8WT0iB5K.js} +2 -2
  37. package/dist/{internal-workspace-header-BKqejstG.js.map → internal-workspace-header-8WT0iB5K.js.map} +1 -1
  38. package/dist/lifecycle-C8fOsQke.js +2 -0
  39. package/dist/lifecycle-D4Yc1aap.js +2 -0
  40. package/dist/{lifecycle-SXaWssN9.js → lifecycle-LeSfa7wH.js} +2 -2
  41. package/dist/{lifecycle-SXaWssN9.js.map → lifecycle-LeSfa7wH.js.map} +1 -1
  42. package/dist/{lifecycle-DbM29FLK.js → lifecycle-nuOHfwgj.js} +2 -2
  43. package/dist/{lifecycle-DbM29FLK.js.map → lifecycle-nuOHfwgj.js.map} +1 -1
  44. package/dist/main.js +17 -17
  45. package/dist/{mcp-workspace-header-DRCCWlOi.js → mcp-workspace-header-q34H_4wL.js} +2 -2
  46. package/dist/{mcp-workspace-header-DRCCWlOi.js.map → mcp-workspace-header-q34H_4wL.js.map} +1 -1
  47. package/dist/{models-Dz8d_SnI.js → models-hhJcrZhr.js} +3 -3
  48. package/dist/{models-Dz8d_SnI.js.map → models-hhJcrZhr.js.map} +1 -1
  49. package/dist/{orchestration-BrJwZxMN.js → orchestration-pzbrKkgD.js} +2 -2
  50. package/dist/{orchestration-BrJwZxMN.js.map → orchestration-pzbrKkgD.js.map} +1 -1
  51. package/dist/{paths-D7_SAaIQ.js → paths-BH4J7slC.js} +4 -4
  52. package/dist/{paths-D7_SAaIQ.js.map → paths-BH4J7slC.js.map} +1 -1
  53. package/dist/paths-DJZoXfAS.js +2 -0
  54. package/dist/{peer-mcp-personas-Bd56EmiO.js → peer-mcp-personas-DhI7ZPSx.js} +569 -136
  55. package/dist/peer-mcp-personas-DhI7ZPSx.js.map +1 -0
  56. package/dist/{plan-review-hook-CVZsG9MZ.js → plan-review-hook-CfcanA7_.js} +3 -3
  57. package/dist/{plan-review-hook-CVZsG9MZ.js.map → plan-review-hook-CfcanA7_.js.map} +1 -1
  58. package/dist/{prompt-submit-hook-BW92FX2D.js → prompt-submit-hook-Bqf9ORgb.js} +3 -3
  59. package/dist/{prompt-submit-hook-BW92FX2D.js.map → prompt-submit-hook-Bqf9ORgb.js.map} +1 -1
  60. package/dist/{provision-B53wbHwa.js → provision-CZJ4EWls.js} +4 -4
  61. package/dist/{provision-B53wbHwa.js.map → provision-CZJ4EWls.js.map} +1 -1
  62. package/dist/{self-invocation-CP_SOkrr.js → self-invocation-DhO1Z8iD.js} +2 -2
  63. package/dist/{self-invocation-CP_SOkrr.js.map → self-invocation-DhO1Z8iD.js.map} +1 -1
  64. package/dist/{serve-aZCEYFe5.js → serve-RxWYzoOl.js} +12 -12
  65. package/dist/{serve-aZCEYFe5.js.map → serve-RxWYzoOl.js.map} +1 -1
  66. package/dist/{server-setup-DlztZAGT.js → server-setup-DbvbW5Ve.js} +627 -85
  67. package/dist/server-setup-DbvbW5Ve.js.map +1 -0
  68. package/dist/{start-DwNiXv5N.js → start-BNcGsXsY.js} +3 -3
  69. package/dist/{start-DwNiXv5N.js.map → start-BNcGsXsY.js.map} +1 -1
  70. package/dist/{stop-gate-hook-DriRc9xN.js → stop-gate-hook-BiBp5aGm.js} +3 -3
  71. package/dist/{stop-gate-hook-DriRc9xN.js.map → stop-gate-hook-BiBp5aGm.js.map} +1 -1
  72. package/dist/{stop-gate-policy-DMPanpoR.js → stop-gate-policy-BGd6b5hR.js} +2 -2
  73. package/dist/{stop-gate-policy-DMPanpoR.js.map → stop-gate-policy-BGd6b5hR.js.map} +1 -1
  74. package/dist/{token-BGCjZwtj.js → token-8drORhXg.js} +33 -3
  75. package/dist/token-8drORhXg.js.map +1 -0
  76. package/dist/{worker-dispatch-BCTMyNE-.js → worker-dispatch-D5fGroNr.js} +2 -2
  77. package/dist/{worker-dispatch-BCTMyNE-.js.map → worker-dispatch-D5fGroNr.js.map} +1 -1
  78. package/package.json +1 -1
  79. package/dist/attribution-settings-Cmz2jt7P.js.map +0 -1
  80. package/dist/claude-C-9xFI4b.js.map +0 -1
  81. package/dist/engine-iEqGdx6T.js +0 -2
  82. package/dist/lifecycle-BTodQvn4.js +0 -2
  83. package/dist/lifecycle-C7JYNz-F.js +0 -2
  84. package/dist/paths-CTr59UC6.js +0 -2
  85. package/dist/peer-mcp-personas-Bd56EmiO.js.map +0 -1
  86. package/dist/server-setup-DlztZAGT.js.map +0 -1
  87. package/dist/token-BGCjZwtj.js.map +0 -1
@@ -1,8 +1,8 @@
1
- import { $ as handleMcpDelete, $t as UPSTREAM_INACTIVITY_TIMEOUT_MS, At as readResponseBodyCapped, B as rememberThinkingHistoryRepair, Bt as provisionTreeSitterAssets, Ct as getTokenCount, Dt as createResponses, Et as resolveMcpToolTimeoutMs, F as injectAdvisorTool, G as isControllerClosedError, Gt as toolbeltPathOverride, H as repairRejectedThinkingHistory, I as isAdvisorRequested, J as relayAnthropicStream, K as logStreamError, L as resolveAdvisorEffort, M as ADVISOR_INTERNAL_TOOL_NAME, Mt as normalizeOpenAIUsage, N as ADVISOR_TOOL_INSTRUCTIONS, Ot as createChatCompletions, P as buildAdvisorStream, Q as clampEffort, Qt as UPSTREAM_FETCH_TIMEOUT_MS, R as resolveAdvisorModel, St as createMessages, T as toolbeltEnabled, Tt as warnOnTokenPriceDrift, U as buildAnthropicErrorEvent, V as repairKnownThinkingHistory, W as buildOpenAIErrorEvent, X as UNKNOWN_EFFORT_ANCHOR, Y as EFFORT_ORDER, Z as bucketEffort, an as upstreamMaxConnections, bt as shimDefaultsToXhigh, cn as withOneMSuffixForLead, en as generateRandomPort, et as handleMcpPost, in as upstreamAllowH2, j as searchWeb, jt as parseJsonOrDiagnose, kt as MAX_RESPONSE_BODY_BYTES, ln as withInstallLock, nt as agentToolsEnabled, on as classifyMessagesRoute, pt as resolveGeminiReviewModel, q as readIteratorWithTimeout, qt as BUDGET_SMALL_FAST_SLUG, r as assertMcpToolSurfaceConsistent, sn as withOneMSuffix, tn as isBudgetClaudeLead, wt as assembleResponsesPayload, xt as countTokens, z as formatThinkingRepairDecline } from "./peer-mcp-personas-Bd56EmiO.js";
1
+ import { $ as UNKNOWN_EFFORT_ANCHOR, $t as provisionTreeSitterAssets, At as shimDefaultsToXhigh, B as resolveAdvisorEffort, Bt as createResponses, Cn as withOneMSuffix, E as toolbeltEnabled, F as FAST_ADVISOR_TOOL_INSTRUCTIONS, G as repairRejectedThinkingHistory, Gt as normalizeOpenAIUsage, H as formatThinkingRepairDecline, Ht as MAX_RESPONSE_BODY_BYTES, I as buildAdvisorStream, J as isControllerClosedError, K as buildAnthropicErrorEvent, L as injectAdvisorTool, Lt as assembleResponsesPayload, M as searchWeb, Mt as createMessages, N as ADVISOR_INTERNAL_TOOL_NAME, Nt as getTokenCount, P as ADVISOR_TOOL_INSTRUCTIONS, Pt as findLaunchBySecret, Q as EFFORT_ORDER, R as isAdvisorRequested, Rt as warnOnTokenPriceDrift, Sn as oneMContextDisabled, Tn as withInstallLock, U as rememberThinkingHistoryRepair, Ut as readResponseBodyCapped, V as resolveAdvisorModel, Vt as createChatCompletions, W as repairKnownThinkingHistory, Wt as parseJsonOrDiagnose, X as readIteratorWithTimeout, Y as logStreamError, Z as relayAnthropicStream, _n as upstreamAllowH2, bn as pickEndpoint, dn as UPSTREAM_FETCH_TIMEOUT_MS, et as bucketEffort, fn as UPSTREAM_INACTIVITY_TIMEOUT_MS, i as assertMcpToolSurfaceConsistent, in as toolbeltPathOverride, jt as countTokens, mn as isBudgetClaudeLead, nt as handleMcpDelete, on as BUDGET_SMALL_FAST_SLUG, pn as generateRandomPort, q as buildOpenAIErrorEvent, rt as handleMcpPost, st as agentToolsEnabled, tt as clampEffort, vn as upstreamMaxConnections, wn as withOneMSuffixForLead, xn as catalogAdvertises1M, yn as classifyMessagesRoute, z as isFastProfileLead, zt as resolveMcpToolTimeoutMs } from "./peer-mcp-personas-DhI7ZPSx.js";
2
2
  import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
3
- import { i as ensurePaths, t as PATHS } from "./paths-D7_SAaIQ.js";
4
- import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-BGCjZwtj.js";
5
- import { t as getCopilotUsage } from "./get-copilot-usage-BjA0nyGR.js";
3
+ import { i as ensurePaths, t as PATHS } from "./paths-BH4J7slC.js";
4
+ import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-8drORhXg.js";
5
+ import { t as getCopilotUsage } from "./get-copilot-usage-CRrf1ZSC.js";
6
6
  import { a as resolveExecutable, n as killManagedTree, o as runCommandCapture, r as parseBoolEnv, s as runCommandVoid } from "./exec-y8C_MU8A.js";
7
7
  import consola from "consola";
8
8
  import * as fs$2 from "node:fs";
@@ -354,6 +354,232 @@ async function runSelfUpdate(opts) {
354
354
  }
355
355
  }
356
356
  //#endregion
357
+ //#region src/lib/launch-profile.ts
358
+ const STANDARD_PROFILE = Object.freeze({
359
+ id: "standard",
360
+ hasCoordinator: true
361
+ });
362
+ /**
363
+ * The `-m fast` roster: exactly four native agents (`scout`, `implementer`,
364
+ * `reviewer`, `planner`), the fast-only `oracle` peer tool, no coordinator,
365
+ * and only `peers`/`search` plus the ordinary opt-in `browser` group.
366
+ * `workers`/`orchestrate`/`decide`/`fleet`/`first-mate` are hard denies even
367
+ * when their independent standard-profile gates pass.
368
+ */
369
+ const FAST_PROFILE = Object.freeze({
370
+ id: "fast",
371
+ nativeRoster: /* @__PURE__ */ new Set([
372
+ "scout",
373
+ "implementer",
374
+ "reviewer",
375
+ "planner"
376
+ ]),
377
+ personaAllowlist: /* @__PURE__ */ new Set(["oracle"]),
378
+ allowedGroups: /* @__PURE__ */ new Set([
379
+ "peers",
380
+ "search",
381
+ "browser"
382
+ ]),
383
+ hasCoordinator: false
384
+ });
385
+ function profileDescriptor(id) {
386
+ return id === "fast" ? FAST_PROFILE : STANDARD_PROFILE;
387
+ }
388
+ /**
389
+ * Resolve the parsed `-m` argument to a launch profile.
390
+ *
391
+ * Deliberately keyed on the RAW alias string (trimmed, case-insensitive
392
+ * `"fast"`), never on a resolved model id: `resolveLeadSlugArg` maps `fast`
393
+ * to `FAST_LEAD_MODEL` (`./port`) before this is of any use to a caller who
394
+ * only has the resolved id, so callers that already resolved the lead must
395
+ * pass the ORIGINAL `-m` value here, not the resolved one. This is what
396
+ * keeps `-m gpt-5.6-luna` (a direct pin of the same underlying model) a
397
+ * standard-surface launch — only the literal alias narrows the surface.
398
+ */
399
+ function resolveLaunchProfile(modelArg) {
400
+ return modelArg?.trim().toLowerCase() === "fast" ? "fast" : "standard";
401
+ }
402
+ /**
403
+ * Router-owned alias id for the fast profile's Sonnet-tier row
404
+ * (`ANTHROPIC_DEFAULT_SONNET_MODEL`). Never sent upstream — canonicalized to
405
+ * `LUNA_REAL_MODEL_ID` by `canonicalizeAliasModel` before the request
406
+ * reaches Copilot.
407
+ */
408
+ const LUNA_DRIVER_ALIAS_ID = "gh-router-luna-driver-max";
409
+ /** Fast native-agent alias ids preserve role-specific effort provenance until
410
+ * the authenticated request boundary. They both canonicalize to Luna, but the
411
+ * scout is fixed high while the implementer is fixed max. */
412
+ const LUNA_SCOUT_ALIAS_ID = "gh-router-luna-scout-high";
413
+ const LUNA_IMPLEMENTER_ALIAS_ID = "gh-router-luna-implementer-max";
414
+ const LUNA_SONNET_ALIAS_ID = "gh-router-luna-sonnet-xhigh";
415
+ /**
416
+ * Router-owned alias id for the fast profile's Haiku-tier row
417
+ * (`ANTHROPIC_DEFAULT_HAIKU_MODEL` / `ANTHROPIC_SMALL_FAST_MODEL`).
418
+ */
419
+ const LUNA_HAIKU_ALIAS_ID = "gh-router-luna-haiku-high";
420
+ /** The real Copilot catalog id every Luna alias (including the driver
421
+ * itself) canonicalizes to. */
422
+ const LUNA_REAL_MODEL_ID = "gpt-5.6-luna";
423
+ /**
424
+ * The full alias table, keyed by `aliasId`. A simpler model-id-only table is
425
+ * rejected by design: the driver, the Sonnet tier, and the Haiku tier all
426
+ * resolve to the SAME Luna catalog id, so after early canonicalization a
427
+ * table keyed on the real id could no longer tell which absent-effort
428
+ * default applies. Alias provenance — which of the three ids the request
429
+ * actually carried — is the minimum discriminator that survives from tier
430
+ * selection through to request preprocessing, which is why canonicalization
431
+ * must happen LAST (in the `/v1/messages` identity preflight), after the
432
+ * effort default has already been read off the alias.
433
+ */
434
+ const MODEL_ALIAS_TABLE = /* @__PURE__ */ new Map([
435
+ [LUNA_DRIVER_ALIAS_ID, {
436
+ aliasId: LUNA_DRIVER_ALIAS_ID,
437
+ realModel: LUNA_REAL_MODEL_ID,
438
+ absentEffortDefault: "max"
439
+ }],
440
+ [LUNA_SCOUT_ALIAS_ID, {
441
+ aliasId: LUNA_SCOUT_ALIAS_ID,
442
+ realModel: LUNA_REAL_MODEL_ID,
443
+ absentEffortDefault: "high"
444
+ }],
445
+ [LUNA_IMPLEMENTER_ALIAS_ID, {
446
+ aliasId: LUNA_IMPLEMENTER_ALIAS_ID,
447
+ realModel: LUNA_REAL_MODEL_ID,
448
+ absentEffortDefault: "max"
449
+ }],
450
+ [LUNA_SONNET_ALIAS_ID, {
451
+ aliasId: LUNA_SONNET_ALIAS_ID,
452
+ realModel: LUNA_REAL_MODEL_ID,
453
+ absentEffortDefault: "xhigh"
454
+ }],
455
+ [LUNA_HAIKU_ALIAS_ID, {
456
+ aliasId: LUNA_HAIKU_ALIAS_ID,
457
+ realModel: LUNA_REAL_MODEL_ID,
458
+ absentEffortDefault: "high"
459
+ }]
460
+ ]);
461
+ /**
462
+ * Look up the alias descriptor for a wire-facing model id (with or without
463
+ * a trailing `[1m]` bracket — the bracket is stripped before the table
464
+ * lookup and is orthogonal to alias identity). Returns undefined for any
465
+ * id that isn't one of the three registered aliases (including the bare
466
+ * `claude-*` ids and every other real Copilot catalog id).
467
+ */
468
+ function resolveModelAlias(id) {
469
+ const bare = id.replace(/\[1m\]$/i, "");
470
+ return MODEL_ALIAS_TABLE.get(bare);
471
+ }
472
+ /**
473
+ * Strip alias provenance and return the real catalog id to send upstream.
474
+ * Idempotent passthrough for any id that isn't a registered alias (a bare
475
+ * `claude-*` slug, an already-real Copilot id, or anything else) — this is
476
+ * safe to call unconditionally on every `body.model` at the outbound
477
+ * boundary. Preserves a trailing `[1m]` bracket: canonicalization only
478
+ * erases ALIAS identity, not the 1M-context accounting decoration.
479
+ */
480
+ function canonicalizeAliasModel(id) {
481
+ const bracket = /\[1m\]$/i.test(id) ? "[1m]" : "";
482
+ const bare = bracket ? id.slice(0, -bracket.length) : id;
483
+ const alias = MODEL_ALIAS_TABLE.get(bare);
484
+ return alias ? `${alias.realModel}${bracket}` : id;
485
+ }
486
+ const FAST_REQUIRED_CONTEXT_TOKENS = 1e6;
487
+ function findModel(catalog, id) {
488
+ return catalog?.data?.find((m) => m.id === id);
489
+ }
490
+ function hasToolCalls(model) {
491
+ return model?.capabilities?.supports?.tool_calls === true;
492
+ }
493
+ function hasContextAtLeast(model, tokens) {
494
+ return (model?.capabilities?.limits?.max_context_window_tokens ?? 0) >= tokens;
495
+ }
496
+ function supportsEffort(model, effort) {
497
+ const list = model?.capabilities?.supports?.reasoning_effort;
498
+ return Array.isArray(list) && list.includes(effort);
499
+ }
500
+ function supportsEndpoint(model, paths) {
501
+ const endpoints = model?.supported_endpoints;
502
+ return Array.isArray(endpoints) && endpoints.some((endpoint) => paths.has(endpoint));
503
+ }
504
+ const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
505
+ const MESSAGES_ENDPOINTS = /* @__PURE__ */ new Set(["/messages", "/v1/messages"]);
506
+ function hasUsablePromptMetadata(model) {
507
+ const prompt = model?.capabilities?.limits?.max_prompt_tokens;
508
+ return typeof prompt === "number" && Number.isFinite(prompt) && prompt > 0;
509
+ }
510
+ /**
511
+ * Validate the live Copilot catalog carries every model the fast profile's
512
+ * EXACT roster depends on, with the specific capabilities each assignment
513
+ * needs. These are capability-availability PREREQUISITES for constructing
514
+ * the roster — not an allowlist of models the user may select later in the
515
+ * session — so a partial catalog fails the whole `-m fast` launch rather
516
+ * than silently substituting or dropping an agent.
517
+ *
518
+ * Checks, per the fast-launch-profile design:
519
+ * - Luna lead/scout/implementer: tool calls, >=1M, high+max, Responses.
520
+ * - Sol planner: tool calls, >=1M, high, Responses.
521
+ * - Grok reviewer: tool calls, medium, Responses, usable prompt metadata.
522
+ * - Gemini Advisor: >=1M, high, chat-completions.
523
+ * - Opus Oracle: exact Opus 5, >=1M, adaptive/high, Messages, prompt metadata.
524
+ *
525
+ * Pure over the passed-in catalog snapshot so it's unit-testable without
526
+ * `state` — callers pass `state.models` at call time.
527
+ */
528
+ function validateFastProfilePrerequisites(catalog) {
529
+ const missing = [];
530
+ const luna = findModel(catalog, LUNA_REAL_MODEL_ID);
531
+ if (!luna) missing.push(`${LUNA_REAL_MODEL_ID}: absent from the live catalog`);
532
+ else {
533
+ if (!hasToolCalls(luna)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise tool_calls`);
534
+ if (!hasContextAtLeast(luna, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push(`${LUNA_REAL_MODEL_ID}: advertised context window is below 1M`);
535
+ if (!supportsEffort(luna, "high") || !supportsEffort(luna, "max")) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise both "high" and "max" reasoning effort`);
536
+ if (!supportsEndpoint(luna, RESPONSES_ENDPOINTS)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise a supported Responses endpoint`);
537
+ }
538
+ const sol = findModel(catalog, "gpt-5.6-sol");
539
+ if (!sol) missing.push("gpt-5.6-sol: absent from the live catalog");
540
+ else {
541
+ if (!hasToolCalls(sol)) missing.push("gpt-5.6-sol: does not advertise tool_calls");
542
+ if (!hasContextAtLeast(sol, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gpt-5.6-sol: advertised context window is below 1M");
543
+ if (!supportsEffort(sol, "high")) missing.push("gpt-5.6-sol: does not advertise a \"high\" reasoning effort");
544
+ if (!supportsEndpoint(sol, RESPONSES_ENDPOINTS)) missing.push("gpt-5.6-sol: does not advertise a supported Responses endpoint");
545
+ }
546
+ const grok = findModel(catalog, "grok-4.6");
547
+ if (!grok) missing.push("grok-4.6: absent from the live catalog");
548
+ else {
549
+ if (!hasToolCalls(grok)) missing.push("grok-4.6: does not advertise tool_calls");
550
+ if (!supportsEffort(grok, "medium")) missing.push("grok-4.6: does not advertise a \"medium\" reasoning effort");
551
+ if (!hasUsablePromptMetadata(grok)) missing.push("grok-4.6: no usable max_prompt_tokens metadata");
552
+ if (!supportsEndpoint(grok, RESPONSES_ENDPOINTS)) missing.push("grok-4.6: does not advertise a supported Responses endpoint");
553
+ }
554
+ const gemini = findModel(catalog, "gemini-3.7-flash");
555
+ if (!gemini) missing.push("gemini-3.7-flash: absent from the live catalog");
556
+ else {
557
+ if (!hasContextAtLeast(gemini, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gemini-3.7-flash: advertised context window is below 1M");
558
+ if (!supportsEffort(gemini, "high")) missing.push("gemini-3.7-flash: does not advertise a \"high\" reasoning effort");
559
+ if (pickEndpoint(gemini) !== "chat") missing.push("gemini-3.7-flash: does not advertise a supported chat-completions endpoint");
560
+ }
561
+ const opus = findModel(catalog, "claude-opus-5");
562
+ if (!opus) missing.push("claude-opus-5: absent from the live catalog");
563
+ else {
564
+ if (!hasContextAtLeast(opus, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("claude-opus-5: advertised context window is below 1M");
565
+ if (!supportsEffort(opus, "high")) missing.push("claude-opus-5: does not advertise a \"high\" reasoning effort");
566
+ if (opus.capabilities?.supports?.adaptive_thinking !== true) missing.push("claude-opus-5: does not advertise adaptive_thinking");
567
+ if (!hasUsablePromptMetadata(opus)) missing.push("claude-opus-5: no usable max_prompt_tokens metadata");
568
+ if (!supportsEndpoint(opus, MESSAGES_ENDPOINTS)) missing.push("claude-opus-5: does not advertise a supported Messages endpoint");
569
+ }
570
+ return {
571
+ ok: missing.length === 0,
572
+ missing
573
+ };
574
+ }
575
+ /**
576
+ * Format `validateFastProfilePrerequisites`'s failure list into the launch
577
+ * error message: every missing/invalid model, plus the rollback command.
578
+ */
579
+ function formatFastPrerequisiteFailure(missing) {
580
+ return "github-router claude -m fast requires the following live-catalog capabilities, which this account's catalog does not fully provide:\n" + missing.map((m) => ` - ${m}`).join("\n") + "\n\nFalling back or silently dropping an agent is not supported for the fast profile's exact roster. Run plain `github-router claude` instead.";
581
+ }
582
+ //#endregion
357
583
  //#region src/lib/file-log-reporter.ts
358
584
  const MAX_LOG_BYTES = 1048576;
359
585
  const DEDUP_MAX = 1e3;
@@ -1430,7 +1656,7 @@ function collectToolFieldKeys(body) {
1430
1656
  //#endregion
1431
1657
  //#region package.json
1432
1658
  var name = "github-router";
1433
- var version = "0.3.289";
1659
+ var version = "0.3.293";
1434
1660
  //#endregion
1435
1661
  //#region src/lib/approval.ts
1436
1662
  const awaitApproval = async () => {
@@ -3063,6 +3289,18 @@ function anthropicMessageToNeutral(msg) {
3063
3289
  name: typeof b.name === "string" ? b.name : "",
3064
3290
  arguments: b.input ?? {}
3065
3291
  });
3292
+ else if (b.type === "server_tool_use" && b.name === "advisor") parts.push({
3293
+ type: "text",
3294
+ text: "[Consulted advisor]"
3295
+ });
3296
+ else if (b.type === "advisor_tool_result") {
3297
+ const resultContent = b.content;
3298
+ const text = resultContent && typeof resultContent === "object" && typeof resultContent.text === "string" ? resultContent.text : "";
3299
+ if (text.length > 0) parts.push({
3300
+ type: "text",
3301
+ text: `[Advisor response]\n${text}`
3302
+ });
3303
+ }
3066
3304
  }
3067
3305
  return [{
3068
3306
  role: "assistant",
@@ -4042,6 +4280,75 @@ function isAsyncIterable(x) {
4042
4280
  return x != null && typeof x[Symbol.asyncIterator] === "function";
4043
4281
  }
4044
4282
  /**
4283
+ * Context-free, caller-abortable core of the non-Claude shim's STREAMING
4284
+ * path for one already-parsed Anthropic request.
4285
+ *
4286
+ * "Context-free": no Hono `Context` dependency, unlike
4287
+ * `handleNonClaudeResponses`/`handleNonClaudeChat` below (which need one to
4288
+ * build their JSON error responses and read `c.req.path` for logging).
4289
+ * "Caller-abortable": takes the caller's OWN `AbortSignal` rather than
4290
+ * constructing an internal `AbortController` — this function never creates
4291
+ * one — and forwards an optional `onCancel` so a caller that DOES own a
4292
+ * controller (the two handlers below, for the initial request) can still
4293
+ * tear it down when the stream it returns is cancelled.
4294
+ *
4295
+ * Shared by:
4296
+ * - `handleNonClaudeResponses` / `handleNonClaudeChat` (this module), for
4297
+ * the initial request — each already knows its own endpoint statically
4298
+ * (they are dispatched by `classifyMessagesRoute`), so this function
4299
+ * takes `endpoint` as an explicit argument rather than re-deriving it
4300
+ * from the catalog (which would also mean re-parsing/re-picking work the
4301
+ * caller already did).
4302
+ * - `makeShimContinueTurn` (below), which `buildAdvisorStream`
4303
+ * (`src/services/advisor/advisor.ts`) injects as its `continueTurn` for
4304
+ * the fast Luna-lead profile, so an advisor continuation on a non-Claude
4305
+ * lead runs through the SAME translation + SSE-synthesis machinery as
4306
+ * the initial turn instead of a parallel, divergent implementation.
4307
+ */
4308
+ async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
4309
+ const routePath = opts.routePath ?? "/v1/messages (advisor lead shim)";
4310
+ if (endpoint === "chat") {
4311
+ const payload = parsedToChatPayload(parsed);
4312
+ const result = await createChatCompletions(payload, opts.model?.requestHeaders, signal, true);
4313
+ if (!isAsyncIterable(result)) throw new Error("Upstream /chat/completions did not return an SSE stream (stream: true expected)");
4314
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromChat(result, { modelId: opts.modelId }), {
4315
+ routePath,
4316
+ onCancel: opts.onCancel,
4317
+ inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4318
+ });
4319
+ return new Response(stream, {
4320
+ status: 200,
4321
+ headers: STREAM_HEADERS
4322
+ });
4323
+ }
4324
+ const payload = parsedToResponsesPayload(parsed);
4325
+ const result = await createResponses(payload, opts.model?.requestHeaders, signal, true);
4326
+ if (!isAsyncIterable(result)) throw new Error("Upstream /responses did not return an SSE stream (stream: true expected)");
4327
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromResponses(result, { modelId: opts.modelId }), {
4328
+ routePath,
4329
+ onCancel: opts.onCancel,
4330
+ inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4331
+ });
4332
+ return new Response(stream, {
4333
+ status: 200,
4334
+ headers: STREAM_HEADERS
4335
+ });
4336
+ }
4337
+ /**
4338
+ * Build an injectable `continueTurn(body, signal)` for `buildAdvisorStream`
4339
+ * (`src/services/advisor/advisor.ts`) that routes a continuation turn
4340
+ * through THIS module's non-Claude shim instead of Claude passthrough — used
4341
+ * for the fast Luna-lead profile's advisor translate-loop. No `onCancel` is
4342
+ * threaded through: the advisor loop's own `aborter` (shared with `signal`
4343
+ * here) already tears down on consumer cancel via `buildAdvisorStream`'s
4344
+ * `cancel()`, so this stream needs no independent teardown hook.
4345
+ */
4346
+ function makeShimContinueTurn(endpoint, opts) {
4347
+ return (body, signal) => {
4348
+ return streamParsedRequestViaShim(parseAnthropicRequest(body, opts.modelId, opts.model), endpoint, opts, signal);
4349
+ };
4350
+ }
4351
+ /**
4045
4352
  * Handle a `/v1/messages` request targeting a non-Claude `/responses` model.
4046
4353
  * Returns a streaming or non-streaming Anthropic-format Response. Upstream
4047
4354
  * non-2xx / abort errors are thrown (as HTTPError) and handled by the route's
@@ -4062,12 +4369,15 @@ async function handleNonClaudeResponses(c, opts) {
4062
4369
  }, 400);
4063
4370
  }
4064
4371
  const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
4065
- const payload = parsedToResponsesPayload(parsed);
4066
4372
  if (consola.level >= 4) consola.debug(`Anthropic-translate → /responses model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
4067
4373
  if (parsed.stream) {
4068
4374
  const aborter = new AbortController();
4069
- const result = await createResponses(payload, opts.model?.requestHeaders, aborter.signal, true);
4070
- if (!isAsyncIterable(result)) throw new Error("Upstream /responses did not return an SSE stream (stream: true expected)");
4375
+ const stream = await streamParsedRequestViaShim(parsed, "responses", {
4376
+ modelId: opts.modelId,
4377
+ model: opts.model,
4378
+ routePath,
4379
+ onCancel: () => aborter.abort()
4380
+ }, aborter.signal);
4071
4381
  logRequest({
4072
4382
  method: "POST",
4073
4383
  path: routePath,
@@ -4076,16 +4386,9 @@ async function handleNonClaudeResponses(c, opts) {
4076
4386
  status: 200,
4077
4387
  streaming: true
4078
4388
  }, opts.model, opts.startTime);
4079
- const stream = anthropicSseStreamFromEvents(synthAnthropicFromResponses(result, { modelId: opts.modelId }), {
4080
- routePath,
4081
- onCancel: () => aborter.abort(),
4082
- inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4083
- });
4084
- return new Response(stream, {
4085
- status: 200,
4086
- headers: STREAM_HEADERS
4087
- });
4389
+ return stream;
4088
4390
  }
4391
+ const payload = parsedToResponsesPayload(parsed);
4089
4392
  const anthropic = responsesResponseToAnthropicMessage(await createResponses(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
4090
4393
  const usage = anthropic.usage;
4091
4394
  logRequest({
@@ -4122,12 +4425,15 @@ async function handleNonClaudeChat(c, opts) {
4122
4425
  }, 400);
4123
4426
  }
4124
4427
  const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
4125
- const payload = parsedToChatPayload(parsed);
4126
4428
  if (consola.level >= 4) consola.debug(`Anthropic-translate → /chat/completions model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
4127
4429
  if (parsed.stream) {
4128
4430
  const aborter = new AbortController();
4129
- const result = await createChatCompletions(payload, opts.model?.requestHeaders, aborter.signal, true);
4130
- if (!isAsyncIterable(result)) throw new Error("Upstream /chat/completions did not return an SSE stream (stream: true expected)");
4431
+ const stream = await streamParsedRequestViaShim(parsed, "chat", {
4432
+ modelId: opts.modelId,
4433
+ model: opts.model,
4434
+ routePath,
4435
+ onCancel: () => aborter.abort()
4436
+ }, aborter.signal);
4131
4437
  logRequest({
4132
4438
  method: "POST",
4133
4439
  path: routePath,
@@ -4136,16 +4442,9 @@ async function handleNonClaudeChat(c, opts) {
4136
4442
  status: 200,
4137
4443
  streaming: true
4138
4444
  }, opts.model, opts.startTime);
4139
- const stream = anthropicSseStreamFromEvents(synthAnthropicFromChat(result, { modelId: opts.modelId }), {
4140
- routePath,
4141
- onCancel: () => aborter.abort(),
4142
- inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4143
- });
4144
- return new Response(stream, {
4145
- status: 200,
4146
- headers: STREAM_HEADERS
4147
- });
4445
+ return stream;
4148
4446
  }
4447
+ const payload = parsedToChatPayload(parsed);
4149
4448
  const anthropic = chatResponseToAnthropicMessage(await createChatCompletions(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
4150
4449
  const usage = anthropic.usage;
4151
4450
  logRequest({
@@ -4162,6 +4461,126 @@ async function handleNonClaudeChat(c, opts) {
4162
4461
  return c.json(anthropic, 200);
4163
4462
  }
4164
4463
  //#endregion
4464
+ //#region src/lib/messages-identity-preflight.ts
4465
+ /**
4466
+ * `/v1/messages` identity-preflight bearer, distinct from the `/mcp` nonce
4467
+ * (`Authorization` header). Delivered to the spawned Claude Code process via
4468
+ * `ANTHROPIC_CUSTOM_HEADERS` (an Anthropic SDK env var already carried
4469
+ * through `getClaudeCodeEnvVars`), so it rides on EVERY `/v1/messages`
4470
+ * request the client sends — main-loop turns, subagents, hooks calling the
4471
+ * loopback endpoint directly, all of it.
4472
+ *
4473
+ * A raw BYO client (`start`/`codex`, or any script hitting `/v1/messages`
4474
+ * directly) never sets this header, and that is intentional: header
4475
+ * PRESENCE is what marks a request as asserting a bound-launch identity.
4476
+ * Absence is not a downgrade from some previously-enforced state — it is
4477
+ * today's status quo for every `/v1/messages` caller, preserved exactly.
4478
+ * Only a request that DOES present the header is held to it: if a matching
4479
+ * registry entry can't be found for it (wrong value, launch already torn
4480
+ * down, a claude session that raced this header against a proxy restart),
4481
+ * that specific request fails closed.
4482
+ */
4483
+ const LAUNCH_SECRET_HEADER = "X-GH-Router-Launch-Secret";
4484
+ /**
4485
+ * Validate the launch-secret header BEFORE any body consumer runs (i.e.
4486
+ * before `c.req.text()`/`c.req.json()` — this function only reads a
4487
+ * header). Callers running this must NOT surface a bare 401 on the
4488
+ * `/v1/messages` boundary: this route observes the same no-401 invariant
4489
+ * `forwardError` enforces for upstream failures (Claude Code's reactive
4490
+ * refresh path fires on ANY 401 and would try to use the synthetic
4491
+ * refresh token, breaking the session). Use `identityPreflightErrorResponse`
4492
+ * below, which answers 403, to reject a failed preflight.
4493
+ */
4494
+ function runMessagesIdentityPreflight(c) {
4495
+ const header = c.req.header(LAUNCH_SECRET_HEADER);
4496
+ if (!header) return { ok: true };
4497
+ const launch = findLaunchBySecret(header);
4498
+ if (!launch) return {
4499
+ ok: false,
4500
+ reason: "X-GH-Router-Launch-Secret header did not match any registered launch (the launch may have been restarted, or the header was tampered with)"
4501
+ };
4502
+ return {
4503
+ ok: true,
4504
+ launch
4505
+ };
4506
+ }
4507
+ /**
4508
+ * Anthropic-shaped rejection for a failed identity preflight. 403, never
4509
+ * 401 — see the no-401 invariant note on `runMessagesIdentityPreflight`.
4510
+ */
4511
+ function identityPreflightErrorResponse(c, reason, path = "/v1/messages") {
4512
+ return c.json({
4513
+ type: "error",
4514
+ error: {
4515
+ type: "permission_error",
4516
+ message: `${path} identity preflight rejected: ${reason}`
4517
+ }
4518
+ }, 403);
4519
+ }
4520
+ //#endregion
4521
+ //#region src/lib/fast-request-preprocess.ts
4522
+ /**
4523
+ * Apply authenticated fast-profile model and effort policy before ordinary model
4524
+ * resolution. Synthetic aliases are refused outside an authenticated fast
4525
+ * launch, so raw/BYO traffic cannot opt itself into private profile semantics.
4526
+ */
4527
+ function preprocessFastRequest(rawBody, launch) {
4528
+ let parsed;
4529
+ try {
4530
+ parsed = JSON.parse(rawBody);
4531
+ } catch {
4532
+ return {
4533
+ body: rawBody,
4534
+ modified: false
4535
+ };
4536
+ }
4537
+ const originalModel = typeof parsed.model === "string" ? parsed.model : void 0;
4538
+ if (!originalModel) return {
4539
+ body: rawBody,
4540
+ modified: false
4541
+ };
4542
+ const alias = resolveModelAlias(originalModel);
4543
+ if (alias && launch?.profileId !== "fast") return {
4544
+ body: rawBody,
4545
+ originalModel,
4546
+ modified: false,
4547
+ rejectedAlias: originalModel
4548
+ };
4549
+ if (launch?.profileId !== "fast") return {
4550
+ body: rawBody,
4551
+ originalModel,
4552
+ modified: false
4553
+ };
4554
+ const bare = originalModel.replace(/\[1m\]$/i, "");
4555
+ let effort;
4556
+ if (alias) {
4557
+ effort = alias.absentEffortDefault;
4558
+ parsed.model = canonicalizeAliasModel(originalModel);
4559
+ } else if (bare === "gpt-5.6-luna") effort = "max";
4560
+ else if (bare === "gpt-5.6-sol") effort = "high";
4561
+ else if (bare === "grok-4.6") effort = "medium";
4562
+ else if (bare === "gemini-3.7-flash") effort = "high";
4563
+ else if (bare === "claude-opus-5") effort = "high";
4564
+ if (!effort && !alias) return {
4565
+ body: rawBody,
4566
+ originalModel,
4567
+ modified: false,
4568
+ rejectedModel: originalModel
4569
+ };
4570
+ const outputConfig = parsed.output_config && typeof parsed.output_config === "object" ? parsed.output_config : {};
4571
+ parsed.output_config = {
4572
+ ...outputConfig,
4573
+ effort
4574
+ };
4575
+ const thinking = parsed.thinking;
4576
+ if (thinking && typeof thinking === "object" && thinking.type === "enabled") parsed.thinking = { type: "adaptive" };
4577
+ return {
4578
+ body: JSON.stringify(parsed),
4579
+ originalModel,
4580
+ modified: true
4581
+ };
4582
+ }
4583
+ //#endregion
4165
4584
  //#region src/routes/messages/handler.ts
4166
4585
  const MAX_THINKING_REPAIR_ATTEMPTS = 5;
4167
4586
  const isWebSearchTool$1 = (tool) => !!tool && typeof tool === "object" && (typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search");
@@ -4222,13 +4641,13 @@ function stripWebSearchTool(body) {
4222
4641
  }
4223
4642
  /**
4224
4643
  * Strip the injected `__anthropic_advisor` tool (and any Anthropic-native
4225
- * `advisor_*` typed tool) from a request body. Used ONLY on the non-Claude
4226
- * shim path: ADVISOR's server-side translate-loop (buildAdvisorStream) lives
4227
- * on the native /v1/messages route, so a non-Claude model has no handler for
4228
- * the tool and it must be removed before forwarding — otherwise the model
4229
- * could emit a tool_use that nothing fulfils. Mirrors stripWebSearchTool's
4230
- * tool_choice cleanup. Returns the original string (same reference) when
4231
- * nothing was removed.
4644
+ * `advisor_*` typed tool) from a request body. Used on non-Claude shim paths
4645
+ * without an advisor handler and on authenticated fast Task-subagent requests,
4646
+ * where Advisor is intentionally lead-only. Mirrors stripWebSearchTool's
4647
+ * tool_choice cleanup. Returns the original string when nothing was removed.
4648
+ * End-to-end evidence for both stripped tool forms and the resulting 200 lives
4649
+ * in probes `shim_advisor_degrade_gpt55` and
4650
+ * `shim_advisor_degrade_gemini35flash`.
4232
4651
  */
4233
4652
  function stripAdvisorTool(rawBody) {
4234
4653
  let body;
@@ -4285,6 +4704,8 @@ async function processWebSearch(rawBody) {
4285
4704
  }
4286
4705
  async function handleCompletion(c) {
4287
4706
  const startTime = Date.now();
4707
+ const identity = runMessagesIdentityPreflight(c);
4708
+ if (!identity.ok) return identityPreflightErrorResponse(c, identity.reason);
4288
4709
  await checkRateLimit(state);
4289
4710
  const rawBody = await c.req.text();
4290
4711
  recordBodySize(rawBody.length);
@@ -4359,9 +4780,25 @@ async function handleCompletion(c) {
4359
4780
  if (state.manualApprove) await awaitApproval();
4360
4781
  const betaHeaders = extractBetaHeaders(c);
4361
4782
  const incomingBeta = c.req.header("anthropic-beta");
4362
- const advisorEnabled = isAdvisorRequested(incomingBeta);
4363
- let finalBody = await processWebSearch(rawBody);
4783
+ const advisorRequested = isAdvisorRequested(incomingBeta);
4784
+ const fastProfileRequest = identity.launch?.profileId === "fast";
4785
+ const fastSubagentRequest = fastProfileRequest && Boolean(c.req.header("x-claude-code-agent-id"));
4786
+ const fastLeadAdvisor = fastProfileRequest && !fastSubagentRequest;
4787
+ const advisorEnabled = advisorRequested && !fastSubagentRequest;
4788
+ const fastPreprocess = preprocessFastRequest(rawBody, identity.launch);
4789
+ if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
4790
+ const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
4791
+ return c.json({
4792
+ type: "error",
4793
+ error: {
4794
+ type: "invalid_request_error",
4795
+ message
4796
+ }
4797
+ }, 400);
4798
+ }
4799
+ let finalBody = await processWebSearch(fastPreprocess.body);
4364
4800
  finalBody = sanitizeAnthropicBody(finalBody);
4801
+ if (fastSubagentRequest) finalBody = stripAdvisorTool(finalBody);
4365
4802
  const loopGuard = guardAnthropicBody(finalBody);
4366
4803
  if (loopGuard.action === "abort") return c.json({
4367
4804
  type: "error",
@@ -4372,7 +4809,7 @@ async function handleCompletion(c) {
4372
4809
  }, 400, { "x-should-retry": "false" });
4373
4810
  if (loopGuard.body !== void 0) finalBody = loopGuard.body;
4374
4811
  if (advisorEnabled) {
4375
- finalBody = injectAdvisorTool(finalBody);
4812
+ finalBody = injectAdvisorTool(finalBody, fastLeadAdvisor ? FAST_ADVISOR_TOOL_INSTRUCTIONS : void 0);
4376
4813
  consola.info("ADVISOR enabled for this request — injecting __anthropic_advisor tool; will translate tool_use → server_tool_use{advisor} on the SSE stream");
4377
4814
  }
4378
4815
  if (finalBody.includes("\"mcp_servers\"")) try {
@@ -4389,6 +4826,55 @@ async function handleCompletion(c) {
4389
4826
  const modelId = resolvedModel ?? originalModel;
4390
4827
  const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel);
4391
4828
  if (messagesRoute !== "claude-passthrough") {
4829
+ const endpoint = messagesRoute === "chat-shim" ? "chat" : "responses";
4830
+ let parsedBase;
4831
+ try {
4832
+ parsedBase = JSON.parse(resolvedBody);
4833
+ } catch {}
4834
+ const wantsStream = parsedBase?.stream === true;
4835
+ if (advisorEnabled && wantsStream && identity.launch?.profileId === "fast" && isFastProfileLead(modelId)) {
4836
+ const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
4837
+ const parsedInitial = parseAnthropicRequest(parsedBase, modelId, selectedModel);
4838
+ const fastAdvisorAborter = new AbortController();
4839
+ const firstResponse = await streamParsedRequestViaShim(parsedInitial, endpoint, {
4840
+ modelId,
4841
+ model: selectedModel,
4842
+ routePath: c.req.path,
4843
+ onCancel: () => fastAdvisorAborter.abort()
4844
+ }, fastAdvisorAborter.signal);
4845
+ logRequest({
4846
+ method: "POST",
4847
+ path: c.req.path,
4848
+ model: originalModel,
4849
+ resolvedModel: modelId,
4850
+ status: 200,
4851
+ streaming: true
4852
+ }, selectedModel, startTime);
4853
+ const advisorChoice = resolveAdvisorModel(modelId, true);
4854
+ return new Response(buildAdvisorStream({
4855
+ firstResponse,
4856
+ initialConversation,
4857
+ baseBody: parsedBase,
4858
+ requestHeaders: {},
4859
+ advisorModel: advisorChoice.model,
4860
+ advisorEscalated: advisorChoice.escalated,
4861
+ advisorFastProfile: fastLeadAdvisor,
4862
+ advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model, true),
4863
+ externalAborter: fastAdvisorAborter,
4864
+ continueTurn: makeShimContinueTurn(endpoint, {
4865
+ modelId,
4866
+ model: selectedModel
4867
+ })
4868
+ }), {
4869
+ status: 200,
4870
+ headers: {
4871
+ "content-type": "text/event-stream",
4872
+ "cache-control": "no-cache",
4873
+ "transfer-encoding": "chunked",
4874
+ connection: "keep-alive"
4875
+ }
4876
+ });
4877
+ }
4392
4878
  const shimBody = stripAdvisorTool(resolvedBody);
4393
4879
  if (advisorEnabled) consola.info("ADVISOR requested with a non-Claude model — stripping the injected __anthropic_advisor tool and proceeding without advisor (Claude-only feature; gracefully degraded).");
4394
4880
  const shimOpts = {
@@ -4482,7 +4968,7 @@ async function handleCompletion(c) {
4482
4968
  parsedBase = JSON.parse(nativeBody);
4483
4969
  } catch {}
4484
4970
  const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
4485
- const advisorChoice = resolveAdvisorModel(originalModel);
4971
+ const advisorChoice = resolveAdvisorModel(originalModel, fastLeadAdvisor);
4486
4972
  return new Response(buildAdvisorStream({
4487
4973
  firstResponse: response,
4488
4974
  initialConversation,
@@ -4490,7 +4976,8 @@ async function handleCompletion(c) {
4490
4976
  requestHeaders,
4491
4977
  advisorModel: advisorChoice.model,
4492
4978
  advisorEscalated: advisorChoice.escalated,
4493
- advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model),
4979
+ advisorFastProfile: fastLeadAdvisor,
4980
+ advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model, fastLeadAdvisor),
4494
4981
  externalAborter: advisorAborter
4495
4982
  }), {
4496
4983
  status: response.status,
@@ -4541,9 +5028,10 @@ function resolveModelInBody$1(rawBody) {
4541
5028
  }
4542
5029
  const originalModel = typeof parsed.model === "string" ? parsed.model : void 0;
4543
5030
  let modified = false;
4544
- if (originalModel) {
4545
- const resolved = resolveModel(originalModel);
4546
- if (resolved !== originalModel) {
5031
+ const resolvedOriginalModel = typeof parsed.model === "string" ? parsed.model : originalModel;
5032
+ if (resolvedOriginalModel) {
5033
+ const resolved = resolveModel(resolvedOriginalModel);
5034
+ if (resolved !== resolvedOriginalModel) {
4547
5035
  parsed.model = resolved;
4548
5036
  modified = true;
4549
5037
  }
@@ -4823,7 +5311,20 @@ function stripWebSearchFromBody(rawBody) {
4823
5311
  */
4824
5312
  async function handleCountTokens(c) {
4825
5313
  const startTime = Date.now();
4826
- const strippedBody = stripWebSearchFromBody(sanitizeAnthropicBody(await c.req.text()));
5314
+ const identity = runMessagesIdentityPreflight(c);
5315
+ if (!identity.ok) return identityPreflightErrorResponse(c, identity.reason, c.req.path);
5316
+ const fastPreprocess = preprocessFastRequest(await c.req.text(), identity.launch);
5317
+ if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
5318
+ const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
5319
+ return c.json({
5320
+ type: "error",
5321
+ error: {
5322
+ type: "invalid_request_error",
5323
+ message
5324
+ }
5325
+ }, 400);
5326
+ }
5327
+ const strippedBody = stripWebSearchFromBody(sanitizeAnthropicBody(fastPreprocess.body));
4827
5328
  if (strippedBody.includes("\"mcp_servers\"")) try {
4828
5329
  const probe = JSON.parse(strippedBody);
4829
5330
  if (Array.isArray(probe.mcp_servers) && probe.mcp_servers.length > 0) return c.json({
@@ -5798,12 +6299,18 @@ function parseSharedArgs(args) {
5798
6299
  }
5799
6300
  /**
5800
6301
  * Non-Claude models we surface as first-class, selectable rows in Claude
5801
- * Code's model picker (Phase 3 of native-non-claude-models). The main
5802
- * agent loop runs on them through the `/v1/messages` translation shim
5803
- * (`src/lib/anthropic-translate/*`, branched in `routes/messages/handler.ts`)
5804
- * that forwards non-Claude targets to Copilot `/responses` (gpt) or
5805
- * `/chat/completions` (gemini). The Gemini review row prefers
5806
- * `gemini-3.1-pro-preview` and degrades to `gemini-3.7-flash`.
6302
+ * Code's model picker (Phase 3 of native-non-claude-models, later
6303
+ * modernized to exactly these four live models by the fast-launch-profile
6304
+ * change). The main agent loop runs on them through the `/v1/messages`
6305
+ * translation shim (`src/lib/anthropic-translate/*`, branched in
6306
+ * `routes/messages/handler.ts`) that forwards non-Claude targets to
6307
+ * Copilot `/responses` (gpt) or `/chat/completions` (gemini/grok).
6308
+ *
6309
+ * This list is EXACT and STATIC — no dynamic Gemini-review append (the
6310
+ * earlier `gemini-3.1-pro-preview`-preferred / `gemini-3.7-flash`-fallback
6311
+ * row is retired: `gemini-3.7-flash` is now a first-class row on its own,
6312
+ * always at this fixed id). A model missing from the catalog is simply
6313
+ * omitted, never substituted — see `nativeSelectableModelsInCatalog`.
5807
6314
  *
5808
6315
  * Display labels only: the gateway-model cache schema Claude Code reads is
5809
6316
  * `{id, display_name?}` per model — there is NO per-model context-window
@@ -5817,48 +6324,52 @@ const NATIVE_NON_CLAUDE_MODELS = [
5817
6324
  displayName: "GPT-5.6 Sol"
5818
6325
  },
5819
6326
  {
5820
- id: "gpt-5.5",
5821
- displayName: "GPT-5.5"
6327
+ id: "gpt-5.6-luna",
6328
+ displayName: "GPT-5.6 Luna"
5822
6329
  },
5823
6330
  {
5824
- id: "gpt-5.3-codex",
5825
- displayName: "GPT-5.3 Codex"
6331
+ id: "gemini-3.7-flash",
6332
+ displayName: "Gemini 3.7 Flash"
5826
6333
  },
5827
6334
  {
5828
- id: "gemini-3.5-flash",
5829
- displayName: "Gemini 3.5 Flash"
6335
+ id: "grok-4.6",
6336
+ displayName: "Grok 4.6"
5830
6337
  }
5831
6338
  ];
5832
6339
  /**
6340
+ * `grok-4.6` never carries `[1m]` — its live-catalog window is 500K total
6341
+ * (372K max prompt), genuinely below the 1M accounting threshold, and this
6342
+ * project deliberately does NOT inject a global
6343
+ * `CLAUDE_CODE_MAX_CONTEXT_TOKENS` override for it (see
6344
+ * `docs/default-models.md` "fast launch profile" once landed): Claude Code
6345
+ * permits arbitrary bare non-Claude ids and runtime `/model` switches, so a
6346
+ * Grok-specific process-global window override would incorrectly follow the
6347
+ * session onto every other model after a switch. Grok is simply left bare,
6348
+ * which is also its true accounting rather than an over- or under-estimate
6349
+ * disguised as one.
6350
+ */
6351
+ const NEVER_1M_MODEL_IDS = /* @__PURE__ */ new Set(["grok-4.6"]);
6352
+ /**
5833
6353
  * The subset of `NATIVE_NON_CLAUDE_MODELS` actually present in the live
5834
- * Copilot catalog. License tiers differ (gpt-5.5 needs
5835
- * pro_plus/business/enterprise/max; gemini-3.5-flash is absent on
5836
- * edu/individual_trial), so a model missing from the catalog is silently
5837
- * dropped the caller then neither enables discovery nor writes a cache
5838
- * for it, and lesser tiers see the unchanged picker. Pure (reads
5839
- * `state.models`), so it is unit-testable without side effects.
6354
+ * Copilot catalog. License tiers differ, so a model missing from the
6355
+ * catalog is silently dropped — the caller then neither enables discovery
6356
+ * nor writes a cache for it, and lesser tiers see the unchanged picker.
6357
+ * Pure (reads `state.models`), so it is unit-testable without side effects.
5840
6358
  *
5841
6359
  * The projected id carries a `[1m]` suffix when the catalog advertises a
5842
- * >=1M window for it, because Claude Code budgets a gateway-discovered row
5843
- * at its 200K default otherwise `gpt-5.6-sol` (1,050,000) would compact at
5844
- * roughly a fifth of its real window. `withOneMSuffix` is catalog-gated, so
5845
- * `gpt-5.3-codex` (400K) stays bare and keeps the conservative accounting;
5846
- * over-budgeting it would trade premature compaction for a hard overflow.
5847
- * The lookup below still keys off the BARE id the decoration is applied
5848
- * only to the value handed to Claude Code.
6360
+ * >=1M window for it AND the id isn't in `NEVER_1M_MODEL_IDS`, because
6361
+ * Claude Code budgets a gateway-discovered row at its 200K default
6362
+ * otherwise. `withOneMSuffix` is catalog-gated on top of that, so a model
6363
+ * whose advertised window shrinks below 1M stays bare regardless. The
6364
+ * lookup below still keys off the BARE id the decoration is applied only
6365
+ * to the value handed to Claude Code.
5849
6366
  */
5850
6367
  function nativeSelectableModelsInCatalog() {
5851
6368
  const catalog = state.models?.data;
5852
6369
  if (!catalog || catalog.length === 0) return [];
5853
6370
  const present = new Set(catalog.map((m) => m.id));
5854
- const models = NATIVE_NON_CLAUDE_MODELS.filter((m) => present.has(m.id));
5855
- const geminiReviewModel = resolveGeminiReviewModel();
5856
- if (geminiReviewModel) models.push({
5857
- id: geminiReviewModel,
5858
- displayName: geminiReviewModel === "gemini-3.7-flash" ? "Gemini 3.7 Flash" : "Gemini 3.1 Pro (preview)"
5859
- });
5860
- return models.map((m) => ({
5861
- id: withOneMSuffix(m.id),
6371
+ return NATIVE_NON_CLAUDE_MODELS.filter((m) => present.has(m.id)).map((m) => ({
6372
+ id: NEVER_1M_MODEL_IDS.has(m.id) ? m.id : withOneMSuffix(m.id),
5862
6373
  display_name: m.displayName
5863
6374
  }));
5864
6375
  }
@@ -5979,7 +6490,19 @@ function clearGatewayModelCache(configDir = PATHS.CLAUDE_CONFIG_DIR) {
5979
6490
  * MUST NOT return 401 on the Anthropic-shape boundary even when
5980
6491
  * upstream Copilot returns 401. See `src/routes/messages/handler.ts`.
5981
6492
  */
5982
- function getClaudeCodeEnvVars(serverUrl, model) {
6493
+ /**
6494
+ * Decorate a Luna-alias id with `[1m]` based on the REAL `gpt-5.6-luna`
6495
+ * catalog entry's advertised window, never on the alias string itself
6496
+ * (which is never a catalog entry — `catalogAdvertises1M`/`resolveModel`
6497
+ * would find nothing and silently leave it bare). Shared by every
6498
+ * fast-profile tier-row seed below so they can't disagree about whether
6499
+ * Luna currently backs 1M.
6500
+ */
6501
+ function oneMSuffixForAlias(aliasId) {
6502
+ if (oneMContextDisabled()) return aliasId;
6503
+ return catalogAdvertises1M("gpt-5.6-luna") ? `${aliasId}[1m]` : aliasId;
6504
+ }
6505
+ function getClaudeCodeEnvVars(serverUrl, model, launchProfileId = "standard") {
5983
6506
  const vars = {
5984
6507
  ANTHROPIC_BASE_URL: serverUrl,
5985
6508
  CLAUDE_CONFIG_DIR: PATHS.CLAUDE_CONFIG_DIR,
@@ -5991,15 +6514,34 @@ function getClaudeCodeEnvVars(serverUrl, model) {
5991
6514
  const mcpToolTimeoutMs = String(resolveMcpToolTimeoutMs());
5992
6515
  if (process.env.MCP_TIMEOUT === void 0) vars.MCP_TIMEOUT = mcpToolTimeoutMs;
5993
6516
  if (process.env.MCP_TOOL_TIMEOUT === void 0) vars.MCP_TOOL_TIMEOUT = mcpToolTimeoutMs;
5994
- const smallFastModel = isBudgetClaudeLead(model) && (state.models?.data?.some((m) => m.id === "claude-haiku-4.5") ?? false) ? BUDGET_SMALL_FAST_SLUG : "claude-sonnet-5";
5995
- if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = smallFastModel;
6517
+ const isFastProfile = launchProfileId === "fast";
6518
+ const smallFastModel = isFastProfile ? LUNA_HAIKU_ALIAS_ID : isBudgetClaudeLead(model) && (state.models?.data?.some((m) => m.id === "claude-haiku-4.5") ?? false) ? BUDGET_SMALL_FAST_SLUG : "claude-sonnet-5";
6519
+ if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = isFastProfile ? oneMSuffixForAlias(smallFastModel) : smallFastModel;
5996
6520
  const seedTierRow = (modelKey, nameKey, bareSlug) => {
5997
6521
  if (process.env[modelKey] !== void 0) return;
5998
6522
  vars[modelKey] = withOneMSuffixForLead(bareSlug);
5999
6523
  if (process.env[nameKey] === void 0) vars[nameKey] = bareSlug;
6000
6524
  };
6001
- seedTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", "claude-sonnet-5");
6002
- seedTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", smallFastModel);
6525
+ const seedFastAliasTierRow = (modelKey, nameKey, aliasId, displayName) => {
6526
+ if (process.env[modelKey] !== void 0) return;
6527
+ vars[modelKey] = oneMSuffixForAlias(aliasId);
6528
+ if (process.env[nameKey] === void 0) vars[nameKey] = displayName;
6529
+ };
6530
+ if (isFastProfile) {
6531
+ seedFastAliasTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", LUNA_SONNET_ALIAS_ID, "GPT-5.6 Luna (xhigh)");
6532
+ seedFastAliasTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", LUNA_HAIKU_ALIAS_ID, "GPT-5.6 Luna (high)");
6533
+ const fastAliasCapabilities = "effort,xhigh_effort,max_effort,thinking,adaptive_thinking,interleaved_thinking";
6534
+ if (process.env.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6535
+ if (process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_DEFAULT_HAIKU_MODEL_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6536
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION === void 0) {
6537
+ vars.ANTHROPIC_CUSTOM_MODEL_OPTION = oneMSuffixForAlias(LUNA_DRIVER_ALIAS_ID);
6538
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION_NAME === void 0) vars.ANTHROPIC_CUSTOM_MODEL_OPTION_NAME = "GPT-5.6 Luna (max)";
6539
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_CUSTOM_MODEL_OPTION_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6540
+ }
6541
+ } else {
6542
+ seedTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", "claude-sonnet-5");
6543
+ seedTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", smallFastModel);
6544
+ }
6003
6545
  seedTierRow("ANTHROPIC_DEFAULT_OPUS_MODEL", "ANTHROPIC_DEFAULT_OPUS_MODEL_NAME", "claude-opus-5");
6004
6546
  if (process.env.CLAUDE_CODE_PLAN_V2_AGENT_COUNT === void 0) vars.CLAUDE_CODE_PLAN_V2_AGENT_COUNT = "7";
6005
6547
  for (const key of [
@@ -6038,6 +6580,6 @@ function getCodexEnvVars(serverUrl) {
6038
6580
  return vars;
6039
6581
  }
6040
6582
  //#endregion
6041
- export { sharedServerArgs as a, listModelsForEndpoint as c, checkClaudeVersion as d, updateClaude as f, setupAndServe as i, enableFileLogging as l, getCodexEnvVars as n, startKeepAwake as o, parseSharedArgs as r, stopKeepAwake as s, getClaudeCodeEnvVars as t, runSelfUpdate as u };
6583
+ export { validateFastProfilePrerequisites as _, sharedServerArgs as a, updateClaude as b, stopKeepAwake as c, LUNA_DRIVER_ALIAS_ID as d, LUNA_IMPLEMENTER_ALIAS_ID as f, resolveLaunchProfile as g, profileDescriptor as h, setupAndServe as i, listModelsForEndpoint as l, formatFastPrerequisiteFailure as m, getCodexEnvVars as n, LAUNCH_SECRET_HEADER as o, LUNA_SCOUT_ALIAS_ID as p, parseSharedArgs as r, startKeepAwake as s, getClaudeCodeEnvVars as t, enableFileLogging as u, runSelfUpdate as v, checkClaudeVersion as y };
6042
6584
 
6043
- //# sourceMappingURL=server-setup-DlztZAGT.js.map
6585
+ //# sourceMappingURL=server-setup-DbvbW5Ve.js.map