github-router 0.3.289 → 0.3.292

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/dist/{attribution-settings-Cmz2jt7P.js → attribution-settings-CpLUCi8R.js} +124 -34
  2. package/dist/attribution-settings-CpLUCi8R.js.map +1 -0
  3. package/dist/{auth-DG4vh8-F.js → auth-BwUHopJz.js} +3 -3
  4. package/dist/{auth-DG4vh8-F.js.map → auth-BwUHopJz.js.map} +1 -1
  5. package/dist/browser-ext/manifest.json +1 -1
  6. package/dist/{check-usage-BqN7mBYv.js → check-usage-BTda5753.js} +4 -4
  7. package/dist/{check-usage-BqN7mBYv.js.map → check-usage-BTda5753.js.map} +1 -1
  8. package/dist/{claude-C-9xFI4b.js → claude-_DYGKCw8.js} +99 -48
  9. package/dist/claude-_DYGKCw8.js.map +1 -0
  10. package/dist/{codex-DRW0yb8x.js → codex-rTJ8jW5G.js} +5 -5
  11. package/dist/{codex-DRW0yb8x.js.map → codex-rTJ8jW5G.js.map} +1 -1
  12. package/dist/{debug-B5TjPTTH.js → debug-B3UrZTHQ.js} +2 -2
  13. package/dist/{debug-B5TjPTTH.js.map → debug-B3UrZTHQ.js.map} +1 -1
  14. package/dist/engine-C9axTIu7.js +2 -0
  15. package/dist/{gate-discovery-Cz6kwIVG.js → gate-discovery-Bjar5dgv.js} +5 -5
  16. package/dist/{gate-discovery-Cz6kwIVG.js.map → gate-discovery-Bjar5dgv.js.map} +1 -1
  17. package/dist/{get-copilot-usage-BjA0nyGR.js → get-copilot-usage-CRrf1ZSC.js} +2 -2
  18. package/dist/{get-copilot-usage-BjA0nyGR.js.map → get-copilot-usage-CRrf1ZSC.js.map} +1 -1
  19. package/dist/{internal-artifact-open-BskmUpnb.js → internal-artifact-open-BsEQRvDi.js} +2 -2
  20. package/dist/{internal-artifact-open-BskmUpnb.js.map → internal-artifact-open-BsEQRvDi.js.map} +1 -1
  21. package/dist/{internal-first-mate-guard-5XEHMaqy.js → internal-first-mate-guard-CVFFJriA.js} +3 -3
  22. package/dist/{internal-first-mate-guard-5XEHMaqy.js.map → internal-first-mate-guard-CVFFJriA.js.map} +1 -1
  23. package/dist/{internal-first-mate-guard-DHDQ6hFz.js → internal-first-mate-guard-DJdX6t48.js} +1 -1
  24. package/dist/{internal-plan-review-BUuMx4ku.js → internal-plan-review-8TKDLco6.js} +3 -3
  25. package/dist/{internal-plan-review-BUuMx4ku.js.map → internal-plan-review-8TKDLco6.js.map} +1 -1
  26. package/dist/{internal-prompt-submit-CQQ15xdO.js → internal-prompt-submit-Da7pxqua.js} +4 -4
  27. package/dist/{internal-prompt-submit-CQQ15xdO.js.map → internal-prompt-submit-Da7pxqua.js.map} +1 -1
  28. package/dist/{internal-session-bind-D04W2yWI.js → internal-session-bind-BOFytA1f.js} +2 -2
  29. package/dist/{internal-session-bind-D04W2yWI.js.map → internal-session-bind-BOFytA1f.js.map} +1 -1
  30. package/dist/{internal-stop-hook-Dvkppwo7.js → internal-stop-hook-DrX2xlj0.js} +5 -5
  31. package/dist/{internal-stop-hook-Dvkppwo7.js.map → internal-stop-hook-DrX2xlj0.js.map} +1 -1
  32. package/dist/{internal-stop-review-CdByyJLc.js → internal-stop-review-CdouacHL.js} +2 -2
  33. package/dist/{internal-stop-review-CdByyJLc.js.map → internal-stop-review-CdouacHL.js.map} +1 -1
  34. package/dist/{internal-worker-guard-BIPN6Rv9.js → internal-worker-guard-Bx-itoP8.js} +2 -2
  35. package/dist/{internal-worker-guard-BIPN6Rv9.js.map → internal-worker-guard-Bx-itoP8.js.map} +1 -1
  36. package/dist/{internal-workspace-header-BKqejstG.js → internal-workspace-header-8WT0iB5K.js} +2 -2
  37. package/dist/{internal-workspace-header-BKqejstG.js.map → internal-workspace-header-8WT0iB5K.js.map} +1 -1
  38. package/dist/lifecycle-C8fOsQke.js +2 -0
  39. package/dist/lifecycle-D4Yc1aap.js +2 -0
  40. package/dist/{lifecycle-SXaWssN9.js → lifecycle-LeSfa7wH.js} +2 -2
  41. package/dist/{lifecycle-SXaWssN9.js.map → lifecycle-LeSfa7wH.js.map} +1 -1
  42. package/dist/{lifecycle-DbM29FLK.js → lifecycle-nuOHfwgj.js} +2 -2
  43. package/dist/{lifecycle-DbM29FLK.js.map → lifecycle-nuOHfwgj.js.map} +1 -1
  44. package/dist/main.js +17 -17
  45. package/dist/{mcp-workspace-header-DRCCWlOi.js → mcp-workspace-header-q34H_4wL.js} +2 -2
  46. package/dist/{mcp-workspace-header-DRCCWlOi.js.map → mcp-workspace-header-q34H_4wL.js.map} +1 -1
  47. package/dist/{models-Dz8d_SnI.js → models-hhJcrZhr.js} +3 -3
  48. package/dist/{models-Dz8d_SnI.js.map → models-hhJcrZhr.js.map} +1 -1
  49. package/dist/{orchestration-BrJwZxMN.js → orchestration-pzbrKkgD.js} +2 -2
  50. package/dist/{orchestration-BrJwZxMN.js.map → orchestration-pzbrKkgD.js.map} +1 -1
  51. package/dist/{paths-D7_SAaIQ.js → paths-BH4J7slC.js} +4 -4
  52. package/dist/{paths-D7_SAaIQ.js.map → paths-BH4J7slC.js.map} +1 -1
  53. package/dist/paths-DJZoXfAS.js +2 -0
  54. package/dist/{peer-mcp-personas-Bd56EmiO.js → peer-mcp-personas-CHbl6MwM.js} +544 -128
  55. package/dist/peer-mcp-personas-CHbl6MwM.js.map +1 -0
  56. package/dist/{plan-review-hook-CVZsG9MZ.js → plan-review-hook-CfcanA7_.js} +3 -3
  57. package/dist/{plan-review-hook-CVZsG9MZ.js.map → plan-review-hook-CfcanA7_.js.map} +1 -1
  58. package/dist/{prompt-submit-hook-BW92FX2D.js → prompt-submit-hook-Bqf9ORgb.js} +3 -3
  59. package/dist/{prompt-submit-hook-BW92FX2D.js.map → prompt-submit-hook-Bqf9ORgb.js.map} +1 -1
  60. package/dist/{provision-B53wbHwa.js → provision-BYFd9nPK.js} +4 -4
  61. package/dist/{provision-B53wbHwa.js.map → provision-BYFd9nPK.js.map} +1 -1
  62. package/dist/{self-invocation-CP_SOkrr.js → self-invocation-DhO1Z8iD.js} +2 -2
  63. package/dist/{self-invocation-CP_SOkrr.js.map → self-invocation-DhO1Z8iD.js.map} +1 -1
  64. package/dist/{serve-aZCEYFe5.js → serve-BWMxDLnD.js} +12 -12
  65. package/dist/{serve-aZCEYFe5.js.map → serve-BWMxDLnD.js.map} +1 -1
  66. package/dist/{server-setup-DlztZAGT.js → server-setup-CqlaZukJ.js} +609 -74
  67. package/dist/server-setup-CqlaZukJ.js.map +1 -0
  68. package/dist/{start-DwNiXv5N.js → start-5MgGT4IF.js} +3 -3
  69. package/dist/{start-DwNiXv5N.js.map → start-5MgGT4IF.js.map} +1 -1
  70. package/dist/{stop-gate-hook-DriRc9xN.js → stop-gate-hook-BiBp5aGm.js} +3 -3
  71. package/dist/{stop-gate-hook-DriRc9xN.js.map → stop-gate-hook-BiBp5aGm.js.map} +1 -1
  72. package/dist/{stop-gate-policy-DMPanpoR.js → stop-gate-policy-BGd6b5hR.js} +2 -2
  73. package/dist/{stop-gate-policy-DMPanpoR.js.map → stop-gate-policy-BGd6b5hR.js.map} +1 -1
  74. package/dist/{token-BGCjZwtj.js → token-8drORhXg.js} +33 -3
  75. package/dist/token-8drORhXg.js.map +1 -0
  76. package/dist/{worker-dispatch-BCTMyNE-.js → worker-dispatch-D5fGroNr.js} +2 -2
  77. package/dist/{worker-dispatch-BCTMyNE-.js.map → worker-dispatch-D5fGroNr.js.map} +1 -1
  78. package/package.json +1 -1
  79. package/dist/attribution-settings-Cmz2jt7P.js.map +0 -1
  80. package/dist/claude-C-9xFI4b.js.map +0 -1
  81. package/dist/engine-iEqGdx6T.js +0 -2
  82. package/dist/lifecycle-BTodQvn4.js +0 -2
  83. package/dist/lifecycle-C7JYNz-F.js +0 -2
  84. package/dist/paths-CTr59UC6.js +0 -2
  85. package/dist/peer-mcp-personas-Bd56EmiO.js.map +0 -1
  86. package/dist/server-setup-DlztZAGT.js.map +0 -1
  87. package/dist/token-BGCjZwtj.js.map +0 -1
@@ -1,8 +1,8 @@
1
- import { $ as handleMcpDelete, $t as UPSTREAM_INACTIVITY_TIMEOUT_MS, At as readResponseBodyCapped, B as rememberThinkingHistoryRepair, Bt as provisionTreeSitterAssets, Ct as getTokenCount, Dt as createResponses, Et as resolveMcpToolTimeoutMs, F as injectAdvisorTool, G as isControllerClosedError, Gt as toolbeltPathOverride, H as repairRejectedThinkingHistory, I as isAdvisorRequested, J as relayAnthropicStream, K as logStreamError, L as resolveAdvisorEffort, M as ADVISOR_INTERNAL_TOOL_NAME, Mt as normalizeOpenAIUsage, N as ADVISOR_TOOL_INSTRUCTIONS, Ot as createChatCompletions, P as buildAdvisorStream, Q as clampEffort, Qt as UPSTREAM_FETCH_TIMEOUT_MS, R as resolveAdvisorModel, St as createMessages, T as toolbeltEnabled, Tt as warnOnTokenPriceDrift, U as buildAnthropicErrorEvent, V as repairKnownThinkingHistory, W as buildOpenAIErrorEvent, X as UNKNOWN_EFFORT_ANCHOR, Y as EFFORT_ORDER, Z as bucketEffort, an as upstreamMaxConnections, bt as shimDefaultsToXhigh, cn as withOneMSuffixForLead, en as generateRandomPort, et as handleMcpPost, in as upstreamAllowH2, j as searchWeb, jt as parseJsonOrDiagnose, kt as MAX_RESPONSE_BODY_BYTES, ln as withInstallLock, nt as agentToolsEnabled, on as classifyMessagesRoute, pt as resolveGeminiReviewModel, q as readIteratorWithTimeout, qt as BUDGET_SMALL_FAST_SLUG, r as assertMcpToolSurfaceConsistent, sn as withOneMSuffix, tn as isBudgetClaudeLead, wt as assembleResponsesPayload, xt as countTokens, z as formatThinkingRepairDecline } from "./peer-mcp-personas-Bd56EmiO.js";
1
+ import { $ as bucketEffort, At as countTokens, B as resolveAdvisorModel, Bt as createChatCompletions, Cn as withOneMSuffixForLead, E as toolbeltEnabled, F as buildAdvisorStream, G as buildAnthropicErrorEvent, H as rememberThinkingHistoryRepair, Ht as readResponseBodyCapped, I as injectAdvisorTool, It as assembleResponsesPayload, J as logStreamError, K as buildOpenAIErrorEvent, L as isAdvisorRequested, Lt as warnOnTokenPriceDrift, M as searchWeb, Mt as getTokenCount, N as ADVISOR_INTERNAL_TOOL_NAME, Nt as findLaunchBySecret, P as ADVISOR_TOOL_INSTRUCTIONS, Q as UNKNOWN_EFFORT_ANCHOR, Qt as provisionTreeSitterAssets, R as isFastProfileLead, Rt as resolveMcpToolTimeoutMs, Sn as withOneMSuffix, U as repairKnownThinkingHistory, Ut as parseJsonOrDiagnose, V as formatThinkingRepairDecline, Vt as MAX_RESPONSE_BODY_BYTES, W as repairRejectedThinkingHistory, Wt as normalizeOpenAIUsage, X as relayAnthropicStream, Y as readIteratorWithTimeout, Z as EFFORT_ORDER, _n as upstreamMaxConnections, an as BUDGET_SMALL_FAST_SLUG, bn as catalogAdvertises1M, dn as UPSTREAM_INACTIVITY_TIMEOUT_MS, et as clampEffort, fn as generateRandomPort, gn as upstreamAllowH2, i as assertMcpToolSurfaceConsistent, jt as createMessages, kt as shimDefaultsToXhigh, nt as handleMcpPost, ot as agentToolsEnabled, pn as isBudgetClaudeLead, q as isControllerClosedError, rn as toolbeltPathOverride, tt as handleMcpDelete, un as UPSTREAM_FETCH_TIMEOUT_MS, vn as classifyMessagesRoute, wn as withInstallLock, xn as oneMContextDisabled, yn as pickEndpoint, z as resolveAdvisorEffort, zt as createResponses } from "./peer-mcp-personas-CHbl6MwM.js";
2
2
  import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
3
- import { i as ensurePaths, t as PATHS } from "./paths-D7_SAaIQ.js";
4
- import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-BGCjZwtj.js";
5
- import { t as getCopilotUsage } from "./get-copilot-usage-BjA0nyGR.js";
3
+ import { i as ensurePaths, t as PATHS } from "./paths-BH4J7slC.js";
4
+ import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-8drORhXg.js";
5
+ import { t as getCopilotUsage } from "./get-copilot-usage-CRrf1ZSC.js";
6
6
  import { a as resolveExecutable, n as killManagedTree, o as runCommandCapture, r as parseBoolEnv, s as runCommandVoid } from "./exec-y8C_MU8A.js";
7
7
  import consola from "consola";
8
8
  import * as fs$2 from "node:fs";
@@ -354,6 +354,232 @@ async function runSelfUpdate(opts) {
354
354
  }
355
355
  }
356
356
  //#endregion
357
+ //#region src/lib/launch-profile.ts
358
+ const STANDARD_PROFILE = Object.freeze({
359
+ id: "standard",
360
+ hasCoordinator: true
361
+ });
362
+ /**
363
+ * The `-m fast` roster: exactly four native agents (`scout`, `implementer`,
364
+ * `reviewer`, `planner`), the fast-only `oracle` peer tool, no coordinator,
365
+ * and only `peers`/`search` plus the ordinary opt-in `browser` group.
366
+ * `workers`/`orchestrate`/`decide`/`fleet`/`first-mate` are hard denies even
367
+ * when their independent standard-profile gates pass.
368
+ */
369
+ const FAST_PROFILE = Object.freeze({
370
+ id: "fast",
371
+ nativeRoster: /* @__PURE__ */ new Set([
372
+ "scout",
373
+ "implementer",
374
+ "reviewer",
375
+ "planner"
376
+ ]),
377
+ personaAllowlist: /* @__PURE__ */ new Set(["oracle"]),
378
+ allowedGroups: /* @__PURE__ */ new Set([
379
+ "peers",
380
+ "search",
381
+ "browser"
382
+ ]),
383
+ hasCoordinator: false
384
+ });
385
+ function profileDescriptor(id) {
386
+ return id === "fast" ? FAST_PROFILE : STANDARD_PROFILE;
387
+ }
388
+ /**
389
+ * Resolve the parsed `-m` argument to a launch profile.
390
+ *
391
+ * Deliberately keyed on the RAW alias string (trimmed, case-insensitive
392
+ * `"fast"`), never on a resolved model id: `resolveLeadSlugArg` maps `fast`
393
+ * to `FAST_LEAD_MODEL` (`./port`) before this is of any use to a caller who
394
+ * only has the resolved id, so callers that already resolved the lead must
395
+ * pass the ORIGINAL `-m` value here, not the resolved one. This is what
396
+ * keeps `-m gpt-5.6-luna` (a direct pin of the same underlying model) a
397
+ * standard-surface launch — only the literal alias narrows the surface.
398
+ */
399
+ function resolveLaunchProfile(modelArg) {
400
+ return modelArg?.trim().toLowerCase() === "fast" ? "fast" : "standard";
401
+ }
402
+ /**
403
+ * Router-owned alias id for the fast profile's Sonnet-tier row
404
+ * (`ANTHROPIC_DEFAULT_SONNET_MODEL`). Never sent upstream — canonicalized to
405
+ * `LUNA_REAL_MODEL_ID` by `canonicalizeAliasModel` before the request
406
+ * reaches Copilot.
407
+ */
408
+ const LUNA_DRIVER_ALIAS_ID = "gh-router-luna-driver-max";
409
+ /** Fast native-agent alias ids preserve role-specific effort provenance until
410
+ * the authenticated request boundary. They both canonicalize to Luna, but the
411
+ * scout is fixed high while the implementer is fixed max. */
412
+ const LUNA_SCOUT_ALIAS_ID = "gh-router-luna-scout-high";
413
+ const LUNA_IMPLEMENTER_ALIAS_ID = "gh-router-luna-implementer-max";
414
+ const LUNA_SONNET_ALIAS_ID = "gh-router-luna-sonnet-xhigh";
415
+ /**
416
+ * Router-owned alias id for the fast profile's Haiku-tier row
417
+ * (`ANTHROPIC_DEFAULT_HAIKU_MODEL` / `ANTHROPIC_SMALL_FAST_MODEL`).
418
+ */
419
+ const LUNA_HAIKU_ALIAS_ID = "gh-router-luna-haiku-high";
420
+ /** The real Copilot catalog id every Luna alias (including the driver
421
+ * itself) canonicalizes to. */
422
+ const LUNA_REAL_MODEL_ID = "gpt-5.6-luna";
423
+ /**
424
+ * The full alias table, keyed by `aliasId`. A simpler model-id-only table is
425
+ * rejected by design: the driver, the Sonnet tier, and the Haiku tier all
426
+ * resolve to the SAME Luna catalog id, so after early canonicalization a
427
+ * table keyed on the real id could no longer tell which absent-effort
428
+ * default applies. Alias provenance — which of the three ids the request
429
+ * actually carried — is the minimum discriminator that survives from tier
430
+ * selection through to request preprocessing, which is why canonicalization
431
+ * must happen LAST (in the `/v1/messages` identity preflight), after the
432
+ * effort default has already been read off the alias.
433
+ */
434
+ const MODEL_ALIAS_TABLE = /* @__PURE__ */ new Map([
435
+ [LUNA_DRIVER_ALIAS_ID, {
436
+ aliasId: LUNA_DRIVER_ALIAS_ID,
437
+ realModel: LUNA_REAL_MODEL_ID,
438
+ absentEffortDefault: "max"
439
+ }],
440
+ [LUNA_SCOUT_ALIAS_ID, {
441
+ aliasId: LUNA_SCOUT_ALIAS_ID,
442
+ realModel: LUNA_REAL_MODEL_ID,
443
+ absentEffortDefault: "high"
444
+ }],
445
+ [LUNA_IMPLEMENTER_ALIAS_ID, {
446
+ aliasId: LUNA_IMPLEMENTER_ALIAS_ID,
447
+ realModel: LUNA_REAL_MODEL_ID,
448
+ absentEffortDefault: "max"
449
+ }],
450
+ [LUNA_SONNET_ALIAS_ID, {
451
+ aliasId: LUNA_SONNET_ALIAS_ID,
452
+ realModel: LUNA_REAL_MODEL_ID,
453
+ absentEffortDefault: "xhigh"
454
+ }],
455
+ [LUNA_HAIKU_ALIAS_ID, {
456
+ aliasId: LUNA_HAIKU_ALIAS_ID,
457
+ realModel: LUNA_REAL_MODEL_ID,
458
+ absentEffortDefault: "high"
459
+ }]
460
+ ]);
461
+ /**
462
+ * Look up the alias descriptor for a wire-facing model id (with or without
463
+ * a trailing `[1m]` bracket — the bracket is stripped before the table
464
+ * lookup and is orthogonal to alias identity). Returns undefined for any
465
+ * id that isn't one of the three registered aliases (including the bare
466
+ * `claude-*` ids and every other real Copilot catalog id).
467
+ */
468
+ function resolveModelAlias(id) {
469
+ const bare = id.replace(/\[1m\]$/i, "");
470
+ return MODEL_ALIAS_TABLE.get(bare);
471
+ }
472
+ /**
473
+ * Strip alias provenance and return the real catalog id to send upstream.
474
+ * Idempotent passthrough for any id that isn't a registered alias (a bare
475
+ * `claude-*` slug, an already-real Copilot id, or anything else) — this is
476
+ * safe to call unconditionally on every `body.model` at the outbound
477
+ * boundary. Preserves a trailing `[1m]` bracket: canonicalization only
478
+ * erases ALIAS identity, not the 1M-context accounting decoration.
479
+ */
480
+ function canonicalizeAliasModel(id) {
481
+ const bracket = /\[1m\]$/i.test(id) ? "[1m]" : "";
482
+ const bare = bracket ? id.slice(0, -bracket.length) : id;
483
+ const alias = MODEL_ALIAS_TABLE.get(bare);
484
+ return alias ? `${alias.realModel}${bracket}` : id;
485
+ }
486
+ const FAST_REQUIRED_CONTEXT_TOKENS = 1e6;
487
+ function findModel(catalog, id) {
488
+ return catalog?.data?.find((m) => m.id === id);
489
+ }
490
+ function hasToolCalls(model) {
491
+ return model?.capabilities?.supports?.tool_calls === true;
492
+ }
493
+ function hasContextAtLeast(model, tokens) {
494
+ return (model?.capabilities?.limits?.max_context_window_tokens ?? 0) >= tokens;
495
+ }
496
+ function supportsEffort(model, effort) {
497
+ const list = model?.capabilities?.supports?.reasoning_effort;
498
+ return Array.isArray(list) && list.includes(effort);
499
+ }
500
+ function supportsEndpoint(model, paths) {
501
+ const endpoints = model?.supported_endpoints;
502
+ return Array.isArray(endpoints) && endpoints.some((endpoint) => paths.has(endpoint));
503
+ }
504
+ const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
505
+ const MESSAGES_ENDPOINTS = /* @__PURE__ */ new Set(["/messages", "/v1/messages"]);
506
+ function hasUsablePromptMetadata(model) {
507
+ const prompt = model?.capabilities?.limits?.max_prompt_tokens;
508
+ return typeof prompt === "number" && Number.isFinite(prompt) && prompt > 0;
509
+ }
510
+ /**
511
+ * Validate the live Copilot catalog carries every model the fast profile's
512
+ * EXACT roster depends on, with the specific capabilities each assignment
513
+ * needs. These are capability-availability PREREQUISITES for constructing
514
+ * the roster — not an allowlist of models the user may select later in the
515
+ * session — so a partial catalog fails the whole `-m fast` launch rather
516
+ * than silently substituting or dropping an agent.
517
+ *
518
+ * Checks, per the fast-launch-profile design:
519
+ * - Luna lead/scout/implementer: tool calls, >=1M, high+max, Responses.
520
+ * - Sol planner: tool calls, >=1M, high, Responses.
521
+ * - Grok reviewer: tool calls, medium, Responses, usable prompt metadata.
522
+ * - Gemini Advisor: >=1M, high, chat-completions.
523
+ * - Opus Oracle: exact Opus 5, >=1M, adaptive/high, Messages, prompt metadata.
524
+ *
525
+ * Pure over the passed-in catalog snapshot so it's unit-testable without
526
+ * `state` — callers pass `state.models` at call time.
527
+ */
528
+ function validateFastProfilePrerequisites(catalog) {
529
+ const missing = [];
530
+ const luna = findModel(catalog, LUNA_REAL_MODEL_ID);
531
+ if (!luna) missing.push(`${LUNA_REAL_MODEL_ID}: absent from the live catalog`);
532
+ else {
533
+ if (!hasToolCalls(luna)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise tool_calls`);
534
+ if (!hasContextAtLeast(luna, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push(`${LUNA_REAL_MODEL_ID}: advertised context window is below 1M`);
535
+ if (!supportsEffort(luna, "high") || !supportsEffort(luna, "max")) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise both "high" and "max" reasoning effort`);
536
+ if (!supportsEndpoint(luna, RESPONSES_ENDPOINTS)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise a supported Responses endpoint`);
537
+ }
538
+ const sol = findModel(catalog, "gpt-5.6-sol");
539
+ if (!sol) missing.push("gpt-5.6-sol: absent from the live catalog");
540
+ else {
541
+ if (!hasToolCalls(sol)) missing.push("gpt-5.6-sol: does not advertise tool_calls");
542
+ if (!hasContextAtLeast(sol, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gpt-5.6-sol: advertised context window is below 1M");
543
+ if (!supportsEffort(sol, "high")) missing.push("gpt-5.6-sol: does not advertise a \"high\" reasoning effort");
544
+ if (!supportsEndpoint(sol, RESPONSES_ENDPOINTS)) missing.push("gpt-5.6-sol: does not advertise a supported Responses endpoint");
545
+ }
546
+ const grok = findModel(catalog, "grok-4.6");
547
+ if (!grok) missing.push("grok-4.6: absent from the live catalog");
548
+ else {
549
+ if (!hasToolCalls(grok)) missing.push("grok-4.6: does not advertise tool_calls");
550
+ if (!supportsEffort(grok, "medium")) missing.push("grok-4.6: does not advertise a \"medium\" reasoning effort");
551
+ if (!hasUsablePromptMetadata(grok)) missing.push("grok-4.6: no usable max_prompt_tokens metadata");
552
+ if (!supportsEndpoint(grok, RESPONSES_ENDPOINTS)) missing.push("grok-4.6: does not advertise a supported Responses endpoint");
553
+ }
554
+ const gemini = findModel(catalog, "gemini-3.7-flash");
555
+ if (!gemini) missing.push("gemini-3.7-flash: absent from the live catalog");
556
+ else {
557
+ if (!hasContextAtLeast(gemini, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gemini-3.7-flash: advertised context window is below 1M");
558
+ if (!supportsEffort(gemini, "high")) missing.push("gemini-3.7-flash: does not advertise a \"high\" reasoning effort");
559
+ if (pickEndpoint(gemini) !== "chat") missing.push("gemini-3.7-flash: does not advertise a supported chat-completions endpoint");
560
+ }
561
+ const opus = findModel(catalog, "claude-opus-5");
562
+ if (!opus) missing.push("claude-opus-5: absent from the live catalog");
563
+ else {
564
+ if (!hasContextAtLeast(opus, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("claude-opus-5: advertised context window is below 1M");
565
+ if (!supportsEffort(opus, "high")) missing.push("claude-opus-5: does not advertise a \"high\" reasoning effort");
566
+ if (opus.capabilities?.supports?.adaptive_thinking !== true) missing.push("claude-opus-5: does not advertise adaptive_thinking");
567
+ if (!hasUsablePromptMetadata(opus)) missing.push("claude-opus-5: no usable max_prompt_tokens metadata");
568
+ if (!supportsEndpoint(opus, MESSAGES_ENDPOINTS)) missing.push("claude-opus-5: does not advertise a supported Messages endpoint");
569
+ }
570
+ return {
571
+ ok: missing.length === 0,
572
+ missing
573
+ };
574
+ }
575
+ /**
576
+ * Format `validateFastProfilePrerequisites`'s failure list into the launch
577
+ * error message: every missing/invalid model, plus the rollback command.
578
+ */
579
+ function formatFastPrerequisiteFailure(missing) {
580
+ return "github-router claude -m fast requires the following live-catalog capabilities, which this account's catalog does not fully provide:\n" + missing.map((m) => ` - ${m}`).join("\n") + "\n\nFalling back or silently dropping an agent is not supported for the fast profile's exact roster. Run plain `github-router claude` instead.";
581
+ }
582
+ //#endregion
357
583
  //#region src/lib/file-log-reporter.ts
358
584
  const MAX_LOG_BYTES = 1048576;
359
585
  const DEDUP_MAX = 1e3;
@@ -1430,7 +1656,7 @@ function collectToolFieldKeys(body) {
1430
1656
  //#endregion
1431
1657
  //#region package.json
1432
1658
  var name = "github-router";
1433
- var version = "0.3.289";
1659
+ var version = "0.3.292";
1434
1660
  //#endregion
1435
1661
  //#region src/lib/approval.ts
1436
1662
  const awaitApproval = async () => {
@@ -3063,6 +3289,18 @@ function anthropicMessageToNeutral(msg) {
3063
3289
  name: typeof b.name === "string" ? b.name : "",
3064
3290
  arguments: b.input ?? {}
3065
3291
  });
3292
+ else if (b.type === "server_tool_use" && b.name === "advisor") parts.push({
3293
+ type: "text",
3294
+ text: "[Consulted advisor]"
3295
+ });
3296
+ else if (b.type === "advisor_tool_result") {
3297
+ const resultContent = b.content;
3298
+ const text = resultContent && typeof resultContent === "object" && typeof resultContent.text === "string" ? resultContent.text : "";
3299
+ if (text.length > 0) parts.push({
3300
+ type: "text",
3301
+ text: `[Advisor response]\n${text}`
3302
+ });
3303
+ }
3066
3304
  }
3067
3305
  return [{
3068
3306
  role: "assistant",
@@ -4042,6 +4280,75 @@ function isAsyncIterable(x) {
4042
4280
  return x != null && typeof x[Symbol.asyncIterator] === "function";
4043
4281
  }
4044
4282
  /**
4283
+ * Context-free, caller-abortable core of the non-Claude shim's STREAMING
4284
+ * path for one already-parsed Anthropic request.
4285
+ *
4286
+ * "Context-free": no Hono `Context` dependency, unlike
4287
+ * `handleNonClaudeResponses`/`handleNonClaudeChat` below (which need one to
4288
+ * build their JSON error responses and read `c.req.path` for logging).
4289
+ * "Caller-abortable": takes the caller's OWN `AbortSignal` rather than
4290
+ * constructing an internal `AbortController` — this function never creates
4291
+ * one — and forwards an optional `onCancel` so a caller that DOES own a
4292
+ * controller (the two handlers below, for the initial request) can still
4293
+ * tear it down when the stream it returns is cancelled.
4294
+ *
4295
+ * Shared by:
4296
+ * - `handleNonClaudeResponses` / `handleNonClaudeChat` (this module), for
4297
+ * the initial request — each already knows its own endpoint statically
4298
+ * (they are dispatched by `classifyMessagesRoute`), so this function
4299
+ * takes `endpoint` as an explicit argument rather than re-deriving it
4300
+ * from the catalog (which would also mean re-parsing/re-picking work the
4301
+ * caller already did).
4302
+ * - `makeShimContinueTurn` (below), which `buildAdvisorStream`
4303
+ * (`src/services/advisor/advisor.ts`) injects as its `continueTurn` for
4304
+ * the fast Luna-lead profile, so an advisor continuation on a non-Claude
4305
+ * lead runs through the SAME translation + SSE-synthesis machinery as
4306
+ * the initial turn instead of a parallel, divergent implementation.
4307
+ */
4308
+ async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
4309
+ const routePath = opts.routePath ?? "/v1/messages (advisor lead shim)";
4310
+ if (endpoint === "chat") {
4311
+ const payload = parsedToChatPayload(parsed);
4312
+ const result = await createChatCompletions(payload, opts.model?.requestHeaders, signal, true);
4313
+ if (!isAsyncIterable(result)) throw new Error("Upstream /chat/completions did not return an SSE stream (stream: true expected)");
4314
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromChat(result, { modelId: opts.modelId }), {
4315
+ routePath,
4316
+ onCancel: opts.onCancel,
4317
+ inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4318
+ });
4319
+ return new Response(stream, {
4320
+ status: 200,
4321
+ headers: STREAM_HEADERS
4322
+ });
4323
+ }
4324
+ const payload = parsedToResponsesPayload(parsed);
4325
+ const result = await createResponses(payload, opts.model?.requestHeaders, signal, true);
4326
+ if (!isAsyncIterable(result)) throw new Error("Upstream /responses did not return an SSE stream (stream: true expected)");
4327
+ const stream = anthropicSseStreamFromEvents(synthAnthropicFromResponses(result, { modelId: opts.modelId }), {
4328
+ routePath,
4329
+ onCancel: opts.onCancel,
4330
+ inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4331
+ });
4332
+ return new Response(stream, {
4333
+ status: 200,
4334
+ headers: STREAM_HEADERS
4335
+ });
4336
+ }
4337
+ /**
4338
+ * Build an injectable `continueTurn(body, signal)` for `buildAdvisorStream`
4339
+ * (`src/services/advisor/advisor.ts`) that routes a continuation turn
4340
+ * through THIS module's non-Claude shim instead of Claude passthrough — used
4341
+ * for the fast Luna-lead profile's advisor translate-loop. No `onCancel` is
4342
+ * threaded through: the advisor loop's own `aborter` (shared with `signal`
4343
+ * here) already tears down on consumer cancel via `buildAdvisorStream`'s
4344
+ * `cancel()`, so this stream needs no independent teardown hook.
4345
+ */
4346
+ function makeShimContinueTurn(endpoint, opts) {
4347
+ return (body, signal) => {
4348
+ return streamParsedRequestViaShim(parseAnthropicRequest(body, opts.modelId, opts.model), endpoint, opts, signal);
4349
+ };
4350
+ }
4351
+ /**
4045
4352
  * Handle a `/v1/messages` request targeting a non-Claude `/responses` model.
4046
4353
  * Returns a streaming or non-streaming Anthropic-format Response. Upstream
4047
4354
  * non-2xx / abort errors are thrown (as HTTPError) and handled by the route's
@@ -4062,12 +4369,15 @@ async function handleNonClaudeResponses(c, opts) {
4062
4369
  }, 400);
4063
4370
  }
4064
4371
  const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
4065
- const payload = parsedToResponsesPayload(parsed);
4066
4372
  if (consola.level >= 4) consola.debug(`Anthropic-translate → /responses model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
4067
4373
  if (parsed.stream) {
4068
4374
  const aborter = new AbortController();
4069
- const result = await createResponses(payload, opts.model?.requestHeaders, aborter.signal, true);
4070
- if (!isAsyncIterable(result)) throw new Error("Upstream /responses did not return an SSE stream (stream: true expected)");
4375
+ const stream = await streamParsedRequestViaShim(parsed, "responses", {
4376
+ modelId: opts.modelId,
4377
+ model: opts.model,
4378
+ routePath,
4379
+ onCancel: () => aborter.abort()
4380
+ }, aborter.signal);
4071
4381
  logRequest({
4072
4382
  method: "POST",
4073
4383
  path: routePath,
@@ -4076,16 +4386,9 @@ async function handleNonClaudeResponses(c, opts) {
4076
4386
  status: 200,
4077
4387
  streaming: true
4078
4388
  }, opts.model, opts.startTime);
4079
- const stream = anthropicSseStreamFromEvents(synthAnthropicFromResponses(result, { modelId: opts.modelId }), {
4080
- routePath,
4081
- onCancel: () => aborter.abort(),
4082
- inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4083
- });
4084
- return new Response(stream, {
4085
- status: 200,
4086
- headers: STREAM_HEADERS
4087
- });
4389
+ return stream;
4088
4390
  }
4391
+ const payload = parsedToResponsesPayload(parsed);
4089
4392
  const anthropic = responsesResponseToAnthropicMessage(await createResponses(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
4090
4393
  const usage = anthropic.usage;
4091
4394
  logRequest({
@@ -4122,12 +4425,15 @@ async function handleNonClaudeChat(c, opts) {
4122
4425
  }, 400);
4123
4426
  }
4124
4427
  const parsed = parseAnthropicRequest(body, opts.modelId, opts.model);
4125
- const payload = parsedToChatPayload(parsed);
4126
4428
  if (consola.level >= 4) consola.debug(`Anthropic-translate → /chat/completions model=${opts.modelId} stream=${parsed.stream} tools=${parsed.tools?.length ?? 0} effort=${parsed.reasoningEffort ?? "none"}`);
4127
4429
  if (parsed.stream) {
4128
4430
  const aborter = new AbortController();
4129
- const result = await createChatCompletions(payload, opts.model?.requestHeaders, aborter.signal, true);
4130
- if (!isAsyncIterable(result)) throw new Error("Upstream /chat/completions did not return an SSE stream (stream: true expected)");
4431
+ const stream = await streamParsedRequestViaShim(parsed, "chat", {
4432
+ modelId: opts.modelId,
4433
+ model: opts.model,
4434
+ routePath,
4435
+ onCancel: () => aborter.abort()
4436
+ }, aborter.signal);
4131
4437
  logRequest({
4132
4438
  method: "POST",
4133
4439
  path: routePath,
@@ -4136,16 +4442,9 @@ async function handleNonClaudeChat(c, opts) {
4136
4442
  status: 200,
4137
4443
  streaming: true
4138
4444
  }, opts.model, opts.startTime);
4139
- const stream = anthropicSseStreamFromEvents(synthAnthropicFromChat(result, { modelId: opts.modelId }), {
4140
- routePath,
4141
- onCancel: () => aborter.abort(),
4142
- inactivityTimeoutMs: UPSTREAM_INACTIVITY_TIMEOUT_MS
4143
- });
4144
- return new Response(stream, {
4145
- status: 200,
4146
- headers: STREAM_HEADERS
4147
- });
4445
+ return stream;
4148
4446
  }
4447
+ const payload = parsedToChatPayload(parsed);
4149
4448
  const anthropic = chatResponseToAnthropicMessage(await createChatCompletions(payload, opts.model?.requestHeaders, void 0, true), opts.modelId);
4150
4449
  const usage = anthropic.usage;
4151
4450
  logRequest({
@@ -4162,6 +4461,126 @@ async function handleNonClaudeChat(c, opts) {
4162
4461
  return c.json(anthropic, 200);
4163
4462
  }
4164
4463
  //#endregion
4464
+ //#region src/lib/messages-identity-preflight.ts
4465
+ /**
4466
+ * `/v1/messages` identity-preflight bearer, distinct from the `/mcp` nonce
4467
+ * (`Authorization` header). Delivered to the spawned Claude Code process via
4468
+ * `ANTHROPIC_CUSTOM_HEADERS` (an Anthropic SDK env var already carried
4469
+ * through `getClaudeCodeEnvVars`), so it rides on EVERY `/v1/messages`
4470
+ * request the client sends — main-loop turns, subagents, hooks calling the
4471
+ * loopback endpoint directly, all of it.
4472
+ *
4473
+ * A raw BYO client (`start`/`codex`, or any script hitting `/v1/messages`
4474
+ * directly) never sets this header, and that is intentional: header
4475
+ * PRESENCE is what marks a request as asserting a bound-launch identity.
4476
+ * Absence is not a downgrade from some previously-enforced state — it is
4477
+ * today's status quo for every `/v1/messages` caller, preserved exactly.
4478
+ * Only a request that DOES present the header is held to it: if a matching
4479
+ * registry entry can't be found for it (wrong value, launch already torn
4480
+ * down, a claude session that raced this header against a proxy restart),
4481
+ * that specific request fails closed.
4482
+ */
4483
+ const LAUNCH_SECRET_HEADER = "X-GH-Router-Launch-Secret";
4484
+ /**
4485
+ * Validate the launch-secret header BEFORE any body consumer runs (i.e.
4486
+ * before `c.req.text()`/`c.req.json()` — this function only reads a
4487
+ * header). Callers running this must NOT surface a bare 401 on the
4488
+ * `/v1/messages` boundary: this route observes the same no-401 invariant
4489
+ * `forwardError` enforces for upstream failures (Claude Code's reactive
4490
+ * refresh path fires on ANY 401 and would try to use the synthetic
4491
+ * refresh token, breaking the session). Use `identityPreflightErrorResponse`
4492
+ * below, which answers 403, to reject a failed preflight.
4493
+ */
4494
+ function runMessagesIdentityPreflight(c) {
4495
+ const header = c.req.header(LAUNCH_SECRET_HEADER);
4496
+ if (!header) return { ok: true };
4497
+ const launch = findLaunchBySecret(header);
4498
+ if (!launch) return {
4499
+ ok: false,
4500
+ reason: "X-GH-Router-Launch-Secret header did not match any registered launch (the launch may have been restarted, or the header was tampered with)"
4501
+ };
4502
+ return {
4503
+ ok: true,
4504
+ launch
4505
+ };
4506
+ }
4507
+ /**
4508
+ * Anthropic-shaped rejection for a failed identity preflight. 403, never
4509
+ * 401 — see the no-401 invariant note on `runMessagesIdentityPreflight`.
4510
+ */
4511
+ function identityPreflightErrorResponse(c, reason, path = "/v1/messages") {
4512
+ return c.json({
4513
+ type: "error",
4514
+ error: {
4515
+ type: "permission_error",
4516
+ message: `${path} identity preflight rejected: ${reason}`
4517
+ }
4518
+ }, 403);
4519
+ }
4520
+ //#endregion
4521
+ //#region src/lib/fast-request-preprocess.ts
4522
+ /**
4523
+ * Apply authenticated fast-profile model and effort policy before ordinary model
4524
+ * resolution. Synthetic aliases are refused outside an authenticated fast
4525
+ * launch, so raw/BYO traffic cannot opt itself into private profile semantics.
4526
+ */
4527
+ function preprocessFastRequest(rawBody, launch) {
4528
+ let parsed;
4529
+ try {
4530
+ parsed = JSON.parse(rawBody);
4531
+ } catch {
4532
+ return {
4533
+ body: rawBody,
4534
+ modified: false
4535
+ };
4536
+ }
4537
+ const originalModel = typeof parsed.model === "string" ? parsed.model : void 0;
4538
+ if (!originalModel) return {
4539
+ body: rawBody,
4540
+ modified: false
4541
+ };
4542
+ const alias = resolveModelAlias(originalModel);
4543
+ if (alias && launch?.profileId !== "fast") return {
4544
+ body: rawBody,
4545
+ originalModel,
4546
+ modified: false,
4547
+ rejectedAlias: originalModel
4548
+ };
4549
+ if (launch?.profileId !== "fast") return {
4550
+ body: rawBody,
4551
+ originalModel,
4552
+ modified: false
4553
+ };
4554
+ const bare = originalModel.replace(/\[1m\]$/i, "");
4555
+ let effort;
4556
+ if (alias) {
4557
+ effort = alias.absentEffortDefault;
4558
+ parsed.model = canonicalizeAliasModel(originalModel);
4559
+ } else if (bare === "gpt-5.6-luna") effort = "max";
4560
+ else if (bare === "gpt-5.6-sol") effort = "high";
4561
+ else if (bare === "grok-4.6") effort = "medium";
4562
+ else if (bare === "gemini-3.7-flash") effort = "high";
4563
+ else if (bare === "claude-opus-5") effort = "high";
4564
+ if (!effort && !alias) return {
4565
+ body: rawBody,
4566
+ originalModel,
4567
+ modified: false,
4568
+ rejectedModel: originalModel
4569
+ };
4570
+ const outputConfig = parsed.output_config && typeof parsed.output_config === "object" ? parsed.output_config : {};
4571
+ parsed.output_config = {
4572
+ ...outputConfig,
4573
+ effort
4574
+ };
4575
+ const thinking = parsed.thinking;
4576
+ if (thinking && typeof thinking === "object" && thinking.type === "enabled") parsed.thinking = { type: "adaptive" };
4577
+ return {
4578
+ body: JSON.stringify(parsed),
4579
+ originalModel,
4580
+ modified: true
4581
+ };
4582
+ }
4583
+ //#endregion
4165
4584
  //#region src/routes/messages/handler.ts
4166
4585
  const MAX_THINKING_REPAIR_ATTEMPTS = 5;
4167
4586
  const isWebSearchTool$1 = (tool) => !!tool && typeof tool === "object" && (typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search");
@@ -4285,6 +4704,8 @@ async function processWebSearch(rawBody) {
4285
4704
  }
4286
4705
  async function handleCompletion(c) {
4287
4706
  const startTime = Date.now();
4707
+ const identity = runMessagesIdentityPreflight(c);
4708
+ if (!identity.ok) return identityPreflightErrorResponse(c, identity.reason);
4288
4709
  await checkRateLimit(state);
4289
4710
  const rawBody = await c.req.text();
4290
4711
  recordBodySize(rawBody.length);
@@ -4360,7 +4781,18 @@ async function handleCompletion(c) {
4360
4781
  const betaHeaders = extractBetaHeaders(c);
4361
4782
  const incomingBeta = c.req.header("anthropic-beta");
4362
4783
  const advisorEnabled = isAdvisorRequested(incomingBeta);
4363
- let finalBody = await processWebSearch(rawBody);
4784
+ const fastPreprocess = preprocessFastRequest(rawBody, identity.launch);
4785
+ if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
4786
+ const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
4787
+ return c.json({
4788
+ type: "error",
4789
+ error: {
4790
+ type: "invalid_request_error",
4791
+ message
4792
+ }
4793
+ }, 400);
4794
+ }
4795
+ let finalBody = await processWebSearch(fastPreprocess.body);
4364
4796
  finalBody = sanitizeAnthropicBody(finalBody);
4365
4797
  const loopGuard = guardAnthropicBody(finalBody);
4366
4798
  if (loopGuard.action === "abort") return c.json({
@@ -4389,6 +4821,54 @@ async function handleCompletion(c) {
4389
4821
  const modelId = resolvedModel ?? originalModel;
4390
4822
  const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel);
4391
4823
  if (messagesRoute !== "claude-passthrough") {
4824
+ const endpoint = messagesRoute === "chat-shim" ? "chat" : "responses";
4825
+ let parsedBase;
4826
+ try {
4827
+ parsedBase = JSON.parse(resolvedBody);
4828
+ } catch {}
4829
+ const wantsStream = parsedBase?.stream === true;
4830
+ if (advisorEnabled && wantsStream && identity.launch?.profileId === "fast" && isFastProfileLead(modelId)) {
4831
+ const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
4832
+ const parsedInitial = parseAnthropicRequest(parsedBase, modelId, selectedModel);
4833
+ const fastAdvisorAborter = new AbortController();
4834
+ const firstResponse = await streamParsedRequestViaShim(parsedInitial, endpoint, {
4835
+ modelId,
4836
+ model: selectedModel,
4837
+ routePath: c.req.path,
4838
+ onCancel: () => fastAdvisorAborter.abort()
4839
+ }, fastAdvisorAborter.signal);
4840
+ logRequest({
4841
+ method: "POST",
4842
+ path: c.req.path,
4843
+ model: originalModel,
4844
+ resolvedModel: modelId,
4845
+ status: 200,
4846
+ streaming: true
4847
+ }, selectedModel, startTime);
4848
+ const advisorChoice = resolveAdvisorModel(modelId, true);
4849
+ return new Response(buildAdvisorStream({
4850
+ firstResponse,
4851
+ initialConversation,
4852
+ baseBody: parsedBase,
4853
+ requestHeaders: {},
4854
+ advisorModel: advisorChoice.model,
4855
+ advisorEscalated: advisorChoice.escalated || advisorChoice.fastProfile,
4856
+ advisorEffort: resolveAdvisorEffort(rawBody, advisorChoice.model, true),
4857
+ externalAborter: fastAdvisorAborter,
4858
+ continueTurn: makeShimContinueTurn(endpoint, {
4859
+ modelId,
4860
+ model: selectedModel
4861
+ })
4862
+ }), {
4863
+ status: 200,
4864
+ headers: {
4865
+ "content-type": "text/event-stream",
4866
+ "cache-control": "no-cache",
4867
+ "transfer-encoding": "chunked",
4868
+ connection: "keep-alive"
4869
+ }
4870
+ });
4871
+ }
4392
4872
  const shimBody = stripAdvisorTool(resolvedBody);
4393
4873
  if (advisorEnabled) consola.info("ADVISOR requested with a non-Claude model — stripping the injected __anthropic_advisor tool and proceeding without advisor (Claude-only feature; gracefully degraded).");
4394
4874
  const shimOpts = {
@@ -4541,9 +5021,10 @@ function resolveModelInBody$1(rawBody) {
4541
5021
  }
4542
5022
  const originalModel = typeof parsed.model === "string" ? parsed.model : void 0;
4543
5023
  let modified = false;
4544
- if (originalModel) {
4545
- const resolved = resolveModel(originalModel);
4546
- if (resolved !== originalModel) {
5024
+ const resolvedOriginalModel = typeof parsed.model === "string" ? parsed.model : originalModel;
5025
+ if (resolvedOriginalModel) {
5026
+ const resolved = resolveModel(resolvedOriginalModel);
5027
+ if (resolved !== resolvedOriginalModel) {
4547
5028
  parsed.model = resolved;
4548
5029
  modified = true;
4549
5030
  }
@@ -4823,7 +5304,20 @@ function stripWebSearchFromBody(rawBody) {
4823
5304
  */
4824
5305
  async function handleCountTokens(c) {
4825
5306
  const startTime = Date.now();
4826
- const strippedBody = stripWebSearchFromBody(sanitizeAnthropicBody(await c.req.text()));
5307
+ const identity = runMessagesIdentityPreflight(c);
5308
+ if (!identity.ok) return identityPreflightErrorResponse(c, identity.reason, c.req.path);
5309
+ const fastPreprocess = preprocessFastRequest(await c.req.text(), identity.launch);
5310
+ if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
5311
+ const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
5312
+ return c.json({
5313
+ type: "error",
5314
+ error: {
5315
+ type: "invalid_request_error",
5316
+ message
5317
+ }
5318
+ }, 400);
5319
+ }
5320
+ const strippedBody = stripWebSearchFromBody(sanitizeAnthropicBody(fastPreprocess.body));
4827
5321
  if (strippedBody.includes("\"mcp_servers\"")) try {
4828
5322
  const probe = JSON.parse(strippedBody);
4829
5323
  if (Array.isArray(probe.mcp_servers) && probe.mcp_servers.length > 0) return c.json({
@@ -5798,12 +6292,18 @@ function parseSharedArgs(args) {
5798
6292
  }
5799
6293
  /**
5800
6294
  * Non-Claude models we surface as first-class, selectable rows in Claude
5801
- * Code's model picker (Phase 3 of native-non-claude-models). The main
5802
- * agent loop runs on them through the `/v1/messages` translation shim
5803
- * (`src/lib/anthropic-translate/*`, branched in `routes/messages/handler.ts`)
5804
- * that forwards non-Claude targets to Copilot `/responses` (gpt) or
5805
- * `/chat/completions` (gemini). The Gemini review row prefers
5806
- * `gemini-3.1-pro-preview` and degrades to `gemini-3.7-flash`.
6295
+ * Code's model picker (Phase 3 of native-non-claude-models, later
6296
+ * modernized to exactly these four live models by the fast-launch-profile
6297
+ * change). The main agent loop runs on them through the `/v1/messages`
6298
+ * translation shim (`src/lib/anthropic-translate/*`, branched in
6299
+ * `routes/messages/handler.ts`) that forwards non-Claude targets to
6300
+ * Copilot `/responses` (gpt) or `/chat/completions` (gemini/grok).
6301
+ *
6302
+ * This list is EXACT and STATIC — no dynamic Gemini-review append (the
6303
+ * earlier `gemini-3.1-pro-preview`-preferred / `gemini-3.7-flash`-fallback
6304
+ * row is retired: `gemini-3.7-flash` is now a first-class row on its own,
6305
+ * always at this fixed id). A model missing from the catalog is simply
6306
+ * omitted, never substituted — see `nativeSelectableModelsInCatalog`.
5807
6307
  *
5808
6308
  * Display labels only: the gateway-model cache schema Claude Code reads is
5809
6309
  * `{id, display_name?}` per model — there is NO per-model context-window
@@ -5817,48 +6317,52 @@ const NATIVE_NON_CLAUDE_MODELS = [
5817
6317
  displayName: "GPT-5.6 Sol"
5818
6318
  },
5819
6319
  {
5820
- id: "gpt-5.5",
5821
- displayName: "GPT-5.5"
6320
+ id: "gpt-5.6-luna",
6321
+ displayName: "GPT-5.6 Luna"
5822
6322
  },
5823
6323
  {
5824
- id: "gpt-5.3-codex",
5825
- displayName: "GPT-5.3 Codex"
6324
+ id: "gemini-3.7-flash",
6325
+ displayName: "Gemini 3.7 Flash"
5826
6326
  },
5827
6327
  {
5828
- id: "gemini-3.5-flash",
5829
- displayName: "Gemini 3.5 Flash"
6328
+ id: "grok-4.6",
6329
+ displayName: "Grok 4.6"
5830
6330
  }
5831
6331
  ];
5832
6332
  /**
6333
+ * `grok-4.6` never carries `[1m]` — its live-catalog window is 500K total
6334
+ * (372K max prompt), genuinely below the 1M accounting threshold, and this
6335
+ * project deliberately does NOT inject a global
6336
+ * `CLAUDE_CODE_MAX_CONTEXT_TOKENS` override for it (see
6337
+ * `docs/default-models.md` "fast launch profile" once landed): Claude Code
6338
+ * permits arbitrary bare non-Claude ids and runtime `/model` switches, so a
6339
+ * Grok-specific process-global window override would incorrectly follow the
6340
+ * session onto every other model after a switch. Grok is simply left bare,
6341
+ * which is also its true accounting rather than an over- or under-estimate
6342
+ * disguised as one.
6343
+ */
6344
+ const NEVER_1M_MODEL_IDS = /* @__PURE__ */ new Set(["grok-4.6"]);
6345
+ /**
5833
6346
  * The subset of `NATIVE_NON_CLAUDE_MODELS` actually present in the live
5834
- * Copilot catalog. License tiers differ (gpt-5.5 needs
5835
- * pro_plus/business/enterprise/max; gemini-3.5-flash is absent on
5836
- * edu/individual_trial), so a model missing from the catalog is silently
5837
- * dropped the caller then neither enables discovery nor writes a cache
5838
- * for it, and lesser tiers see the unchanged picker. Pure (reads
5839
- * `state.models`), so it is unit-testable without side effects.
6347
+ * Copilot catalog. License tiers differ, so a model missing from the
6348
+ * catalog is silently dropped — the caller then neither enables discovery
6349
+ * nor writes a cache for it, and lesser tiers see the unchanged picker.
6350
+ * Pure (reads `state.models`), so it is unit-testable without side effects.
5840
6351
  *
5841
6352
  * The projected id carries a `[1m]` suffix when the catalog advertises a
5842
- * >=1M window for it, because Claude Code budgets a gateway-discovered row
5843
- * at its 200K default otherwise `gpt-5.6-sol` (1,050,000) would compact at
5844
- * roughly a fifth of its real window. `withOneMSuffix` is catalog-gated, so
5845
- * `gpt-5.3-codex` (400K) stays bare and keeps the conservative accounting;
5846
- * over-budgeting it would trade premature compaction for a hard overflow.
5847
- * The lookup below still keys off the BARE id the decoration is applied
5848
- * only to the value handed to Claude Code.
6353
+ * >=1M window for it AND the id isn't in `NEVER_1M_MODEL_IDS`, because
6354
+ * Claude Code budgets a gateway-discovered row at its 200K default
6355
+ * otherwise. `withOneMSuffix` is catalog-gated on top of that, so a model
6356
+ * whose advertised window shrinks below 1M stays bare regardless. The
6357
+ * lookup below still keys off the BARE id the decoration is applied only
6358
+ * to the value handed to Claude Code.
5849
6359
  */
5850
6360
  function nativeSelectableModelsInCatalog() {
5851
6361
  const catalog = state.models?.data;
5852
6362
  if (!catalog || catalog.length === 0) return [];
5853
6363
  const present = new Set(catalog.map((m) => m.id));
5854
- const models = NATIVE_NON_CLAUDE_MODELS.filter((m) => present.has(m.id));
5855
- const geminiReviewModel = resolveGeminiReviewModel();
5856
- if (geminiReviewModel) models.push({
5857
- id: geminiReviewModel,
5858
- displayName: geminiReviewModel === "gemini-3.7-flash" ? "Gemini 3.7 Flash" : "Gemini 3.1 Pro (preview)"
5859
- });
5860
- return models.map((m) => ({
5861
- id: withOneMSuffix(m.id),
6364
+ return NATIVE_NON_CLAUDE_MODELS.filter((m) => present.has(m.id)).map((m) => ({
6365
+ id: NEVER_1M_MODEL_IDS.has(m.id) ? m.id : withOneMSuffix(m.id),
5862
6366
  display_name: m.displayName
5863
6367
  }));
5864
6368
  }
@@ -5979,7 +6483,19 @@ function clearGatewayModelCache(configDir = PATHS.CLAUDE_CONFIG_DIR) {
5979
6483
  * MUST NOT return 401 on the Anthropic-shape boundary even when
5980
6484
  * upstream Copilot returns 401. See `src/routes/messages/handler.ts`.
5981
6485
  */
5982
- function getClaudeCodeEnvVars(serverUrl, model) {
6486
+ /**
6487
+ * Decorate a Luna-alias id with `[1m]` based on the REAL `gpt-5.6-luna`
6488
+ * catalog entry's advertised window, never on the alias string itself
6489
+ * (which is never a catalog entry — `catalogAdvertises1M`/`resolveModel`
6490
+ * would find nothing and silently leave it bare). Shared by every
6491
+ * fast-profile tier-row seed below so they can't disagree about whether
6492
+ * Luna currently backs 1M.
6493
+ */
6494
+ function oneMSuffixForAlias(aliasId) {
6495
+ if (oneMContextDisabled()) return aliasId;
6496
+ return catalogAdvertises1M("gpt-5.6-luna") ? `${aliasId}[1m]` : aliasId;
6497
+ }
6498
+ function getClaudeCodeEnvVars(serverUrl, model, launchProfileId = "standard") {
5983
6499
  const vars = {
5984
6500
  ANTHROPIC_BASE_URL: serverUrl,
5985
6501
  CLAUDE_CONFIG_DIR: PATHS.CLAUDE_CONFIG_DIR,
@@ -5991,15 +6507,34 @@ function getClaudeCodeEnvVars(serverUrl, model) {
5991
6507
  const mcpToolTimeoutMs = String(resolveMcpToolTimeoutMs());
5992
6508
  if (process.env.MCP_TIMEOUT === void 0) vars.MCP_TIMEOUT = mcpToolTimeoutMs;
5993
6509
  if (process.env.MCP_TOOL_TIMEOUT === void 0) vars.MCP_TOOL_TIMEOUT = mcpToolTimeoutMs;
5994
- const smallFastModel = isBudgetClaudeLead(model) && (state.models?.data?.some((m) => m.id === "claude-haiku-4.5") ?? false) ? BUDGET_SMALL_FAST_SLUG : "claude-sonnet-5";
5995
- if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = smallFastModel;
6510
+ const isFastProfile = launchProfileId === "fast";
6511
+ const smallFastModel = isFastProfile ? LUNA_HAIKU_ALIAS_ID : isBudgetClaudeLead(model) && (state.models?.data?.some((m) => m.id === "claude-haiku-4.5") ?? false) ? BUDGET_SMALL_FAST_SLUG : "claude-sonnet-5";
6512
+ if (process.env.ANTHROPIC_SMALL_FAST_MODEL === void 0) vars.ANTHROPIC_SMALL_FAST_MODEL = isFastProfile ? oneMSuffixForAlias(smallFastModel) : smallFastModel;
5996
6513
  const seedTierRow = (modelKey, nameKey, bareSlug) => {
5997
6514
  if (process.env[modelKey] !== void 0) return;
5998
6515
  vars[modelKey] = withOneMSuffixForLead(bareSlug);
5999
6516
  if (process.env[nameKey] === void 0) vars[nameKey] = bareSlug;
6000
6517
  };
6001
- seedTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", "claude-sonnet-5");
6002
- seedTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", smallFastModel);
6518
+ const seedFastAliasTierRow = (modelKey, nameKey, aliasId, displayName) => {
6519
+ if (process.env[modelKey] !== void 0) return;
6520
+ vars[modelKey] = oneMSuffixForAlias(aliasId);
6521
+ if (process.env[nameKey] === void 0) vars[nameKey] = displayName;
6522
+ };
6523
+ if (isFastProfile) {
6524
+ seedFastAliasTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", LUNA_SONNET_ALIAS_ID, "GPT-5.6 Luna (xhigh)");
6525
+ seedFastAliasTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", LUNA_HAIKU_ALIAS_ID, "GPT-5.6 Luna (high)");
6526
+ const fastAliasCapabilities = "effort,xhigh_effort,max_effort,thinking,adaptive_thinking,interleaved_thinking";
6527
+ if (process.env.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_DEFAULT_SONNET_MODEL_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6528
+ if (process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_DEFAULT_HAIKU_MODEL_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6529
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION === void 0) {
6530
+ vars.ANTHROPIC_CUSTOM_MODEL_OPTION = oneMSuffixForAlias(LUNA_DRIVER_ALIAS_ID);
6531
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION_NAME === void 0) vars.ANTHROPIC_CUSTOM_MODEL_OPTION_NAME = "GPT-5.6 Luna (max)";
6532
+ if (process.env.ANTHROPIC_CUSTOM_MODEL_OPTION_SUPPORTED_CAPABILITIES === void 0) vars.ANTHROPIC_CUSTOM_MODEL_OPTION_SUPPORTED_CAPABILITIES = fastAliasCapabilities;
6533
+ }
6534
+ } else {
6535
+ seedTierRow("ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_SONNET_MODEL_NAME", "claude-sonnet-5");
6536
+ seedTierRow("ANTHROPIC_DEFAULT_HAIKU_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL_NAME", smallFastModel);
6537
+ }
6003
6538
  seedTierRow("ANTHROPIC_DEFAULT_OPUS_MODEL", "ANTHROPIC_DEFAULT_OPUS_MODEL_NAME", "claude-opus-5");
6004
6539
  if (process.env.CLAUDE_CODE_PLAN_V2_AGENT_COUNT === void 0) vars.CLAUDE_CODE_PLAN_V2_AGENT_COUNT = "7";
6005
6540
  for (const key of [
@@ -6038,6 +6573,6 @@ function getCodexEnvVars(serverUrl) {
6038
6573
  return vars;
6039
6574
  }
6040
6575
  //#endregion
6041
- export { sharedServerArgs as a, listModelsForEndpoint as c, checkClaudeVersion as d, updateClaude as f, setupAndServe as i, enableFileLogging as l, getCodexEnvVars as n, startKeepAwake as o, parseSharedArgs as r, stopKeepAwake as s, getClaudeCodeEnvVars as t, runSelfUpdate as u };
6576
+ export { validateFastProfilePrerequisites as _, sharedServerArgs as a, updateClaude as b, stopKeepAwake as c, LUNA_DRIVER_ALIAS_ID as d, LUNA_IMPLEMENTER_ALIAS_ID as f, resolveLaunchProfile as g, profileDescriptor as h, setupAndServe as i, listModelsForEndpoint as l, formatFastPrerequisiteFailure as m, getCodexEnvVars as n, LAUNCH_SECRET_HEADER as o, LUNA_SCOUT_ALIAS_ID as p, parseSharedArgs as r, startKeepAwake as s, getClaudeCodeEnvVars as t, enableFileLogging as u, runSelfUpdate as v, checkClaudeVersion as y };
6042
6577
 
6043
- //# sourceMappingURL=server-setup-DlztZAGT.js.map
6578
+ //# sourceMappingURL=server-setup-CqlaZukJ.js.map