github-router 0.3.293 → 0.3.297
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{attribution-settings-Dx8TBPE8.js → attribution-settings-B70GmZ68.js} +72 -32
- package/dist/attribution-settings-B70GmZ68.js.map +1 -0
- package/dist/{auth-BwUHopJz.js → auth-CluG5e-l.js} +3 -3
- package/dist/{auth-BwUHopJz.js.map → auth-CluG5e-l.js.map} +1 -1
- package/dist/browser-ext/manifest.json +1 -1
- package/dist/{check-usage-BTda5753.js → check-usage-iOzfnKJe.js} +4 -4
- package/dist/{check-usage-BTda5753.js.map → check-usage-iOzfnKJe.js.map} +1 -1
- package/dist/{claude-BoAd_L2W.js → claude-DMMfvfns.js} +225 -61
- package/dist/claude-DMMfvfns.js.map +1 -0
- package/dist/{codex-BdJgGT6Q.js → codex-DVOyVv2B.js} +5 -5
- package/dist/{codex-BdJgGT6Q.js.map → codex-DVOyVv2B.js.map} +1 -1
- package/dist/{debug-B3UrZTHQ.js → debug-DsEwleGg.js} +2 -2
- package/dist/{debug-B3UrZTHQ.js.map → debug-DsEwleGg.js.map} +1 -1
- package/dist/engine-BkGRuUNs.js +2 -0
- package/dist/fast-profile-contract-K3P58nj3.js +57 -0
- package/dist/fast-profile-contract-K3P58nj3.js.map +1 -0
- package/dist/{gate-discovery-C30rfnyo.js → gate-discovery-G6PAG61j.js} +5 -5
- package/dist/{gate-discovery-C30rfnyo.js.map → gate-discovery-G6PAG61j.js.map} +1 -1
- package/dist/{get-copilot-usage-CRrf1ZSC.js → get-copilot-usage-B7Z4XVWj.js} +2 -2
- package/dist/{get-copilot-usage-CRrf1ZSC.js.map → get-copilot-usage-B7Z4XVWj.js.map} +1 -1
- package/dist/hooks.mjs +228 -3
- package/dist/hooks.sha256 +1 -1
- package/dist/{internal-artifact-open-BsEQRvDi.js → internal-artifact-open-D6tKMg8v.js} +2 -2
- package/dist/{internal-artifact-open-BsEQRvDi.js.map → internal-artifact-open-D6tKMg8v.js.map} +1 -1
- package/dist/internal-fast-dispatch-guard-Bkr_iGOu.js +191 -0
- package/dist/internal-fast-dispatch-guard-Bkr_iGOu.js.map +1 -0
- package/dist/internal-fast-dispatch-guard-N-RaLrOn.js +2 -0
- package/dist/{internal-first-mate-guard-CVFFJriA.js → internal-first-mate-guard-C9BV7J4U.js} +3 -3
- package/dist/{internal-first-mate-guard-CVFFJriA.js.map → internal-first-mate-guard-C9BV7J4U.js.map} +1 -1
- package/dist/{internal-first-mate-guard-DJdX6t48.js → internal-first-mate-guard-WJApyc8j.js} +1 -1
- package/dist/{internal-plan-review-8TKDLco6.js → internal-plan-review-t3PuNfuL.js} +3 -3
- package/dist/{internal-plan-review-8TKDLco6.js.map → internal-plan-review-t3PuNfuL.js.map} +1 -1
- package/dist/{internal-prompt-submit-Da7pxqua.js → internal-prompt-submit-C3KovHQG.js} +4 -4
- package/dist/{internal-prompt-submit-Da7pxqua.js.map → internal-prompt-submit-C3KovHQG.js.map} +1 -1
- package/dist/{internal-session-bind-BOFytA1f.js → internal-session-bind-d2uoja8I.js} +2 -2
- package/dist/{internal-session-bind-BOFytA1f.js.map → internal-session-bind-d2uoja8I.js.map} +1 -1
- package/dist/{internal-stop-hook-OnfK3BxE.js → internal-stop-hook-BjzrF1zd.js} +5 -5
- package/dist/{internal-stop-hook-OnfK3BxE.js.map → internal-stop-hook-BjzrF1zd.js.map} +1 -1
- package/dist/{internal-stop-review-CdouacHL.js → internal-stop-review-YtDUGXpX.js} +2 -2
- package/dist/{internal-stop-review-CdouacHL.js.map → internal-stop-review-YtDUGXpX.js.map} +1 -1
- package/dist/{internal-worker-guard-Bx-itoP8.js → internal-worker-guard-CdkeS9PN.js} +2 -2
- package/dist/{internal-worker-guard-Bx-itoP8.js.map → internal-worker-guard-CdkeS9PN.js.map} +1 -1
- package/dist/{internal-workspace-header-8WT0iB5K.js → internal-workspace-header-tuBXJV22.js} +2 -2
- package/dist/{internal-workspace-header-8WT0iB5K.js.map → internal-workspace-header-tuBXJV22.js.map} +1 -1
- package/dist/{lifecycle-nuOHfwgj.js → lifecycle-Bg6doY3-.js} +2 -2
- package/dist/{lifecycle-nuOHfwgj.js.map → lifecycle-Bg6doY3-.js.map} +1 -1
- package/dist/lifecycle-CGLt1cJQ.js +2 -0
- package/dist/{lifecycle-LeSfa7wH.js → lifecycle-CM9eTzvk.js} +2 -2
- package/dist/{lifecycle-LeSfa7wH.js.map → lifecycle-CM9eTzvk.js.map} +1 -1
- package/dist/lifecycle-DoUwpVDB.js +2 -0
- package/dist/main.js +19 -18
- package/dist/main.js.map +1 -1
- package/dist/{mcp-workspace-header-q34H_4wL.js → mcp-workspace-header-dERl2YTT.js} +2 -2
- package/dist/{mcp-workspace-header-q34H_4wL.js.map → mcp-workspace-header-dERl2YTT.js.map} +1 -1
- package/dist/{models-hhJcrZhr.js → models-DZ4hYsR7.js} +3 -3
- package/dist/{models-hhJcrZhr.js.map → models-DZ4hYsR7.js.map} +1 -1
- package/dist/{orchestration-pzbrKkgD.js → orchestration-BiGCEwbH.js} +2 -2
- package/dist/{orchestration-pzbrKkgD.js.map → orchestration-BiGCEwbH.js.map} +1 -1
- package/dist/{paths-BH4J7slC.js → paths-Ci485qSJ.js} +4 -4
- package/dist/{paths-BH4J7slC.js.map → paths-Ci485qSJ.js.map} +1 -1
- package/dist/paths-GD7bgGGy.js +2 -0
- package/dist/{peer-mcp-personas-DhI7ZPSx.js → peer-mcp-personas-ce6_YMnX.js} +407 -96
- package/dist/peer-mcp-personas-ce6_YMnX.js.map +1 -0
- package/dist/{plan-review-hook-CfcanA7_.js → plan-review-hook-DuW0pSoM.js} +3 -3
- package/dist/{plan-review-hook-CfcanA7_.js.map → plan-review-hook-DuW0pSoM.js.map} +1 -1
- package/dist/{prompt-submit-hook-Bqf9ORgb.js → prompt-submit-hook-IMl7ssfr.js} +3 -3
- package/dist/{prompt-submit-hook-Bqf9ORgb.js.map → prompt-submit-hook-IMl7ssfr.js.map} +1 -1
- package/dist/{provision-CZJ4EWls.js → provision-BhCtm3md.js} +4 -4
- package/dist/{provision-CZJ4EWls.js.map → provision-BhCtm3md.js.map} +1 -1
- package/dist/{self-invocation-DhO1Z8iD.js → self-invocation-DAB_od0C.js} +2 -2
- package/dist/{self-invocation-DhO1Z8iD.js.map → self-invocation-DAB_od0C.js.map} +1 -1
- package/dist/{serve-RxWYzoOl.js → serve-Blj_dRSr.js} +12 -12
- package/dist/{serve-RxWYzoOl.js.map → serve-Blj_dRSr.js.map} +1 -1
- package/dist/{server-setup-DbvbW5Ve.js → server-setup-DhG3YZPj.js} +436 -277
- package/dist/server-setup-DhG3YZPj.js.map +1 -0
- package/dist/{start-BNcGsXsY.js → start-8QaqICtd.js} +3 -3
- package/dist/{start-BNcGsXsY.js.map → start-8QaqICtd.js.map} +1 -1
- package/dist/{stop-gate-hook-BiBp5aGm.js → stop-gate-hook-DQnh_KfI.js} +3 -3
- package/dist/{stop-gate-hook-BiBp5aGm.js.map → stop-gate-hook-DQnh_KfI.js.map} +1 -1
- package/dist/{stop-gate-policy-BGd6b5hR.js → stop-gate-policy-DMr3KVmw.js} +2 -2
- package/dist/{stop-gate-policy-BGd6b5hR.js.map → stop-gate-policy-DMr3KVmw.js.map} +1 -1
- package/dist/{token-8drORhXg.js → token-BtJhjXXu.js} +54 -7
- package/dist/token-BtJhjXXu.js.map +1 -0
- package/dist/{worker-dispatch-D5fGroNr.js → worker-dispatch-BEHkTbnH.js} +2 -2
- package/dist/{worker-dispatch-D5fGroNr.js.map → worker-dispatch-BEHkTbnH.js.map} +1 -1
- package/package.json +1 -1
- package/dist/attribution-settings-Dx8TBPE8.js.map +0 -1
- package/dist/claude-BoAd_L2W.js.map +0 -1
- package/dist/engine-CK2b_cTt.js +0 -2
- package/dist/lifecycle-C8fOsQke.js +0 -2
- package/dist/lifecycle-D4Yc1aap.js +0 -2
- package/dist/paths-DJZoXfAS.js +0 -2
- package/dist/peer-mcp-personas-DhI7ZPSx.js.map +0 -1
- package/dist/server-setup-DbvbW5Ve.js.map +0 -1
- package/dist/token-8drORhXg.js.map +0 -1
|
@@ -1,9 +1,10 @@
|
|
|
1
|
-
import { $ as
|
|
1
|
+
import { $ as bucketEffort, An as generateRandomPort, B as resolveAdvisorModel, Bn as withOneMSuffixForLead, Cn as BUDGET_SMALL_FAST_SLUG, E as toolbeltEnabled, F as FAST_ADVISOR_TOOL_INSTRUCTIONS, Fn as upstreamMaxConnections, G as buildAnthropicErrorEvent, Gt as shimDefaultsToXhigh, H as rememberThinkingHistoryRepair, Hn as withInstallLock, I as buildAdvisorStream, In as classifyMessagesRoute, It as LUNA_REAL_MODEL_ID, J as logStreamError, Jt as getTextTokenCount, K as buildOpenAIErrorEvent, Kt as countTokens, L as injectAdvisorTool, Ln as catalogAdvertises1M, M as searchWeb, N as ADVISOR_INTERNAL_TOOL_NAME, Nt as LUNA_DRIVER_ALIAS_ID, On as UPSTREAM_FETCH_TIMEOUT_MS, P as ADVISOR_TOOL_INSTRUCTIONS, Pn as upstreamAllowH2, Pt as LUNA_HAIKU_ALIAS_ID, Q as UNKNOWN_EFFORT_ANCHOR, R as isAdvisorRequested, Rn as oneMContextDisabled, Rt as LUNA_SONNET_ALIAS_ID, U as repairKnownThinkingHistory, Ut as resolveModelAlias, V as formatThinkingRepairDecline, Vn as stripTrailingOneMSuffix, W as repairRejectedThinkingHistory, X as relayAnthropicStream, Xt as getTokenizerFromModel, Y as readIteratorWithTimeout, Yt as getTokenCount, Z as EFFORT_ORDER, Zt as findLaunchBySecret, an as MAX_RESPONSE_BODY_BYTES, cn as normalizeOpenAIUsage, ct as agentToolsEnabled, en as assembleResponsesPayload, et as clampEffort, gn as provisionTreeSitterAssets, i as assertMcpToolSurfaceConsistent, in as createChatCompletions, jn as isBudgetClaudeLead, kn as UPSTREAM_INACTIVITY_TIMEOUT_MS, nn as resolveMcpToolTimeoutMs, nt as handleMcpPost, on as readResponseBodyCapped, q as isControllerClosedError, qt as createMessages, rn as createResponses, sn as parseJsonOrDiagnose, tn as warnOnTokenPriceDrift, tt as handleMcpDelete, xn as toolbeltPathOverride, z as resolveAdvisorEffort, zn as withOneMSuffix, zt as canonicalizeAliasModel } from "./peer-mcp-personas-ce6_YMnX.js";
|
|
2
2
|
import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
|
|
3
|
-
import { i as ensurePaths, t as PATHS } from "./paths-
|
|
4
|
-
import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-
|
|
5
|
-
import { t as getCopilotUsage } from "./get-copilot-usage-
|
|
3
|
+
import { i as ensurePaths, t as PATHS } from "./paths-Ci485qSJ.js";
|
|
4
|
+
import { A as state, T as copilotHeaders, _ as forwardError, a as cacheCopilotVersion, c as filterBetaHeader, d as resolveModel, f as sleep, g as HTTPError, i as tryRefreshAndRetry, l as isNullish, n as setupGitHubAgentToken, o as cacheModels, r as setupGitHubToken, s as cacheVSCodeVersion, t as setupCopilotToken, w as copilotBaseUrl, y as fetchWithTransientRetry } from "./token-BtJhjXXu.js";
|
|
5
|
+
import { t as getCopilotUsage } from "./get-copilot-usage-B7Z4XVWj.js";
|
|
6
6
|
import { a as resolveExecutable, n as killManagedTree, o as runCommandCapture, r as parseBoolEnv, s as runCommandVoid } from "./exec-y8C_MU8A.js";
|
|
7
|
+
import { r as FAST_PROFILE_ADVISOR_MODEL } from "./fast-profile-contract-K3P58nj3.js";
|
|
7
8
|
import consola from "consola";
|
|
8
9
|
import * as fs$2 from "node:fs";
|
|
9
10
|
import fs, { existsSync } from "node:fs";
|
|
@@ -354,232 +355,6 @@ async function runSelfUpdate(opts) {
|
|
|
354
355
|
}
|
|
355
356
|
}
|
|
356
357
|
//#endregion
|
|
357
|
-
//#region src/lib/launch-profile.ts
|
|
358
|
-
const STANDARD_PROFILE = Object.freeze({
|
|
359
|
-
id: "standard",
|
|
360
|
-
hasCoordinator: true
|
|
361
|
-
});
|
|
362
|
-
/**
|
|
363
|
-
* The `-m fast` roster: exactly four native agents (`scout`, `implementer`,
|
|
364
|
-
* `reviewer`, `planner`), the fast-only `oracle` peer tool, no coordinator,
|
|
365
|
-
* and only `peers`/`search` plus the ordinary opt-in `browser` group.
|
|
366
|
-
* `workers`/`orchestrate`/`decide`/`fleet`/`first-mate` are hard denies even
|
|
367
|
-
* when their independent standard-profile gates pass.
|
|
368
|
-
*/
|
|
369
|
-
const FAST_PROFILE = Object.freeze({
|
|
370
|
-
id: "fast",
|
|
371
|
-
nativeRoster: /* @__PURE__ */ new Set([
|
|
372
|
-
"scout",
|
|
373
|
-
"implementer",
|
|
374
|
-
"reviewer",
|
|
375
|
-
"planner"
|
|
376
|
-
]),
|
|
377
|
-
personaAllowlist: /* @__PURE__ */ new Set(["oracle"]),
|
|
378
|
-
allowedGroups: /* @__PURE__ */ new Set([
|
|
379
|
-
"peers",
|
|
380
|
-
"search",
|
|
381
|
-
"browser"
|
|
382
|
-
]),
|
|
383
|
-
hasCoordinator: false
|
|
384
|
-
});
|
|
385
|
-
function profileDescriptor(id) {
|
|
386
|
-
return id === "fast" ? FAST_PROFILE : STANDARD_PROFILE;
|
|
387
|
-
}
|
|
388
|
-
/**
|
|
389
|
-
* Resolve the parsed `-m` argument to a launch profile.
|
|
390
|
-
*
|
|
391
|
-
* Deliberately keyed on the RAW alias string (trimmed, case-insensitive
|
|
392
|
-
* `"fast"`), never on a resolved model id: `resolveLeadSlugArg` maps `fast`
|
|
393
|
-
* to `FAST_LEAD_MODEL` (`./port`) before this is of any use to a caller who
|
|
394
|
-
* only has the resolved id, so callers that already resolved the lead must
|
|
395
|
-
* pass the ORIGINAL `-m` value here, not the resolved one. This is what
|
|
396
|
-
* keeps `-m gpt-5.6-luna` (a direct pin of the same underlying model) a
|
|
397
|
-
* standard-surface launch — only the literal alias narrows the surface.
|
|
398
|
-
*/
|
|
399
|
-
function resolveLaunchProfile(modelArg) {
|
|
400
|
-
return modelArg?.trim().toLowerCase() === "fast" ? "fast" : "standard";
|
|
401
|
-
}
|
|
402
|
-
/**
|
|
403
|
-
* Router-owned alias id for the fast profile's Sonnet-tier row
|
|
404
|
-
* (`ANTHROPIC_DEFAULT_SONNET_MODEL`). Never sent upstream — canonicalized to
|
|
405
|
-
* `LUNA_REAL_MODEL_ID` by `canonicalizeAliasModel` before the request
|
|
406
|
-
* reaches Copilot.
|
|
407
|
-
*/
|
|
408
|
-
const LUNA_DRIVER_ALIAS_ID = "gh-router-luna-driver-max";
|
|
409
|
-
/** Fast native-agent alias ids preserve role-specific effort provenance until
|
|
410
|
-
* the authenticated request boundary. They both canonicalize to Luna, but the
|
|
411
|
-
* scout is fixed high while the implementer is fixed max. */
|
|
412
|
-
const LUNA_SCOUT_ALIAS_ID = "gh-router-luna-scout-high";
|
|
413
|
-
const LUNA_IMPLEMENTER_ALIAS_ID = "gh-router-luna-implementer-max";
|
|
414
|
-
const LUNA_SONNET_ALIAS_ID = "gh-router-luna-sonnet-xhigh";
|
|
415
|
-
/**
|
|
416
|
-
* Router-owned alias id for the fast profile's Haiku-tier row
|
|
417
|
-
* (`ANTHROPIC_DEFAULT_HAIKU_MODEL` / `ANTHROPIC_SMALL_FAST_MODEL`).
|
|
418
|
-
*/
|
|
419
|
-
const LUNA_HAIKU_ALIAS_ID = "gh-router-luna-haiku-high";
|
|
420
|
-
/** The real Copilot catalog id every Luna alias (including the driver
|
|
421
|
-
* itself) canonicalizes to. */
|
|
422
|
-
const LUNA_REAL_MODEL_ID = "gpt-5.6-luna";
|
|
423
|
-
/**
|
|
424
|
-
* The full alias table, keyed by `aliasId`. A simpler model-id-only table is
|
|
425
|
-
* rejected by design: the driver, the Sonnet tier, and the Haiku tier all
|
|
426
|
-
* resolve to the SAME Luna catalog id, so after early canonicalization a
|
|
427
|
-
* table keyed on the real id could no longer tell which absent-effort
|
|
428
|
-
* default applies. Alias provenance — which of the three ids the request
|
|
429
|
-
* actually carried — is the minimum discriminator that survives from tier
|
|
430
|
-
* selection through to request preprocessing, which is why canonicalization
|
|
431
|
-
* must happen LAST (in the `/v1/messages` identity preflight), after the
|
|
432
|
-
* effort default has already been read off the alias.
|
|
433
|
-
*/
|
|
434
|
-
const MODEL_ALIAS_TABLE = /* @__PURE__ */ new Map([
|
|
435
|
-
[LUNA_DRIVER_ALIAS_ID, {
|
|
436
|
-
aliasId: LUNA_DRIVER_ALIAS_ID,
|
|
437
|
-
realModel: LUNA_REAL_MODEL_ID,
|
|
438
|
-
absentEffortDefault: "max"
|
|
439
|
-
}],
|
|
440
|
-
[LUNA_SCOUT_ALIAS_ID, {
|
|
441
|
-
aliasId: LUNA_SCOUT_ALIAS_ID,
|
|
442
|
-
realModel: LUNA_REAL_MODEL_ID,
|
|
443
|
-
absentEffortDefault: "high"
|
|
444
|
-
}],
|
|
445
|
-
[LUNA_IMPLEMENTER_ALIAS_ID, {
|
|
446
|
-
aliasId: LUNA_IMPLEMENTER_ALIAS_ID,
|
|
447
|
-
realModel: LUNA_REAL_MODEL_ID,
|
|
448
|
-
absentEffortDefault: "max"
|
|
449
|
-
}],
|
|
450
|
-
[LUNA_SONNET_ALIAS_ID, {
|
|
451
|
-
aliasId: LUNA_SONNET_ALIAS_ID,
|
|
452
|
-
realModel: LUNA_REAL_MODEL_ID,
|
|
453
|
-
absentEffortDefault: "xhigh"
|
|
454
|
-
}],
|
|
455
|
-
[LUNA_HAIKU_ALIAS_ID, {
|
|
456
|
-
aliasId: LUNA_HAIKU_ALIAS_ID,
|
|
457
|
-
realModel: LUNA_REAL_MODEL_ID,
|
|
458
|
-
absentEffortDefault: "high"
|
|
459
|
-
}]
|
|
460
|
-
]);
|
|
461
|
-
/**
|
|
462
|
-
* Look up the alias descriptor for a wire-facing model id (with or without
|
|
463
|
-
* a trailing `[1m]` bracket — the bracket is stripped before the table
|
|
464
|
-
* lookup and is orthogonal to alias identity). Returns undefined for any
|
|
465
|
-
* id that isn't one of the three registered aliases (including the bare
|
|
466
|
-
* `claude-*` ids and every other real Copilot catalog id).
|
|
467
|
-
*/
|
|
468
|
-
function resolveModelAlias(id) {
|
|
469
|
-
const bare = id.replace(/\[1m\]$/i, "");
|
|
470
|
-
return MODEL_ALIAS_TABLE.get(bare);
|
|
471
|
-
}
|
|
472
|
-
/**
|
|
473
|
-
* Strip alias provenance and return the real catalog id to send upstream.
|
|
474
|
-
* Idempotent passthrough for any id that isn't a registered alias (a bare
|
|
475
|
-
* `claude-*` slug, an already-real Copilot id, or anything else) — this is
|
|
476
|
-
* safe to call unconditionally on every `body.model` at the outbound
|
|
477
|
-
* boundary. Preserves a trailing `[1m]` bracket: canonicalization only
|
|
478
|
-
* erases ALIAS identity, not the 1M-context accounting decoration.
|
|
479
|
-
*/
|
|
480
|
-
function canonicalizeAliasModel(id) {
|
|
481
|
-
const bracket = /\[1m\]$/i.test(id) ? "[1m]" : "";
|
|
482
|
-
const bare = bracket ? id.slice(0, -bracket.length) : id;
|
|
483
|
-
const alias = MODEL_ALIAS_TABLE.get(bare);
|
|
484
|
-
return alias ? `${alias.realModel}${bracket}` : id;
|
|
485
|
-
}
|
|
486
|
-
const FAST_REQUIRED_CONTEXT_TOKENS = 1e6;
|
|
487
|
-
function findModel(catalog, id) {
|
|
488
|
-
return catalog?.data?.find((m) => m.id === id);
|
|
489
|
-
}
|
|
490
|
-
function hasToolCalls(model) {
|
|
491
|
-
return model?.capabilities?.supports?.tool_calls === true;
|
|
492
|
-
}
|
|
493
|
-
function hasContextAtLeast(model, tokens) {
|
|
494
|
-
return (model?.capabilities?.limits?.max_context_window_tokens ?? 0) >= tokens;
|
|
495
|
-
}
|
|
496
|
-
function supportsEffort(model, effort) {
|
|
497
|
-
const list = model?.capabilities?.supports?.reasoning_effort;
|
|
498
|
-
return Array.isArray(list) && list.includes(effort);
|
|
499
|
-
}
|
|
500
|
-
function supportsEndpoint(model, paths) {
|
|
501
|
-
const endpoints = model?.supported_endpoints;
|
|
502
|
-
return Array.isArray(endpoints) && endpoints.some((endpoint) => paths.has(endpoint));
|
|
503
|
-
}
|
|
504
|
-
const RESPONSES_ENDPOINTS = /* @__PURE__ */ new Set(["/responses", "/v1/responses"]);
|
|
505
|
-
const MESSAGES_ENDPOINTS = /* @__PURE__ */ new Set(["/messages", "/v1/messages"]);
|
|
506
|
-
function hasUsablePromptMetadata(model) {
|
|
507
|
-
const prompt = model?.capabilities?.limits?.max_prompt_tokens;
|
|
508
|
-
return typeof prompt === "number" && Number.isFinite(prompt) && prompt > 0;
|
|
509
|
-
}
|
|
510
|
-
/**
|
|
511
|
-
* Validate the live Copilot catalog carries every model the fast profile's
|
|
512
|
-
* EXACT roster depends on, with the specific capabilities each assignment
|
|
513
|
-
* needs. These are capability-availability PREREQUISITES for constructing
|
|
514
|
-
* the roster — not an allowlist of models the user may select later in the
|
|
515
|
-
* session — so a partial catalog fails the whole `-m fast` launch rather
|
|
516
|
-
* than silently substituting or dropping an agent.
|
|
517
|
-
*
|
|
518
|
-
* Checks, per the fast-launch-profile design:
|
|
519
|
-
* - Luna lead/scout/implementer: tool calls, >=1M, high+max, Responses.
|
|
520
|
-
* - Sol planner: tool calls, >=1M, high, Responses.
|
|
521
|
-
* - Grok reviewer: tool calls, medium, Responses, usable prompt metadata.
|
|
522
|
-
* - Gemini Advisor: >=1M, high, chat-completions.
|
|
523
|
-
* - Opus Oracle: exact Opus 5, >=1M, adaptive/high, Messages, prompt metadata.
|
|
524
|
-
*
|
|
525
|
-
* Pure over the passed-in catalog snapshot so it's unit-testable without
|
|
526
|
-
* `state` — callers pass `state.models` at call time.
|
|
527
|
-
*/
|
|
528
|
-
function validateFastProfilePrerequisites(catalog) {
|
|
529
|
-
const missing = [];
|
|
530
|
-
const luna = findModel(catalog, LUNA_REAL_MODEL_ID);
|
|
531
|
-
if (!luna) missing.push(`${LUNA_REAL_MODEL_ID}: absent from the live catalog`);
|
|
532
|
-
else {
|
|
533
|
-
if (!hasToolCalls(luna)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise tool_calls`);
|
|
534
|
-
if (!hasContextAtLeast(luna, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push(`${LUNA_REAL_MODEL_ID}: advertised context window is below 1M`);
|
|
535
|
-
if (!supportsEffort(luna, "high") || !supportsEffort(luna, "max")) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise both "high" and "max" reasoning effort`);
|
|
536
|
-
if (!supportsEndpoint(luna, RESPONSES_ENDPOINTS)) missing.push(`${LUNA_REAL_MODEL_ID}: does not advertise a supported Responses endpoint`);
|
|
537
|
-
}
|
|
538
|
-
const sol = findModel(catalog, "gpt-5.6-sol");
|
|
539
|
-
if (!sol) missing.push("gpt-5.6-sol: absent from the live catalog");
|
|
540
|
-
else {
|
|
541
|
-
if (!hasToolCalls(sol)) missing.push("gpt-5.6-sol: does not advertise tool_calls");
|
|
542
|
-
if (!hasContextAtLeast(sol, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gpt-5.6-sol: advertised context window is below 1M");
|
|
543
|
-
if (!supportsEffort(sol, "high")) missing.push("gpt-5.6-sol: does not advertise a \"high\" reasoning effort");
|
|
544
|
-
if (!supportsEndpoint(sol, RESPONSES_ENDPOINTS)) missing.push("gpt-5.6-sol: does not advertise a supported Responses endpoint");
|
|
545
|
-
}
|
|
546
|
-
const grok = findModel(catalog, "grok-4.6");
|
|
547
|
-
if (!grok) missing.push("grok-4.6: absent from the live catalog");
|
|
548
|
-
else {
|
|
549
|
-
if (!hasToolCalls(grok)) missing.push("grok-4.6: does not advertise tool_calls");
|
|
550
|
-
if (!supportsEffort(grok, "medium")) missing.push("grok-4.6: does not advertise a \"medium\" reasoning effort");
|
|
551
|
-
if (!hasUsablePromptMetadata(grok)) missing.push("grok-4.6: no usable max_prompt_tokens metadata");
|
|
552
|
-
if (!supportsEndpoint(grok, RESPONSES_ENDPOINTS)) missing.push("grok-4.6: does not advertise a supported Responses endpoint");
|
|
553
|
-
}
|
|
554
|
-
const gemini = findModel(catalog, "gemini-3.7-flash");
|
|
555
|
-
if (!gemini) missing.push("gemini-3.7-flash: absent from the live catalog");
|
|
556
|
-
else {
|
|
557
|
-
if (!hasContextAtLeast(gemini, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("gemini-3.7-flash: advertised context window is below 1M");
|
|
558
|
-
if (!supportsEffort(gemini, "high")) missing.push("gemini-3.7-flash: does not advertise a \"high\" reasoning effort");
|
|
559
|
-
if (pickEndpoint(gemini) !== "chat") missing.push("gemini-3.7-flash: does not advertise a supported chat-completions endpoint");
|
|
560
|
-
}
|
|
561
|
-
const opus = findModel(catalog, "claude-opus-5");
|
|
562
|
-
if (!opus) missing.push("claude-opus-5: absent from the live catalog");
|
|
563
|
-
else {
|
|
564
|
-
if (!hasContextAtLeast(opus, FAST_REQUIRED_CONTEXT_TOKENS)) missing.push("claude-opus-5: advertised context window is below 1M");
|
|
565
|
-
if (!supportsEffort(opus, "high")) missing.push("claude-opus-5: does not advertise a \"high\" reasoning effort");
|
|
566
|
-
if (opus.capabilities?.supports?.adaptive_thinking !== true) missing.push("claude-opus-5: does not advertise adaptive_thinking");
|
|
567
|
-
if (!hasUsablePromptMetadata(opus)) missing.push("claude-opus-5: no usable max_prompt_tokens metadata");
|
|
568
|
-
if (!supportsEndpoint(opus, MESSAGES_ENDPOINTS)) missing.push("claude-opus-5: does not advertise a supported Messages endpoint");
|
|
569
|
-
}
|
|
570
|
-
return {
|
|
571
|
-
ok: missing.length === 0,
|
|
572
|
-
missing
|
|
573
|
-
};
|
|
574
|
-
}
|
|
575
|
-
/**
|
|
576
|
-
* Format `validateFastProfilePrerequisites`'s failure list into the launch
|
|
577
|
-
* error message: every missing/invalid model, plus the rollback command.
|
|
578
|
-
*/
|
|
579
|
-
function formatFastPrerequisiteFailure(missing) {
|
|
580
|
-
return "github-router claude -m fast requires the following live-catalog capabilities, which this account's catalog does not fully provide:\n" + missing.map((m) => ` - ${m}`).join("\n") + "\n\nFalling back or silently dropping an agent is not supported for the fast profile's exact roster. Run plain `github-router claude` instead.";
|
|
581
|
-
}
|
|
582
|
-
//#endregion
|
|
583
358
|
//#region src/lib/file-log-reporter.ts
|
|
584
359
|
const MAX_LOG_BYTES = 1048576;
|
|
585
360
|
const DEDUP_MAX = 1e3;
|
|
@@ -1306,6 +1081,73 @@ function wireDaemonTeardown(handle, opts = {}) {
|
|
|
1306
1081
|
});
|
|
1307
1082
|
}
|
|
1308
1083
|
//#endregion
|
|
1084
|
+
//#region src/lib/grok-context.ts
|
|
1085
|
+
/** Claude Code's fixed output-token reserve when computing its own
|
|
1086
|
+
* compaction-window assumption. Grok 4.6's 128K max output stays well
|
|
1087
|
+
* above this, so the reserve — not the model's real output ceiling — is
|
|
1088
|
+
* what caps the second term. */
|
|
1089
|
+
const CLIENT_OUTPUT_RESERVE_TOKENS = 2e4;
|
|
1090
|
+
/**
|
|
1091
|
+
* The client's additional flat reserve between the output-reduced window and
|
|
1092
|
+
* its reactive compaction threshold. Read from the installed Claude Code
|
|
1093
|
+
* 2.1.247 bundle: `xWe(e,t)` computes `n = e - 13000` and returns it when no
|
|
1094
|
+
* percentage override is set, where `e` is already
|
|
1095
|
+
* `window - min(maxOutput, 20_000)`.
|
|
1096
|
+
*/
|
|
1097
|
+
const CLIENT_THRESHOLD_RESERVE_TOKENS = 13e3;
|
|
1098
|
+
/**
|
|
1099
|
+
* Bounds the `CLAUDE_CODE_AUTO_COMPACT_WINDOW` env path applies to whatever it
|
|
1100
|
+
* parses: `IWe = 1e5` floor, `ZRt = 1e6` cap. A value below the floor is
|
|
1101
|
+
* silently RAISED to it, so deriving something smaller would be a lie; a value
|
|
1102
|
+
* above the cap is clamped. Both read from the same 2.1.247 bundle.
|
|
1103
|
+
*/
|
|
1104
|
+
const ENV_WINDOW_MIN_TOKENS = 1e5;
|
|
1105
|
+
const ENV_WINDOW_MAX_TOKENS = 1e6;
|
|
1106
|
+
/**
|
|
1107
|
+
* Derive the conservative 85%-of-prompt compaction trigger and the client
|
|
1108
|
+
* window that would need to be assumed to make that trigger point natural.
|
|
1109
|
+
*
|
|
1110
|
+
* `target trigger = floor(max_prompt_tokens * 0.85)`
|
|
1111
|
+
* `assumed client window = target trigger + min(max_output_tokens, 20_000)`
|
|
1112
|
+
*
|
|
1113
|
+
* For Grok 4.6's current catalog entry (max_prompt_tokens=372_000,
|
|
1114
|
+
* max_output_tokens=128_000) this yields `{ triggerTokens: 316_200,
|
|
1115
|
+
* assumedClientWindowTokens: 336_200 }` — the exact figures the plan
|
|
1116
|
+
* document records. Pure and side-effect-free; callers are responsible for
|
|
1117
|
+
* sourcing `maxPromptTokens`/`maxOutputTokens` from the live catalog.
|
|
1118
|
+
*/
|
|
1119
|
+
function computeConservativeCompactionTrigger(maxPromptTokens, maxOutputTokens) {
|
|
1120
|
+
const triggerTokens = Math.floor(maxPromptTokens * .85);
|
|
1121
|
+
return {
|
|
1122
|
+
triggerTokens,
|
|
1123
|
+
assumedClientWindowTokens: triggerTokens + Math.min(maxOutputTokens, CLIENT_OUTPUT_RESERVE_TOKENS)
|
|
1124
|
+
};
|
|
1125
|
+
}
|
|
1126
|
+
/**
|
|
1127
|
+
* The integer `CLAUDE_CODE_AUTO_COMPACT_WINDOW` value that puts the client's
|
|
1128
|
+
* REACTIVE compaction trigger at 85% of the provider's real prompt ceiling.
|
|
1129
|
+
*
|
|
1130
|
+
* The client computes its trigger as
|
|
1131
|
+
* `window - min(maxOutput, 20_000) - 13_000`, so we invert that:
|
|
1132
|
+
* `window = assumedClientWindowTokens + 13_000`.
|
|
1133
|
+
*
|
|
1134
|
+
* Worked, against the live catalog at time of writing:
|
|
1135
|
+
* Luna 922_000 prompt / 128_000 out -> trigger 783_700, window 816_700
|
|
1136
|
+
* Opus 5 936_000 prompt / 64_000 out -> trigger 795_600, window 828_600
|
|
1137
|
+
*
|
|
1138
|
+
* Returns undefined when the catalog metadata is missing or nonsensical, so
|
|
1139
|
+
* the caller omits the variable entirely rather than exporting a guess. That
|
|
1140
|
+
* degrades to today's behaviour (client budgets against the total window)
|
|
1141
|
+
* instead of to a wrong number, which is the safer direction for a value the
|
|
1142
|
+
* client silently floors.
|
|
1143
|
+
*/
|
|
1144
|
+
function deriveAutoCompactWindowTokens(maxPromptTokens, maxOutputTokens) {
|
|
1145
|
+
if (typeof maxPromptTokens !== "number" || !Number.isFinite(maxPromptTokens) || maxPromptTokens <= 0) return;
|
|
1146
|
+
const { assumedClientWindowTokens } = computeConservativeCompactionTrigger(maxPromptTokens, typeof maxOutputTokens === "number" && Number.isFinite(maxOutputTokens) && maxOutputTokens > 0 ? maxOutputTokens : CLIENT_OUTPUT_RESERVE_TOKENS);
|
|
1147
|
+
const window = assumedClientWindowTokens + CLIENT_THRESHOLD_RESERVE_TOKENS;
|
|
1148
|
+
return Math.min(ENV_WINDOW_MAX_TOKENS, Math.max(ENV_WINDOW_MIN_TOKENS, window));
|
|
1149
|
+
}
|
|
1150
|
+
//#endregion
|
|
1309
1151
|
//#region src/lib/proxy.ts
|
|
1310
1152
|
let socketCounter = 0;
|
|
1311
1153
|
/**
|
|
@@ -1656,7 +1498,7 @@ function collectToolFieldKeys(body) {
|
|
|
1656
1498
|
//#endregion
|
|
1657
1499
|
//#region package.json
|
|
1658
1500
|
var name = "github-router";
|
|
1659
|
-
var version = "0.3.
|
|
1501
|
+
var version = "0.3.297";
|
|
1660
1502
|
//#endregion
|
|
1661
1503
|
//#region src/lib/approval.ts
|
|
1662
1504
|
const awaitApproval = async () => {
|
|
@@ -1758,7 +1600,7 @@ const NO_LOOP = {
|
|
|
1758
1600
|
tier: null,
|
|
1759
1601
|
repeats: 0
|
|
1760
1602
|
};
|
|
1761
|
-
function isRecord(value) {
|
|
1603
|
+
function isRecord$1(value) {
|
|
1762
1604
|
return typeof value === "object" && value !== null;
|
|
1763
1605
|
}
|
|
1764
1606
|
function digest(value) {
|
|
@@ -1788,7 +1630,7 @@ function normalizeResultBody(content) {
|
|
|
1788
1630
|
}
|
|
1789
1631
|
function normalizeResultBlock(block) {
|
|
1790
1632
|
if (typeof block === "string") return `s:${boundedText(block)}`;
|
|
1791
|
-
if (!isRecord(block)) return `j:${boundedText(safeJson(block))}`;
|
|
1633
|
+
if (!isRecord$1(block)) return `j:${boundedText(safeJson(block))}`;
|
|
1792
1634
|
const type = typeof block.type === "string" ? block.type : "unknown";
|
|
1793
1635
|
if (type === "text" && typeof block.text === "string") return `t:${boundedText(block.text)}`;
|
|
1794
1636
|
return `${type}:${digest(safeJson(block))}`;
|
|
@@ -1899,15 +1741,15 @@ function mayContainToolTraffic(rawBody) {
|
|
|
1899
1741
|
return rawBody.includes("tool_result") || rawBody.includes("tool_call_id") || rawBody.includes("function_call_output");
|
|
1900
1742
|
}
|
|
1901
1743
|
function blockType(block) {
|
|
1902
|
-
return isRecord(block) && typeof block.type === "string" ? block.type : void 0;
|
|
1744
|
+
return isRecord$1(block) && typeof block.type === "string" ? block.type : void 0;
|
|
1903
1745
|
}
|
|
1904
1746
|
/** Anthropic Messages: assistant `tool_use` blocks ↔ user `tool_result` blocks. */
|
|
1905
1747
|
function extractAnthropicTurns(body) {
|
|
1906
|
-
if (!isRecord(body) || !Array.isArray(body.messages)) return [];
|
|
1748
|
+
if (!isRecord$1(body) || !Array.isArray(body.messages)) return [];
|
|
1907
1749
|
const turns = [];
|
|
1908
1750
|
for (let i = 0; i < body.messages.length; i++) {
|
|
1909
1751
|
const message = body.messages[i];
|
|
1910
|
-
if (!isRecord(message) || message.role !== "assistant") continue;
|
|
1752
|
+
if (!isRecord$1(message) || message.role !== "assistant") continue;
|
|
1911
1753
|
if (!Array.isArray(message.content)) continue;
|
|
1912
1754
|
const uses = message.content.filter((b) => blockType(b) === "tool_use");
|
|
1913
1755
|
if (uses.length === 0) {
|
|
@@ -1920,17 +1762,17 @@ function extractAnthropicTurns(body) {
|
|
|
1920
1762
|
const results = /* @__PURE__ */ new Map();
|
|
1921
1763
|
for (let j = i + 1; j < body.messages.length; j++) {
|
|
1922
1764
|
const next = body.messages[j];
|
|
1923
|
-
if (!isRecord(next) || next.role !== "user") break;
|
|
1765
|
+
if (!isRecord$1(next) || next.role !== "user") break;
|
|
1924
1766
|
if (!Array.isArray(next.content)) break;
|
|
1925
1767
|
for (const block of next.content) {
|
|
1926
|
-
if (!isRecord(block) || blockType(block) !== "tool_result") continue;
|
|
1768
|
+
if (!isRecord$1(block) || blockType(block) !== "tool_result") continue;
|
|
1927
1769
|
const id = block.tool_use_id;
|
|
1928
1770
|
if (typeof id === "string") results.set(id, block);
|
|
1929
1771
|
}
|
|
1930
1772
|
}
|
|
1931
1773
|
const calls = [];
|
|
1932
1774
|
for (const use of uses) {
|
|
1933
|
-
if (!isRecord(use)) continue;
|
|
1775
|
+
if (!isRecord$1(use)) continue;
|
|
1934
1776
|
const name = typeof use.name === "string" ? use.name : "unknown";
|
|
1935
1777
|
const id = typeof use.id === "string" ? use.id : void 0;
|
|
1936
1778
|
const result = id === void 0 ? void 0 : results.get(id);
|
|
@@ -1949,16 +1791,16 @@ function anthropicHasNarration(content) {
|
|
|
1949
1791
|
return content.some((block) => {
|
|
1950
1792
|
const type = blockType(block);
|
|
1951
1793
|
if (type === "thinking" || type === "redacted_thinking") return true;
|
|
1952
|
-
return type === "text" && isRecord(block) && typeof block.text === "string" && block.text.trim() !== "";
|
|
1794
|
+
return type === "text" && isRecord$1(block) && typeof block.text === "string" && block.text.trim() !== "";
|
|
1953
1795
|
});
|
|
1954
1796
|
}
|
|
1955
1797
|
/** OpenAI Chat Completions: assistant `tool_calls[]` ↔ `role:"tool"` messages. */
|
|
1956
1798
|
function extractChatTurns(body) {
|
|
1957
|
-
if (!isRecord(body) || !Array.isArray(body.messages)) return [];
|
|
1799
|
+
if (!isRecord$1(body) || !Array.isArray(body.messages)) return [];
|
|
1958
1800
|
const turns = [];
|
|
1959
1801
|
for (let i = 0; i < body.messages.length; i++) {
|
|
1960
1802
|
const message = body.messages[i];
|
|
1961
|
-
if (!isRecord(message) || message.role !== "assistant") continue;
|
|
1803
|
+
if (!isRecord$1(message) || message.role !== "assistant") continue;
|
|
1962
1804
|
if (!Array.isArray(message.tool_calls) || message.tool_calls.length === 0) {
|
|
1963
1805
|
turns.push({
|
|
1964
1806
|
calls: [],
|
|
@@ -1969,14 +1811,14 @@ function extractChatTurns(body) {
|
|
|
1969
1811
|
const results = /* @__PURE__ */ new Map();
|
|
1970
1812
|
for (let j = i + 1; j < body.messages.length; j++) {
|
|
1971
1813
|
const next = body.messages[j];
|
|
1972
|
-
if (!isRecord(next) || next.role !== "tool") break;
|
|
1814
|
+
if (!isRecord$1(next) || next.role !== "tool") break;
|
|
1973
1815
|
const id = next.tool_call_id;
|
|
1974
1816
|
if (typeof id === "string") results.set(id, next);
|
|
1975
1817
|
}
|
|
1976
1818
|
const calls = [];
|
|
1977
1819
|
for (const call of message.tool_calls) {
|
|
1978
|
-
if (!isRecord(call)) continue;
|
|
1979
|
-
const fn = isRecord(call.function) ? call.function : void 0;
|
|
1820
|
+
if (!isRecord$1(call)) continue;
|
|
1821
|
+
const fn = isRecord$1(call.function) ? call.function : void 0;
|
|
1980
1822
|
const name = typeof fn?.name === "string" ? fn.name : "unknown";
|
|
1981
1823
|
const id = typeof call.id === "string" ? call.id : void 0;
|
|
1982
1824
|
const result = id === void 0 ? void 0 : results.get(id);
|
|
@@ -1999,15 +1841,15 @@ function extractChatTurns(body) {
|
|
|
1999
1841
|
function chatHasNarration(content) {
|
|
2000
1842
|
if (typeof content === "string") return content.trim() !== "";
|
|
2001
1843
|
if (!Array.isArray(content)) return false;
|
|
2002
|
-
return content.some((part) => isRecord(part) && typeof part.text === "string" && part.text.trim() !== "");
|
|
1844
|
+
return content.some((part) => isRecord$1(part) && typeof part.text === "string" && part.text.trim() !== "");
|
|
2003
1845
|
}
|
|
2004
1846
|
/** OpenAI Responses: `function_call` items ↔ `function_call_output` items. */
|
|
2005
1847
|
function extractResponsesTurns(body) {
|
|
2006
|
-
if (!isRecord(body) || !Array.isArray(body.input)) return [];
|
|
1848
|
+
if (!isRecord$1(body) || !Array.isArray(body.input)) return [];
|
|
2007
1849
|
const items = body.input;
|
|
2008
1850
|
const results = /* @__PURE__ */ new Map();
|
|
2009
1851
|
for (const item of items) {
|
|
2010
|
-
if (!isRecord(item) || item.type !== "function_call_output") continue;
|
|
1852
|
+
if (!isRecord$1(item) || item.type !== "function_call_output") continue;
|
|
2011
1853
|
const id = item.call_id;
|
|
2012
1854
|
if (typeof id === "string") results.set(id, item);
|
|
2013
1855
|
}
|
|
@@ -2015,14 +1857,14 @@ function extractResponsesTurns(body) {
|
|
|
2015
1857
|
let i = 0;
|
|
2016
1858
|
while (i < items.length) {
|
|
2017
1859
|
const item = items[i];
|
|
2018
|
-
if (!isRecord(item) || item.type !== "function_call") {
|
|
1860
|
+
if (!isRecord$1(item) || item.type !== "function_call") {
|
|
2019
1861
|
i++;
|
|
2020
1862
|
continue;
|
|
2021
1863
|
}
|
|
2022
1864
|
const batch = [];
|
|
2023
1865
|
while (i < items.length) {
|
|
2024
1866
|
const candidate = items[i];
|
|
2025
|
-
if (!isRecord(candidate) || candidate.type !== "function_call") break;
|
|
1867
|
+
if (!isRecord$1(candidate) || candidate.type !== "function_call") break;
|
|
2026
1868
|
batch.push(candidate);
|
|
2027
1869
|
i++;
|
|
2028
1870
|
}
|
|
@@ -2050,7 +1892,7 @@ function extractResponsesTurns(body) {
|
|
|
2050
1892
|
function responsesHasNarration(items, batchStart) {
|
|
2051
1893
|
for (let i = batchStart - 1; i >= 0; i--) {
|
|
2052
1894
|
const item = items[i];
|
|
2053
|
-
if (!isRecord(item)) return false;
|
|
1895
|
+
if (!isRecord$1(item)) return false;
|
|
2054
1896
|
if (item.type === "function_call_output") continue;
|
|
2055
1897
|
if (item.type === "reasoning") return true;
|
|
2056
1898
|
if (item.type === "message" || item.role === "assistant") return responsesItemHasText(item);
|
|
@@ -2061,13 +1903,13 @@ function responsesHasNarration(items, batchStart) {
|
|
|
2061
1903
|
function responsesItemHasText(item) {
|
|
2062
1904
|
if (typeof item.content === "string") return item.content.trim() !== "";
|
|
2063
1905
|
if (!Array.isArray(item.content)) return false;
|
|
2064
|
-
return item.content.some((block) => isRecord(block) && typeof block.text === "string" && block.text.trim() !== "");
|
|
1906
|
+
return item.content.some((block) => isRecord$1(block) && typeof block.text === "string" && block.text.trim() !== "");
|
|
2065
1907
|
}
|
|
2066
1908
|
function injectAnthropicNudge(body, text) {
|
|
2067
|
-
if (!isRecord(body) || !Array.isArray(body.messages)) return false;
|
|
1909
|
+
if (!isRecord$1(body) || !Array.isArray(body.messages)) return false;
|
|
2068
1910
|
const messages = body.messages;
|
|
2069
1911
|
const last = messages[messages.length - 1];
|
|
2070
|
-
if (!isRecord(last) || last.role !== "user") return false;
|
|
1912
|
+
if (!isRecord$1(last) || last.role !== "user") return false;
|
|
2071
1913
|
if (!Array.isArray(last.content)) return false;
|
|
2072
1914
|
messages[messages.length - 1] = {
|
|
2073
1915
|
...last,
|
|
@@ -2079,7 +1921,7 @@ function injectAnthropicNudge(body, text) {
|
|
|
2079
1921
|
return true;
|
|
2080
1922
|
}
|
|
2081
1923
|
function injectChatNudge(body, text) {
|
|
2082
|
-
if (!isRecord(body) || !Array.isArray(body.messages)) return false;
|
|
1924
|
+
if (!isRecord$1(body) || !Array.isArray(body.messages)) return false;
|
|
2083
1925
|
body.messages = [...body.messages, {
|
|
2084
1926
|
role: "user",
|
|
2085
1927
|
content: text
|
|
@@ -2087,7 +1929,7 @@ function injectChatNudge(body, text) {
|
|
|
2087
1929
|
return true;
|
|
2088
1930
|
}
|
|
2089
1931
|
function injectResponsesNudge(body, text) {
|
|
2090
|
-
if (!isRecord(body) || !Array.isArray(body.input)) return false;
|
|
1932
|
+
if (!isRecord$1(body) || !Array.isArray(body.input)) return false;
|
|
2091
1933
|
body.input = [...body.input, {
|
|
2092
1934
|
role: "user",
|
|
2093
1935
|
content: [{
|
|
@@ -4300,10 +4142,10 @@ function isAsyncIterable(x) {
|
|
|
4300
4142
|
* from the catalog (which would also mean re-parsing/re-picking work the
|
|
4301
4143
|
* caller already did).
|
|
4302
4144
|
* - `makeShimContinueTurn` (below), which `buildAdvisorStream`
|
|
4303
|
-
* (`src/services/advisor/advisor.ts`) injects
|
|
4304
|
-
*
|
|
4305
|
-
*
|
|
4306
|
-
* the initial turn instead of a parallel, divergent implementation.
|
|
4145
|
+
* (`src/services/advisor/advisor.ts`) injects for any non-Claude model
|
|
4146
|
+
* selected by an authenticated fast primary lead, so its Advisor
|
|
4147
|
+
* continuation runs through the SAME translation + SSE-synthesis machinery
|
|
4148
|
+
* as the initial turn instead of a parallel, divergent implementation.
|
|
4307
4149
|
*/
|
|
4308
4150
|
async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
|
|
4309
4151
|
const routePath = opts.routePath ?? "/v1/messages (advisor lead shim)";
|
|
@@ -4338,8 +4180,8 @@ async function streamParsedRequestViaShim(parsed, endpoint, opts, signal) {
|
|
|
4338
4180
|
* Build an injectable `continueTurn(body, signal)` for `buildAdvisorStream`
|
|
4339
4181
|
* (`src/services/advisor/advisor.ts`) that routes a continuation turn
|
|
4340
4182
|
* through THIS module's non-Claude shim instead of Claude passthrough — used
|
|
4341
|
-
*
|
|
4342
|
-
* threaded through: the advisor loop's own `aborter` (shared with `signal`
|
|
4183
|
+
* by any non-Claude model selected in an authenticated fast primary lead.
|
|
4184
|
+
* No `onCancel` is threaded through: the advisor loop's own `aborter` (shared with `signal`
|
|
4343
4185
|
* here) already tears down on consumer cancel via `buildAdvisorStream`'s
|
|
4344
4186
|
* `cancel()`, so this stream needs no independent teardown hook.
|
|
4345
4187
|
*/
|
|
@@ -4551,7 +4393,7 @@ function preprocessFastRequest(rawBody, launch) {
|
|
|
4551
4393
|
originalModel,
|
|
4552
4394
|
modified: false
|
|
4553
4395
|
};
|
|
4554
|
-
const bare = originalModel
|
|
4396
|
+
const { base: bare } = stripTrailingOneMSuffix(originalModel);
|
|
4555
4397
|
let effort;
|
|
4556
4398
|
if (alias) {
|
|
4557
4399
|
effort = alias.absentEffortDefault;
|
|
@@ -4581,6 +4423,216 @@ function preprocessFastRequest(rawBody, launch) {
|
|
|
4581
4423
|
};
|
|
4582
4424
|
}
|
|
4583
4425
|
//#endregion
|
|
4426
|
+
//#region src/lib/prompt-window-salvage.ts
|
|
4427
|
+
/**
|
|
4428
|
+
* Keep enough room for Copilot's message framing and for small differences
|
|
4429
|
+
* between counting the JSON text here and counting the decoded request there.
|
|
4430
|
+
*/
|
|
4431
|
+
const PROMPT_WINDOW_RESERVE = 2e3;
|
|
4432
|
+
const TOOL_RESULT_STUB = "[earlier tool output elided to fit prompt window]";
|
|
4433
|
+
const MESSAGE_TEXT_STUB = "[earlier message text elided to fit prompt window]";
|
|
4434
|
+
const MARKER_RE = /^\[github-router: elided ~([0-9]+) tokens of older tool output to fit this model's prompt window\]$/;
|
|
4435
|
+
function markerText(tokens) {
|
|
4436
|
+
return `[github-router: elided ~${tokens} tokens of older tool output to fit this model's prompt window]`;
|
|
4437
|
+
}
|
|
4438
|
+
function isRecord(value) {
|
|
4439
|
+
return typeof value === "object" && value !== null;
|
|
4440
|
+
}
|
|
4441
|
+
function textReplacement(block, replacement) {
|
|
4442
|
+
if (block.type !== "text" || typeof block.text !== "string") return void 0;
|
|
4443
|
+
if (block.text === replacement) return void 0;
|
|
4444
|
+
return {
|
|
4445
|
+
original: block.text,
|
|
4446
|
+
replacement,
|
|
4447
|
+
apply: () => {
|
|
4448
|
+
block.text = replacement;
|
|
4449
|
+
}
|
|
4450
|
+
};
|
|
4451
|
+
}
|
|
4452
|
+
function collectToolResultReplacements(messages, lastMessageIndex) {
|
|
4453
|
+
const replacements = [];
|
|
4454
|
+
for (let i = 0; i < lastMessageIndex; i += 1) {
|
|
4455
|
+
const message = messages[i];
|
|
4456
|
+
if (!isRecord(message) || message.role !== "user" || !Array.isArray(message.content)) continue;
|
|
4457
|
+
for (const block of message.content) {
|
|
4458
|
+
if (!isRecord(block) || block.type !== "tool_result") continue;
|
|
4459
|
+
if (typeof block.content === "string") {
|
|
4460
|
+
if (block.content === TOOL_RESULT_STUB) continue;
|
|
4461
|
+
const original = block.content;
|
|
4462
|
+
replacements.push({
|
|
4463
|
+
original,
|
|
4464
|
+
replacement: TOOL_RESULT_STUB,
|
|
4465
|
+
apply: () => {
|
|
4466
|
+
block.content = TOOL_RESULT_STUB;
|
|
4467
|
+
}
|
|
4468
|
+
});
|
|
4469
|
+
continue;
|
|
4470
|
+
}
|
|
4471
|
+
if (!Array.isArray(block.content)) continue;
|
|
4472
|
+
for (const nested of block.content) {
|
|
4473
|
+
if (!isRecord(nested)) continue;
|
|
4474
|
+
const replacement = textReplacement(nested, TOOL_RESULT_STUB);
|
|
4475
|
+
if (replacement) replacements.push(replacement);
|
|
4476
|
+
}
|
|
4477
|
+
}
|
|
4478
|
+
}
|
|
4479
|
+
return replacements;
|
|
4480
|
+
}
|
|
4481
|
+
function collectMessageTextReplacements(messages, lastMessageIndex) {
|
|
4482
|
+
const replacements = [];
|
|
4483
|
+
for (let i = 0; i < lastMessageIndex; i += 1) {
|
|
4484
|
+
const message = messages[i];
|
|
4485
|
+
if (!isRecord(message)) continue;
|
|
4486
|
+
if (message.role !== "assistant" && message.role !== "user") continue;
|
|
4487
|
+
if (typeof message.content === "string") {
|
|
4488
|
+
if (message.content === MESSAGE_TEXT_STUB) continue;
|
|
4489
|
+
const original = message.content;
|
|
4490
|
+
replacements.push({
|
|
4491
|
+
original,
|
|
4492
|
+
replacement: MESSAGE_TEXT_STUB,
|
|
4493
|
+
apply: () => {
|
|
4494
|
+
message.content = MESSAGE_TEXT_STUB;
|
|
4495
|
+
}
|
|
4496
|
+
});
|
|
4497
|
+
continue;
|
|
4498
|
+
}
|
|
4499
|
+
if (!Array.isArray(message.content)) continue;
|
|
4500
|
+
for (const block of message.content) {
|
|
4501
|
+
if (!isRecord(block)) continue;
|
|
4502
|
+
const replacement = textReplacement(block, MESSAGE_TEXT_STUB);
|
|
4503
|
+
if (replacement) replacements.push(replacement);
|
|
4504
|
+
}
|
|
4505
|
+
}
|
|
4506
|
+
return replacements;
|
|
4507
|
+
}
|
|
4508
|
+
function hasSalvageStub(messages) {
|
|
4509
|
+
for (const message of messages.slice(0, -1)) {
|
|
4510
|
+
if (!isRecord(message)) continue;
|
|
4511
|
+
if (message.content === MESSAGE_TEXT_STUB) return true;
|
|
4512
|
+
if (!Array.isArray(message.content)) continue;
|
|
4513
|
+
for (const block of message.content) {
|
|
4514
|
+
if (!isRecord(block)) continue;
|
|
4515
|
+
if (block.type === "text" && block.text === MESSAGE_TEXT_STUB) return true;
|
|
4516
|
+
if (block.type !== "tool_result") continue;
|
|
4517
|
+
if (block.content === TOOL_RESULT_STUB) return true;
|
|
4518
|
+
if (Array.isArray(block.content) && block.content.some((nested) => isRecord(nested) && nested.type === "text" && nested.text === TOOL_RESULT_STUB)) return true;
|
|
4519
|
+
}
|
|
4520
|
+
}
|
|
4521
|
+
return false;
|
|
4522
|
+
}
|
|
4523
|
+
function ensureMarker(messages) {
|
|
4524
|
+
const last = messages.at(-1);
|
|
4525
|
+
if (!isRecord(last) || last.role !== "user") return void 0;
|
|
4526
|
+
if (typeof last.content === "string") last.content = [{
|
|
4527
|
+
type: "text",
|
|
4528
|
+
text: last.content
|
|
4529
|
+
}];
|
|
4530
|
+
if (!Array.isArray(last.content)) return void 0;
|
|
4531
|
+
for (const contentBlock of last.content) {
|
|
4532
|
+
if (!isRecord(contentBlock)) continue;
|
|
4533
|
+
if (contentBlock.type !== "text" || typeof contentBlock.text !== "string") continue;
|
|
4534
|
+
const match = MARKER_RE.exec(contentBlock.text);
|
|
4535
|
+
if (!match) continue;
|
|
4536
|
+
if (!hasSalvageStub(messages)) return void 0;
|
|
4537
|
+
const parsed = Number.parseInt(match[1], 10);
|
|
4538
|
+
if (!Number.isSafeInteger(parsed) || parsed < 0) return void 0;
|
|
4539
|
+
return {
|
|
4540
|
+
block: contentBlock,
|
|
4541
|
+
priorElidedTokens: parsed
|
|
4542
|
+
};
|
|
4543
|
+
}
|
|
4544
|
+
const block = {
|
|
4545
|
+
type: "text",
|
|
4546
|
+
text: markerText(0)
|
|
4547
|
+
};
|
|
4548
|
+
last.content.push(block);
|
|
4549
|
+
return {
|
|
4550
|
+
block,
|
|
4551
|
+
priorElidedTokens: 0
|
|
4552
|
+
};
|
|
4553
|
+
}
|
|
4554
|
+
async function fragmentTokenSavings(replacement, encoding) {
|
|
4555
|
+
const [before, after] = await Promise.all([getTextTokenCount(JSON.stringify(replacement.original), encoding), getTextTokenCount(JSON.stringify(replacement.replacement), encoding)]);
|
|
4556
|
+
return before - after;
|
|
4557
|
+
}
|
|
4558
|
+
/**
|
|
4559
|
+
* Last-resort history salvage for requests that escaped the client's normal
|
|
4560
|
+
* compaction path. Protected content is never rewritten, and the original
|
|
4561
|
+
* string is returned unless a complete, valid-looking salvage fits the live
|
|
4562
|
+
* model budget.
|
|
4563
|
+
*/
|
|
4564
|
+
async function salvageOversizedPrompt(rawBody, model) {
|
|
4565
|
+
const unchanged = {
|
|
4566
|
+
body: rawBody,
|
|
4567
|
+
salvaged: false
|
|
4568
|
+
};
|
|
4569
|
+
if (!model) return unchanged;
|
|
4570
|
+
const maxPromptTokens = model.capabilities?.limits?.max_prompt_tokens;
|
|
4571
|
+
if (typeof maxPromptTokens !== "number" || !Number.isFinite(maxPromptTokens) || maxPromptTokens <= 0) return unchanged;
|
|
4572
|
+
const budget = Math.floor(maxPromptTokens) - PROMPT_WINDOW_RESERVE;
|
|
4573
|
+
if (budget <= 0) return unchanged;
|
|
4574
|
+
if (Buffer.byteLength(rawBody, "utf8") <= budget) return unchanged;
|
|
4575
|
+
const encoding = getTokenizerFromModel(model);
|
|
4576
|
+
let initialTokens;
|
|
4577
|
+
try {
|
|
4578
|
+
initialTokens = await getTextTokenCount(rawBody, encoding);
|
|
4579
|
+
} catch (error) {
|
|
4580
|
+
consola.debug("Prompt-window salvage tokenization failed; allowing request:", error);
|
|
4581
|
+
return unchanged;
|
|
4582
|
+
}
|
|
4583
|
+
if (initialTokens <= budget) return unchanged;
|
|
4584
|
+
let body;
|
|
4585
|
+
try {
|
|
4586
|
+
const parsed = JSON.parse(rawBody);
|
|
4587
|
+
if (!isRecord(parsed) || !Array.isArray(parsed.messages) || parsed.messages.length === 0) return unchanged;
|
|
4588
|
+
body = parsed;
|
|
4589
|
+
} catch {
|
|
4590
|
+
return unchanged;
|
|
4591
|
+
}
|
|
4592
|
+
const messages = body.messages;
|
|
4593
|
+
const marker = ensureMarker(messages);
|
|
4594
|
+
if (!marker) return unchanged;
|
|
4595
|
+
const replacements = [...collectToolResultReplacements(messages, messages.length - 1), ...collectMessageTextReplacements(messages, messages.length - 1)];
|
|
4596
|
+
let newlyElidedTokens = 0;
|
|
4597
|
+
let estimatedTokens = initialTokens;
|
|
4598
|
+
for (const replacement of replacements) {
|
|
4599
|
+
let savings;
|
|
4600
|
+
try {
|
|
4601
|
+
savings = await fragmentTokenSavings(replacement, encoding);
|
|
4602
|
+
} catch (error) {
|
|
4603
|
+
consola.debug("Prompt-window salvage tokenization failed; allowing request:", error);
|
|
4604
|
+
return unchanged;
|
|
4605
|
+
}
|
|
4606
|
+
if (savings <= 0) continue;
|
|
4607
|
+
replacement.apply();
|
|
4608
|
+
newlyElidedTokens += savings;
|
|
4609
|
+
estimatedTokens -= savings;
|
|
4610
|
+
const totalElidedTokens = marker.priorElidedTokens + newlyElidedTokens;
|
|
4611
|
+
marker.block.text = markerText(totalElidedTokens);
|
|
4612
|
+
if (estimatedTokens > budget) continue;
|
|
4613
|
+
let serialized;
|
|
4614
|
+
let finalTokens;
|
|
4615
|
+
try {
|
|
4616
|
+
serialized = JSON.stringify(body);
|
|
4617
|
+
finalTokens = await getTextTokenCount(serialized, encoding);
|
|
4618
|
+
} catch (error) {
|
|
4619
|
+
consola.debug("Prompt-window salvage serialization failed; allowing request:", error);
|
|
4620
|
+
return unchanged;
|
|
4621
|
+
}
|
|
4622
|
+
if (finalTokens > budget) {
|
|
4623
|
+
estimatedTokens = finalTokens;
|
|
4624
|
+
continue;
|
|
4625
|
+
}
|
|
4626
|
+
consola.warn(`Prompt-window salvage: model=${model.id} tokens=${initialTokens} budget=${budget} elided=${totalElidedTokens}`);
|
|
4627
|
+
return {
|
|
4628
|
+
body: serialized,
|
|
4629
|
+
salvaged: true,
|
|
4630
|
+
elidedTokens: totalElidedTokens
|
|
4631
|
+
};
|
|
4632
|
+
}
|
|
4633
|
+
return unchanged;
|
|
4634
|
+
}
|
|
4635
|
+
//#endregion
|
|
4584
4636
|
//#region src/routes/messages/handler.ts
|
|
4585
4637
|
const MAX_THINKING_REPAIR_ATTEMPTS = 5;
|
|
4586
4638
|
const isWebSearchTool$1 = (tool) => !!tool && typeof tool === "object" && (typeof tool.type === "string" && tool.type.startsWith("web_search") || tool.name === "web_search");
|
|
@@ -4649,6 +4701,32 @@ function stripWebSearchTool(body) {
|
|
|
4649
4701
|
* in probes `shim_advisor_degrade_gpt55` and
|
|
4650
4702
|
* `shim_advisor_degrade_gemini35flash`.
|
|
4651
4703
|
*/
|
|
4704
|
+
function hasNonEmptyTools(rawBody) {
|
|
4705
|
+
let body;
|
|
4706
|
+
try {
|
|
4707
|
+
body = JSON.parse(rawBody);
|
|
4708
|
+
} catch {
|
|
4709
|
+
return false;
|
|
4710
|
+
}
|
|
4711
|
+
return Array.isArray(body.tools) && body.tools.length > 0;
|
|
4712
|
+
}
|
|
4713
|
+
function fastAdvisorMetadataMismatch(rawBody) {
|
|
4714
|
+
let body;
|
|
4715
|
+
try {
|
|
4716
|
+
body = JSON.parse(rawBody);
|
|
4717
|
+
} catch {
|
|
4718
|
+
return;
|
|
4719
|
+
}
|
|
4720
|
+
if (!Array.isArray(body.tools) || body.tools.length === 0) return void 0;
|
|
4721
|
+
for (const tool of body.tools) {
|
|
4722
|
+
if (!tool || typeof tool !== "object") continue;
|
|
4723
|
+
const type = tool.type;
|
|
4724
|
+
if (typeof type !== "string" || !type.startsWith("advisor_")) continue;
|
|
4725
|
+
const model = tool.model;
|
|
4726
|
+
if (typeof model !== "string") return "the native Advisor tool omitted its fixed model";
|
|
4727
|
+
if (stripTrailingOneMSuffix(model).base !== FAST_PROFILE_ADVISOR_MODEL) return `the native Advisor tool requested ${JSON.stringify(model)} instead of ${JSON.stringify(FAST_PROFILE_ADVISOR_MODEL)}`;
|
|
4728
|
+
}
|
|
4729
|
+
}
|
|
4652
4730
|
function stripAdvisorTool(rawBody) {
|
|
4653
4731
|
let body;
|
|
4654
4732
|
try {
|
|
@@ -4785,6 +4863,30 @@ async function handleCompletion(c) {
|
|
|
4785
4863
|
const fastSubagentRequest = fastProfileRequest && Boolean(c.req.header("x-claude-code-agent-id"));
|
|
4786
4864
|
const fastLeadAdvisor = fastProfileRequest && !fastSubagentRequest;
|
|
4787
4865
|
const advisorEnabled = advisorRequested && !fastSubagentRequest;
|
|
4866
|
+
const fastAdvisorEnabled = fastLeadAdvisor && advisorRequested && hasNonEmptyTools(rawBody);
|
|
4867
|
+
const advisorBehaviorEnabled = fastLeadAdvisor ? fastAdvisorEnabled : advisorEnabled;
|
|
4868
|
+
let fastAdvisorChoice;
|
|
4869
|
+
if (fastAdvisorEnabled) {
|
|
4870
|
+
const mismatch = fastAdvisorMetadataMismatch(rawBody);
|
|
4871
|
+
if (mismatch) return c.json({
|
|
4872
|
+
type: "error",
|
|
4873
|
+
error: {
|
|
4874
|
+
type: "invalid_request_error",
|
|
4875
|
+
message: `Fast Advisor model mismatch: ${mismatch}. Run \`/advisor ${FAST_PROFILE_ADVISOR_MODEL}[1m]\` to restore the fixed fast profile, or relaunch with \`github-router claude -m fast\`.`
|
|
4876
|
+
}
|
|
4877
|
+
}, 400, { "x-should-retry": "false" });
|
|
4878
|
+
try {
|
|
4879
|
+
fastAdvisorChoice = resolveAdvisorModel(void 0, true);
|
|
4880
|
+
} catch (error) {
|
|
4881
|
+
return c.json({
|
|
4882
|
+
type: "error",
|
|
4883
|
+
error: {
|
|
4884
|
+
type: "api_error",
|
|
4885
|
+
message: error instanceof Error ? error.message : String(error)
|
|
4886
|
+
}
|
|
4887
|
+
}, 503);
|
|
4888
|
+
}
|
|
4889
|
+
}
|
|
4788
4890
|
const fastPreprocess = preprocessFastRequest(rawBody, identity.launch);
|
|
4789
4891
|
if (fastPreprocess.rejectedAlias || fastPreprocess.rejectedModel) {
|
|
4790
4892
|
const message = fastPreprocess.rejectedAlias ? `Router-owned model alias ${JSON.stringify(fastPreprocess.rejectedAlias)} is valid only for an authenticated -m fast launch.` : `Model ${JSON.stringify(fastPreprocess.rejectedModel)} is outside the fixed -m fast model set.`;
|
|
@@ -4808,7 +4910,7 @@ async function handleCompletion(c) {
|
|
|
4808
4910
|
}
|
|
4809
4911
|
}, 400, { "x-should-retry": "false" });
|
|
4810
4912
|
if (loopGuard.body !== void 0) finalBody = loopGuard.body;
|
|
4811
|
-
if (
|
|
4913
|
+
if (advisorBehaviorEnabled) {
|
|
4812
4914
|
finalBody = injectAdvisorTool(finalBody, fastLeadAdvisor ? FAST_ADVISOR_TOOL_INSTRUCTIONS : void 0);
|
|
4813
4915
|
consola.info("ADVISOR enabled for this request — injecting __anthropic_advisor tool; will translate tool_use → server_tool_use{advisor} on the SSE stream");
|
|
4814
4916
|
}
|
|
@@ -4824,15 +4926,16 @@ async function handleCompletion(c) {
|
|
|
4824
4926
|
} catch {}
|
|
4825
4927
|
const { body: resolvedBody, originalModel, resolvedModel, selectedModel } = resolveModelInBody$1(finalBody);
|
|
4826
4928
|
const modelId = resolvedModel ?? originalModel;
|
|
4827
|
-
const
|
|
4929
|
+
const { body: promptWindowBody } = await salvageOversizedPrompt(resolvedBody, selectedModel);
|
|
4930
|
+
const messagesRoute = classifyMessagesRoute(modelId, selectedModel, originalModel, fastProfileRequest);
|
|
4828
4931
|
if (messagesRoute !== "claude-passthrough") {
|
|
4829
4932
|
const endpoint = messagesRoute === "chat-shim" ? "chat" : "responses";
|
|
4830
4933
|
let parsedBase;
|
|
4831
4934
|
try {
|
|
4832
|
-
parsedBase = JSON.parse(
|
|
4935
|
+
parsedBase = JSON.parse(promptWindowBody);
|
|
4833
4936
|
} catch {}
|
|
4834
4937
|
const wantsStream = parsedBase?.stream === true;
|
|
4835
|
-
if (
|
|
4938
|
+
if (fastAdvisorEnabled && wantsStream) {
|
|
4836
4939
|
const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
|
|
4837
4940
|
const parsedInitial = parseAnthropicRequest(parsedBase, modelId, selectedModel);
|
|
4838
4941
|
const fastAdvisorAborter = new AbortController();
|
|
@@ -4850,7 +4953,7 @@ async function handleCompletion(c) {
|
|
|
4850
4953
|
status: 200,
|
|
4851
4954
|
streaming: true
|
|
4852
4955
|
}, selectedModel, startTime);
|
|
4853
|
-
const advisorChoice =
|
|
4956
|
+
const advisorChoice = fastAdvisorChoice;
|
|
4854
4957
|
return new Response(buildAdvisorStream({
|
|
4855
4958
|
firstResponse,
|
|
4856
4959
|
initialConversation,
|
|
@@ -4875,8 +4978,8 @@ async function handleCompletion(c) {
|
|
|
4875
4978
|
}
|
|
4876
4979
|
});
|
|
4877
4980
|
}
|
|
4878
|
-
const shimBody = stripAdvisorTool(
|
|
4879
|
-
if (
|
|
4981
|
+
const shimBody = stripAdvisorTool(promptWindowBody);
|
|
4982
|
+
if (advisorBehaviorEnabled) consola.info("ADVISOR requested with a non-Claude model — stripping the injected __anthropic_advisor tool and proceeding without advisor (Claude-only feature; gracefully degraded).");
|
|
4880
4983
|
const shimOpts = {
|
|
4881
4984
|
rawBody: shimBody,
|
|
4882
4985
|
modelId,
|
|
@@ -4888,12 +4991,12 @@ async function handleCompletion(c) {
|
|
|
4888
4991
|
}
|
|
4889
4992
|
if (modelId) logEndpointMismatch(modelId, "/v1/messages");
|
|
4890
4993
|
const effectiveBetas = applyDefaultBetas(betaHeaders, resolvedModel ?? originalModel);
|
|
4891
|
-
const advisorAborter =
|
|
4994
|
+
const advisorAborter = advisorBehaviorEnabled ? new AbortController() : void 0;
|
|
4892
4995
|
const requestHeaders = {
|
|
4893
4996
|
...selectedModel?.requestHeaders,
|
|
4894
4997
|
...effectiveBetas
|
|
4895
4998
|
};
|
|
4896
|
-
let nativeBody =
|
|
4999
|
+
let nativeBody = promptWindowBody;
|
|
4897
5000
|
const knownThinkingRepair = repairKnownThinkingHistory(nativeBody);
|
|
4898
5001
|
if (knownThinkingRepair) {
|
|
4899
5002
|
nativeBody = knownThinkingRepair.body;
|
|
@@ -4962,13 +5065,13 @@ async function handleCompletion(c) {
|
|
|
4962
5065
|
if (requestId) streamHeaders["x-request-id"] = requestId;
|
|
4963
5066
|
const reqId = response.headers.get("request-id");
|
|
4964
5067
|
if (reqId) streamHeaders["request-id"] = reqId;
|
|
4965
|
-
if (
|
|
5068
|
+
if (advisorBehaviorEnabled && response.body) {
|
|
4966
5069
|
let parsedBase = {};
|
|
4967
5070
|
try {
|
|
4968
5071
|
parsedBase = JSON.parse(nativeBody);
|
|
4969
5072
|
} catch {}
|
|
4970
5073
|
const initialConversation = Array.isArray(parsedBase.messages) ? parsedBase.messages : [];
|
|
4971
|
-
const advisorChoice = resolveAdvisorModel(originalModel,
|
|
5074
|
+
const advisorChoice = fastAdvisorChoice ?? resolveAdvisorModel(originalModel, false);
|
|
4972
5075
|
return new Response(buildAdvisorStream({
|
|
4973
5076
|
firstResponse: response,
|
|
4974
5077
|
initialConversation,
|
|
@@ -6500,7 +6603,7 @@ function clearGatewayModelCache(configDir = PATHS.CLAUDE_CONFIG_DIR) {
|
|
|
6500
6603
|
*/
|
|
6501
6604
|
function oneMSuffixForAlias(aliasId) {
|
|
6502
6605
|
if (oneMContextDisabled()) return aliasId;
|
|
6503
|
-
return catalogAdvertises1M(
|
|
6606
|
+
return catalogAdvertises1M(LUNA_REAL_MODEL_ID) ? `${aliasId}[1m]` : aliasId;
|
|
6504
6607
|
}
|
|
6505
6608
|
function getClaudeCodeEnvVars(serverUrl, model, launchProfileId = "standard") {
|
|
6506
6609
|
const vars = {
|
|
@@ -6555,10 +6658,66 @@ function getClaudeCodeEnvVars(serverUrl, model, launchProfileId = "standard") {
|
|
|
6555
6658
|
if (nativeModels.length > 0) {
|
|
6556
6659
|
if (seedGatewayModelCache(serverUrl, nativeModels) && process.env.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0 && vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY === void 0) vars.CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY = "1";
|
|
6557
6660
|
} else clearGatewayModelCache();
|
|
6661
|
+
applyAutoCompactWindow(vars);
|
|
6558
6662
|
if (toolbeltEnabled()) Object.assign(vars, toolbeltPathOverride(process.env, PATHS.TOOLBELT_BIN_DIR));
|
|
6559
6663
|
return vars;
|
|
6560
6664
|
}
|
|
6561
6665
|
/**
|
|
6666
|
+
* Every env var whose value can become the ACTIVE main-loop model id, and so
|
|
6667
|
+
* whose prompt ceiling the compaction window has to respect. `ANTHROPIC_MODEL`
|
|
6668
|
+
* is the lead; the tier rows and the custom option each become the active id
|
|
6669
|
+
* verbatim when picked from `/model`.
|
|
6670
|
+
*
|
|
6671
|
+
* `ANTHROPIC_SMALL_FAST_MODEL` is deliberately absent: background ops run
|
|
6672
|
+
* there, but compaction itself runs on the main-loop model (verified in the
|
|
6673
|
+
* 2.1.247 bundle — the summarizer starts from `r.options.mainLoopModel`).
|
|
6674
|
+
*/
|
|
6675
|
+
const LEAD_CAPABLE_MODEL_ENV_KEYS = [
|
|
6676
|
+
"ANTHROPIC_MODEL",
|
|
6677
|
+
"ANTHROPIC_DEFAULT_OPUS_MODEL",
|
|
6678
|
+
"ANTHROPIC_DEFAULT_SONNET_MODEL",
|
|
6679
|
+
"ANTHROPIC_DEFAULT_HAIKU_MODEL",
|
|
6680
|
+
"ANTHROPIC_CUSTOM_MODEL_OPTION"
|
|
6681
|
+
];
|
|
6682
|
+
/**
|
|
6683
|
+
* Resolve one seeded env value to its live catalog entry: strip the `[1m]`
|
|
6684
|
+
* accounting bracket, erase router-owned alias provenance, then translate the
|
|
6685
|
+
* Anthropic-dashed slug onto Copilot's catalog id.
|
|
6686
|
+
*/
|
|
6687
|
+
function catalogEntryForSeededModel(value) {
|
|
6688
|
+
const { base } = stripTrailingOneMSuffix(canonicalizeAliasModel(value));
|
|
6689
|
+
const id = resolveModel(base);
|
|
6690
|
+
return state.models?.data?.find((m) => m.id === id);
|
|
6691
|
+
}
|
|
6692
|
+
/**
|
|
6693
|
+
* Set `CLAUDE_CODE_AUTO_COMPACT_WINDOW` to the smallest complete derived
|
|
6694
|
+
* window across every 1M-accounted model this launch can reach. Each candidate
|
|
6695
|
+
* puts the client's reactive trigger at 85% of that model's prompt ceiling;
|
|
6696
|
+
* the minimum is therefore safe after any `/model` switch.
|
|
6697
|
+
*
|
|
6698
|
+
* Only `[1m]`-decorated candidates participate. An undecorated row already
|
|
6699
|
+
* budgets at the client's conservative 200K default, which is below every
|
|
6700
|
+
* prompt ceiling in the lineup, so including it would drag the window down for
|
|
6701
|
+
* no benefit. When nothing is decorated, or no candidate carries usable
|
|
6702
|
+
* catalog limits, the variable is omitted entirely rather than guessed.
|
|
6703
|
+
*
|
|
6704
|
+
* Presence-guarded on the parent env, symmetric with every other guard here.
|
|
6705
|
+
*/
|
|
6706
|
+
function applyAutoCompactWindow(vars) {
|
|
6707
|
+
if (process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW !== void 0) return;
|
|
6708
|
+
let tightestWindow;
|
|
6709
|
+
const consider = (value) => {
|
|
6710
|
+
if (!value || !/\[1m\]/i.test(value)) return;
|
|
6711
|
+
const limits = catalogEntryForSeededModel(value)?.capabilities?.limits;
|
|
6712
|
+
const candidateWindow = deriveAutoCompactWindowTokens(limits?.max_prompt_tokens, limits?.max_output_tokens);
|
|
6713
|
+
if (candidateWindow !== void 0 && (tightestWindow === void 0 || candidateWindow < tightestWindow)) tightestWindow = candidateWindow;
|
|
6714
|
+
};
|
|
6715
|
+
for (const key of LEAD_CAPABLE_MODEL_ENV_KEYS) consider(vars[key] ?? process.env[key]);
|
|
6716
|
+
for (const model of nativeSelectableModelsInCatalog()) consider(model.id);
|
|
6717
|
+
if (tightestWindow === void 0) return;
|
|
6718
|
+
vars.CLAUDE_CODE_AUTO_COMPACT_WINDOW = String(tightestWindow);
|
|
6719
|
+
}
|
|
6720
|
+
/**
|
|
6562
6721
|
* Build environment variables for Codex CLI.
|
|
6563
6722
|
*
|
|
6564
6723
|
* Like `getClaudeCodeEnvVars`, the parent env is sanitized of
|
|
@@ -6580,6 +6739,6 @@ function getCodexEnvVars(serverUrl) {
|
|
|
6580
6739
|
return vars;
|
|
6581
6740
|
}
|
|
6582
6741
|
//#endregion
|
|
6583
|
-
export {
|
|
6742
|
+
export { sharedServerArgs as a, stopKeepAwake as c, runSelfUpdate as d, checkClaudeVersion as f, setupAndServe as i, listModelsForEndpoint as l, getCodexEnvVars as n, LAUNCH_SECRET_HEADER as o, updateClaude as p, parseSharedArgs as r, startKeepAwake as s, getClaudeCodeEnvVars as t, enableFileLogging as u };
|
|
6584
6743
|
|
|
6585
|
-
//# sourceMappingURL=server-setup-
|
|
6744
|
+
//# sourceMappingURL=server-setup-DhG3YZPj.js.map
|