github-router 0.3.288 → 0.3.292
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{attribution-settings-CofaqSzw.js → attribution-settings-CpLUCi8R.js} +124 -34
- package/dist/attribution-settings-CpLUCi8R.js.map +1 -0
- package/dist/{auth-DG4vh8-F.js → auth-BwUHopJz.js} +3 -3
- package/dist/{auth-DG4vh8-F.js.map → auth-BwUHopJz.js.map} +1 -1
- package/dist/browser-ext/manifest.json +1 -1
- package/dist/{check-usage-BqN7mBYv.js → check-usage-BTda5753.js} +4 -4
- package/dist/{check-usage-BqN7mBYv.js.map → check-usage-BTda5753.js.map} +1 -1
- package/dist/{claude-CoZKRNB8.js → claude-_DYGKCw8.js} +101 -49
- package/dist/claude-_DYGKCw8.js.map +1 -0
- package/dist/{codex-y2OyLIdv.js → codex-rTJ8jW5G.js} +5 -5
- package/dist/{codex-y2OyLIdv.js.map → codex-rTJ8jW5G.js.map} +1 -1
- package/dist/{debug-B5TjPTTH.js → debug-B3UrZTHQ.js} +2 -2
- package/dist/{debug-B5TjPTTH.js.map → debug-B3UrZTHQ.js.map} +1 -1
- package/dist/engine-C9axTIu7.js +2 -0
- package/dist/{gate-discovery-LQZ-enJa.js → gate-discovery-Bjar5dgv.js} +5 -5
- package/dist/{gate-discovery-LQZ-enJa.js.map → gate-discovery-Bjar5dgv.js.map} +1 -1
- package/dist/{get-copilot-usage-BjA0nyGR.js → get-copilot-usage-CRrf1ZSC.js} +2 -2
- package/dist/{get-copilot-usage-BjA0nyGR.js.map → get-copilot-usage-CRrf1ZSC.js.map} +1 -1
- package/dist/{internal-artifact-open-BskmUpnb.js → internal-artifact-open-BsEQRvDi.js} +2 -2
- package/dist/{internal-artifact-open-BskmUpnb.js.map → internal-artifact-open-BsEQRvDi.js.map} +1 -1
- package/dist/{internal-first-mate-guard-5XEHMaqy.js → internal-first-mate-guard-CVFFJriA.js} +3 -3
- package/dist/{internal-first-mate-guard-5XEHMaqy.js.map → internal-first-mate-guard-CVFFJriA.js.map} +1 -1
- package/dist/{internal-first-mate-guard-DHDQ6hFz.js → internal-first-mate-guard-DJdX6t48.js} +1 -1
- package/dist/{internal-plan-review-BUuMx4ku.js → internal-plan-review-8TKDLco6.js} +3 -3
- package/dist/{internal-plan-review-BUuMx4ku.js.map → internal-plan-review-8TKDLco6.js.map} +1 -1
- package/dist/{internal-prompt-submit-CQQ15xdO.js → internal-prompt-submit-Da7pxqua.js} +4 -4
- package/dist/{internal-prompt-submit-CQQ15xdO.js.map → internal-prompt-submit-Da7pxqua.js.map} +1 -1
- package/dist/{internal-session-bind-D04W2yWI.js → internal-session-bind-BOFytA1f.js} +2 -2
- package/dist/{internal-session-bind-D04W2yWI.js.map → internal-session-bind-BOFytA1f.js.map} +1 -1
- package/dist/{internal-stop-hook-DSbaDb_m.js → internal-stop-hook-DrX2xlj0.js} +5 -5
- package/dist/{internal-stop-hook-DSbaDb_m.js.map → internal-stop-hook-DrX2xlj0.js.map} +1 -1
- package/dist/{internal-stop-review-CdByyJLc.js → internal-stop-review-CdouacHL.js} +2 -2
- package/dist/{internal-stop-review-CdByyJLc.js.map → internal-stop-review-CdouacHL.js.map} +1 -1
- package/dist/{internal-worker-guard-BIPN6Rv9.js → internal-worker-guard-Bx-itoP8.js} +2 -2
- package/dist/{internal-worker-guard-BIPN6Rv9.js.map → internal-worker-guard-Bx-itoP8.js.map} +1 -1
- package/dist/{internal-workspace-header-BKqejstG.js → internal-workspace-header-8WT0iB5K.js} +2 -2
- package/dist/{internal-workspace-header-BKqejstG.js.map → internal-workspace-header-8WT0iB5K.js.map} +1 -1
- package/dist/lifecycle-C8fOsQke.js +2 -0
- package/dist/lifecycle-D4Yc1aap.js +2 -0
- package/dist/{lifecycle-SXaWssN9.js → lifecycle-LeSfa7wH.js} +2 -2
- package/dist/{lifecycle-SXaWssN9.js.map → lifecycle-LeSfa7wH.js.map} +1 -1
- package/dist/{lifecycle-DbM29FLK.js → lifecycle-nuOHfwgj.js} +2 -2
- package/dist/{lifecycle-DbM29FLK.js.map → lifecycle-nuOHfwgj.js.map} +1 -1
- package/dist/main.js +17 -17
- package/dist/{mcp-workspace-header-DRCCWlOi.js → mcp-workspace-header-q34H_4wL.js} +2 -2
- package/dist/{mcp-workspace-header-DRCCWlOi.js.map → mcp-workspace-header-q34H_4wL.js.map} +1 -1
- package/dist/{models-Dz8d_SnI.js → models-hhJcrZhr.js} +3 -3
- package/dist/{models-Dz8d_SnI.js.map → models-hhJcrZhr.js.map} +1 -1
- package/dist/{orchestration-BrJwZxMN.js → orchestration-pzbrKkgD.js} +2 -2
- package/dist/{orchestration-BrJwZxMN.js.map → orchestration-pzbrKkgD.js.map} +1 -1
- package/dist/{paths-D7_SAaIQ.js → paths-BH4J7slC.js} +4 -4
- package/dist/{paths-D7_SAaIQ.js.map → paths-BH4J7slC.js.map} +1 -1
- package/dist/paths-DJZoXfAS.js +2 -0
- package/dist/{peer-mcp-personas-B5Wp6wIn.js → peer-mcp-personas-CHbl6MwM.js} +927 -147
- package/dist/peer-mcp-personas-CHbl6MwM.js.map +1 -0
- package/dist/{plan-review-hook-CVZsG9MZ.js → plan-review-hook-CfcanA7_.js} +3 -3
- package/dist/{plan-review-hook-CVZsG9MZ.js.map → plan-review-hook-CfcanA7_.js.map} +1 -1
- package/dist/{prompt-submit-hook-BW92FX2D.js → prompt-submit-hook-Bqf9ORgb.js} +3 -3
- package/dist/{prompt-submit-hook-BW92FX2D.js.map → prompt-submit-hook-Bqf9ORgb.js.map} +1 -1
- package/dist/{provision-CUqPki1z.js → provision-BYFd9nPK.js} +4 -4
- package/dist/{provision-CUqPki1z.js.map → provision-BYFd9nPK.js.map} +1 -1
- package/dist/{self-invocation-CP_SOkrr.js → self-invocation-DhO1Z8iD.js} +2 -2
- package/dist/{self-invocation-CP_SOkrr.js.map → self-invocation-DhO1Z8iD.js.map} +1 -1
- package/dist/{serve-BASqoXb3.js → serve-BWMxDLnD.js} +12 -12
- package/dist/{serve-BASqoXb3.js.map → serve-BWMxDLnD.js.map} +1 -1
- package/dist/{server-setup-D5hilphf.js → server-setup-CqlaZukJ.js} +852 -160
- package/dist/server-setup-CqlaZukJ.js.map +1 -0
- package/dist/{start-Rfim4TeF.js → start-5MgGT4IF.js} +3 -3
- package/dist/{start-Rfim4TeF.js.map → start-5MgGT4IF.js.map} +1 -1
- package/dist/{stop-gate-hook-DriRc9xN.js → stop-gate-hook-BiBp5aGm.js} +3 -3
- package/dist/{stop-gate-hook-DriRc9xN.js.map → stop-gate-hook-BiBp5aGm.js.map} +1 -1
- package/dist/{stop-gate-policy-DMPanpoR.js → stop-gate-policy-BGd6b5hR.js} +2 -2
- package/dist/{stop-gate-policy-DMPanpoR.js.map → stop-gate-policy-BGd6b5hR.js.map} +1 -1
- package/dist/{token-BGCjZwtj.js → token-8drORhXg.js} +33 -3
- package/dist/token-8drORhXg.js.map +1 -0
- package/dist/{worker-dispatch-BCTMyNE-.js → worker-dispatch-D5fGroNr.js} +2 -2
- package/dist/{worker-dispatch-BCTMyNE-.js.map → worker-dispatch-D5fGroNr.js.map} +1 -1
- package/package.json +2 -1
- package/dist/attribution-settings-CofaqSzw.js.map +0 -1
- package/dist/claude-CoZKRNB8.js.map +0 -1
- package/dist/engine-B5nVGH4b.js +0 -2
- package/dist/lifecycle-BTodQvn4.js +0 -2
- package/dist/lifecycle-C7JYNz-F.js +0 -2
- package/dist/paths-CTr59UC6.js +0 -2
- package/dist/peer-mcp-personas-B5Wp6wIn.js.map +0 -1
- package/dist/server-setup-D5hilphf.js.map +0 -1
- package/dist/token-BGCjZwtj.js.map +0 -1
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
|
|
2
2
|
import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
|
|
3
|
-
import { t as PATHS } from "./paths-
|
|
4
|
-
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-
|
|
3
|
+
import { t as PATHS } from "./paths-BH4J7slC.js";
|
|
4
|
+
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-8drORhXg.js";
|
|
5
5
|
import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
|
|
6
|
-
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-
|
|
7
|
-
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-
|
|
6
|
+
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-nuOHfwgj.js";
|
|
7
|
+
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-q34H_4wL.js";
|
|
8
8
|
import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
|
|
9
|
-
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-
|
|
10
|
-
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-
|
|
11
|
-
import { t as liveExec } from "./orchestration-
|
|
9
|
+
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-LeSfa7wH.js";
|
|
10
|
+
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-BiBp5aGm.js";
|
|
11
|
+
import { t as liveExec } from "./orchestration-pzbrKkgD.js";
|
|
12
12
|
import { createRequire } from "node:module";
|
|
13
13
|
import consola from "consola";
|
|
14
14
|
import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
|
|
@@ -447,9 +447,17 @@ const DEFAULT_CLAUDE_MODEL_FALLBACKS = [
|
|
|
447
447
|
* variant" — defaulting safe-side preserves the pre-change behavior).
|
|
448
448
|
*/
|
|
449
449
|
const DEFAULT_OPUS_FAMILY = "5";
|
|
450
|
-
/**
|
|
451
|
-
*
|
|
452
|
-
|
|
450
|
+
/**
|
|
451
|
+
* The lead `-m fast` selects. `gpt-5.6-luna` — a distinct Luna-driven
|
|
452
|
+
* profile (see `./launch-profile`), NOT a Claude Sonnet budget lead. This
|
|
453
|
+
* REPLACES the earlier `-m fast` → `claude-sonnet-5` mapping: `fast` now
|
|
454
|
+
* names a deliberately lean Luna surface (three native agents, one peer
|
|
455
|
+
* persona, `peers`/`search` MCP groups only), not "budget Sonnet with the
|
|
456
|
+
* full standard surface". `resolveLaunchProfile` in `./launch-profile`
|
|
457
|
+
* keys off the same raw `-m` argument this constant is selected by, so the
|
|
458
|
+
* two can never disagree about which launches count as "fast".
|
|
459
|
+
*/
|
|
460
|
+
const FAST_LEAD_MODEL = "gpt-5.6-luna";
|
|
453
461
|
/** Small/fast tier for a budget lead, in the two forms this codebase needs.
|
|
454
462
|
*
|
|
455
463
|
* `SLUG` is the Anthropic-published DASHED form and is what goes into
|
|
@@ -464,28 +472,29 @@ const BUDGET_SMALL_FAST_CATALOG_ID = "claude-haiku-4.5";
|
|
|
464
472
|
/**
|
|
465
473
|
* Resolve the `-m` argument to the lead slug to launch with.
|
|
466
474
|
*
|
|
467
|
-
* - `fast` → `
|
|
475
|
+
* - `fast` → `FAST_LEAD_MODEL` (the fast Luna profile — see
|
|
476
|
+
* `./launch-profile`, NOT the retired Sonnet budget lead)
|
|
468
477
|
* - `N.M` → the best variant of that Opus family, via `pickClaudeDefault`
|
|
469
478
|
* - a full slug → unchanged, including Copilot slugs a power user pins
|
|
470
479
|
* - absent → the ordinary default
|
|
471
480
|
*
|
|
472
481
|
* Every branch is `[1m]`-decorated against the live catalog, by
|
|
473
482
|
* `pickClaudeDefault` on the two Opus-family branches and by
|
|
474
|
-
* `withOneMSuffixForLead` on the other two.
|
|
475
|
-
*
|
|
476
|
-
*
|
|
477
|
-
*
|
|
478
|
-
*
|
|
479
|
-
*
|
|
480
|
-
* (
|
|
481
|
-
*
|
|
482
|
-
*
|
|
483
|
-
*
|
|
484
|
-
*
|
|
485
|
-
*
|
|
486
|
-
*
|
|
487
|
-
*
|
|
488
|
-
*
|
|
483
|
+
* `withOneMSuffixForLead` on the other two. `gpt-5.6-luna` advertises a 1M
|
|
484
|
+
* window, so `-m fast` gets local 1M accounting exactly like every other
|
|
485
|
+
* branch here; the decoration is catalog-gated per model, so a genuinely
|
|
486
|
+
* 200K model (`claude-haiku-4.5`) still comes back bare.
|
|
487
|
+
*
|
|
488
|
+
* `fast` resolves to an ordinary slug rather than setting a mode flag —
|
|
489
|
+
* `resolveLaunchProfile` (`./launch-profile`) is keyed off the SAME raw
|
|
490
|
+
* argument this function receives, so the two can never disagree about
|
|
491
|
+
* which launches are "fast". `isBudgetClaudeLead` (below) stays
|
|
492
|
+
* Claude-family-only and is UNRELATED to the fast profile: `gpt-5.6-luna`
|
|
493
|
+
* is not a Claude model, so `isBudgetClaudeLead(resolveLeadSlugArg("fast"))`
|
|
494
|
+
* is false — the old Sonnet "budget lead" surfaces (advisor escalation,
|
|
495
|
+
* delegation prose, small/fast Haiku tier) simply don't engage for `-m
|
|
496
|
+
* fast` any more; the fast profile has its own separate roster/tier
|
|
497
|
+
* mechanism instead.
|
|
489
498
|
*
|
|
490
499
|
* Callers must keep treating any explicit `-m` as explicit: the
|
|
491
500
|
* `DEFAULT_CLAUDE_MODEL_FALLBACKS` walk applies to the implicit-default path
|
|
@@ -494,7 +503,7 @@ const BUDGET_SMALL_FAST_CATALOG_ID = "claude-haiku-4.5";
|
|
|
494
503
|
function resolveLeadSlugArg(modelArg) {
|
|
495
504
|
const arg = modelArg?.trim();
|
|
496
505
|
if (!arg) return pickClaudeDefault();
|
|
497
|
-
if (arg.toLowerCase() === "fast") return withOneMSuffixForLead(
|
|
506
|
+
if (arg.toLowerCase() === "fast") return withOneMSuffixForLead(FAST_LEAD_MODEL);
|
|
498
507
|
const opusFamilyShorthand = arg.match(/^(\d+\.\d+)$/)?.[1];
|
|
499
508
|
if (opusFamilyShorthand) return pickClaudeDefault(opusFamilyShorthand);
|
|
500
509
|
return withOneMSuffixForLead(arg);
|
|
@@ -18917,7 +18926,7 @@ function logAudit$1(record) {
|
|
|
18917
18926
|
try {
|
|
18918
18927
|
const fs = await import("node:fs/promises");
|
|
18919
18928
|
const path = await import("node:path");
|
|
18920
|
-
const { PATHS } = await import("./paths-
|
|
18929
|
+
const { PATHS } = await import("./paths-DJZoXfAS.js");
|
|
18921
18930
|
const dir = path.join(PATHS.APP_DIR, "browser-mcp");
|
|
18922
18931
|
await fs.mkdir(dir, { recursive: true });
|
|
18923
18932
|
const line = JSON.stringify({
|
|
@@ -19497,6 +19506,354 @@ function currentInFlight() {
|
|
|
19497
19506
|
return inFlight$2;
|
|
19498
19507
|
}
|
|
19499
19508
|
//#endregion
|
|
19509
|
+
//#region src/lib/prompt-cache.ts
|
|
19510
|
+
/**
|
|
19511
|
+
* Conservative eligibility floor, in UTF-8 BYTES — never `.length`, which
|
|
19512
|
+
* counts UTF-16 code units and undercounts anything outside the BMP (an
|
|
19513
|
+
* emoji is 2 code units but 4 bytes). This is a proxy for "the prefix is
|
|
19514
|
+
* obviously large enough that marking it as a cache breakpoint is worth one
|
|
19515
|
+
* of the scarce marker slots," and it deliberately does NOT claim to be a
|
|
19516
|
+
* token count: byte-to-token density varies by tokenizer and content — CJK
|
|
19517
|
+
* text carries MORE tokens per byte than ASCII prose (undercounting risk is
|
|
19518
|
+
* the SAFE direction: we'd skip a marker that might have qualified), while a
|
|
19519
|
+
* long run of a repeated character or repeated whitespace carries FEWER
|
|
19520
|
+
* tokens per byte than either, since BPE merges long runs into very few
|
|
19521
|
+
* tokens (overcounting risk: a byte count clearing the floor doesn't
|
|
19522
|
+
* guarantee the real token count clears Anthropic's or Copilot's per-model
|
|
19523
|
+
* minimum). No fixed byte threshold can bound that adversarial case; this
|
|
19524
|
+
* value is chosen so ordinary Claude Code system prompts and tool schemas
|
|
19525
|
+
* (natural-language / JSON, not deliberately repetitive) reliably qualify,
|
|
19526
|
+
* while genuinely small prefixes never burn a marker for no benefit.
|
|
19527
|
+
*/
|
|
19528
|
+
const MIN_CACHEABLE_PREFIX_BYTES = 4096;
|
|
19529
|
+
const CACHE_KEY_NAMESPACE = "ghr-cache-v1";
|
|
19530
|
+
const CACHE_DIAGNOSTIC_LIMIT = 128;
|
|
19531
|
+
const GPT56_EXPLICIT_CACHE_MODELS = /* @__PURE__ */ new Set([
|
|
19532
|
+
"gpt-5.6-sol",
|
|
19533
|
+
"gpt-5.6-terra",
|
|
19534
|
+
"gpt-5.6-luna"
|
|
19535
|
+
]);
|
|
19536
|
+
const priorSignatures = /* @__PURE__ */ new Map();
|
|
19537
|
+
function nonNegativeInt(value) {
|
|
19538
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return 0;
|
|
19539
|
+
return Math.floor(value);
|
|
19540
|
+
}
|
|
19541
|
+
/**
|
|
19542
|
+
* Pick the first genuinely POSITIVE numeric candidate from an ordered,
|
|
19543
|
+
* priority-ranked list of usage-shape fields. `??`-chaining these fields is
|
|
19544
|
+
* wrong: it stops at the first field that is merely PRESENT, and a provider
|
|
19545
|
+
* surface that always populates a nested detail object with `0` as a
|
|
19546
|
+
* placeholder (while the real, positive count is reported only in a
|
|
19547
|
+
* lower-priority field, e.g. the top-level one) would have its explicit zero
|
|
19548
|
+
* silently shadow that populated count. Falling through zeros to find a real
|
|
19549
|
+
* positive value fixes that; when every candidate is zero, absent, or
|
|
19550
|
+
* non-numeric, this returns `0` — a genuine all-zero reading, never
|
|
19551
|
+
* `undefined` — so downstream `nonNegativeInt` always has a countable value.
|
|
19552
|
+
*/
|
|
19553
|
+
function firstPositive(...candidates) {
|
|
19554
|
+
for (const c of candidates) if (typeof c === "number" && Number.isFinite(c) && c > 0) return c;
|
|
19555
|
+
return 0;
|
|
19556
|
+
}
|
|
19557
|
+
function usageDetails(usage) {
|
|
19558
|
+
const input = usage.input_tokens_details ?? {};
|
|
19559
|
+
const prompt = usage.prompt_tokens_details ?? {};
|
|
19560
|
+
return {
|
|
19561
|
+
cached_tokens: firstPositive(input.cached_tokens, prompt.cached_tokens, usage.cache_read_input_tokens),
|
|
19562
|
+
cache_write_tokens: firstPositive(input.cache_write_tokens, input.cache_creation_tokens, prompt.cache_write_tokens, prompt.cache_creation_tokens, usage.cache_write_tokens, usage.cache_creation_input_tokens),
|
|
19563
|
+
cache_ttl_seconds: firstPositive(input.cache_ttl_seconds, prompt.cache_ttl_seconds, usage.cache_ttl_seconds)
|
|
19564
|
+
};
|
|
19565
|
+
}
|
|
19566
|
+
/**
|
|
19567
|
+
* OpenAI totals INCLUDE cached and cache-write tokens. Normalize them into
|
|
19568
|
+
* mutually exclusive buckets so Anthropic and Pi consumers do not count the
|
|
19569
|
+
* same input twice.
|
|
19570
|
+
*/
|
|
19571
|
+
function normalizeOpenAIUsage(usage) {
|
|
19572
|
+
if (!usage) return {
|
|
19573
|
+
totalInput: 0,
|
|
19574
|
+
uncachedInput: 0,
|
|
19575
|
+
output: 0,
|
|
19576
|
+
cacheRead: 0,
|
|
19577
|
+
cacheWrite: 0,
|
|
19578
|
+
totalTokens: 0
|
|
19579
|
+
};
|
|
19580
|
+
const totalInput = nonNegativeInt(usage.input_tokens ?? usage.prompt_tokens);
|
|
19581
|
+
const output = nonNegativeInt(usage.output_tokens ?? usage.completion_tokens);
|
|
19582
|
+
const details = usageDetails(usage);
|
|
19583
|
+
const cacheRead = Math.min(totalInput, nonNegativeInt(details.cached_tokens));
|
|
19584
|
+
const remaining = Math.max(0, totalInput - cacheRead);
|
|
19585
|
+
const cacheWrite = Math.min(remaining, nonNegativeInt(details.cache_write_tokens ?? details.cache_creation_tokens));
|
|
19586
|
+
const uncachedInput = Math.max(0, totalInput - cacheRead - cacheWrite);
|
|
19587
|
+
const reportedTotal = nonNegativeInt(usage.total_tokens);
|
|
19588
|
+
const cacheTtlSeconds = nonNegativeInt(details.cache_ttl_seconds);
|
|
19589
|
+
return {
|
|
19590
|
+
totalInput,
|
|
19591
|
+
uncachedInput,
|
|
19592
|
+
output,
|
|
19593
|
+
cacheRead,
|
|
19594
|
+
cacheWrite,
|
|
19595
|
+
totalTokens: Math.max(reportedTotal, totalInput + output),
|
|
19596
|
+
...cacheTtlSeconds > 0 ? { cacheTtlSeconds } : {}
|
|
19597
|
+
};
|
|
19598
|
+
}
|
|
19599
|
+
function hash(value) {
|
|
19600
|
+
return createHash("sha256").update(value).digest("hex");
|
|
19601
|
+
}
|
|
19602
|
+
function signatureFor(value) {
|
|
19603
|
+
return hash(typeof value === "string" ? value : JSON.stringify(value ?? null));
|
|
19604
|
+
}
|
|
19605
|
+
function serializedBytes(value) {
|
|
19606
|
+
return Buffer.byteLength(typeof value === "string" ? value : JSON.stringify(value ?? null));
|
|
19607
|
+
}
|
|
19608
|
+
function logCacheSignature(args) {
|
|
19609
|
+
if (parseBoolEnv(process.env.GH_ROUTER_LOG_CACHE) !== true) return;
|
|
19610
|
+
const key = `${args.endpoint}:${args.model}:${args.workload}`;
|
|
19611
|
+
const current = {
|
|
19612
|
+
system: signatureFor(args.system),
|
|
19613
|
+
tools: signatureFor(args.tools),
|
|
19614
|
+
messages: signatureFor(args.messages)
|
|
19615
|
+
};
|
|
19616
|
+
const previous = priorSignatures.get(key);
|
|
19617
|
+
let changed = "cold";
|
|
19618
|
+
if (previous) changed = previous.system !== current.system ? "system" : previous.tools !== current.tools ? "tools" : previous.messages !== current.messages ? "messages" : "none";
|
|
19619
|
+
priorSignatures.set(key, current);
|
|
19620
|
+
if (priorSignatures.size > CACHE_DIAGNOSTIC_LIMIT) {
|
|
19621
|
+
const oldest = priorSignatures.keys().next().value;
|
|
19622
|
+
if (oldest !== void 0) priorSignatures.delete(oldest);
|
|
19623
|
+
}
|
|
19624
|
+
consola.info(`cache-signature endpoint=${args.endpoint} model=${args.model} workload=${args.workload} changed=${changed} system_bytes=${serializedBytes(args.system ?? "")} tools_bytes=${serializedBytes(args.tools ?? [])} messages_bytes=${serializedBytes(args.messages ?? [])}`);
|
|
19625
|
+
}
|
|
19626
|
+
function hasResponsesBreakpoint(input) {
|
|
19627
|
+
const visit = (value) => {
|
|
19628
|
+
if (!value || typeof value !== "object") return false;
|
|
19629
|
+
if (Array.isArray(value)) return value.some(visit);
|
|
19630
|
+
const record = value;
|
|
19631
|
+
return record.prompt_cache_breakpoint !== void 0 || Object.values(record).some(visit);
|
|
19632
|
+
};
|
|
19633
|
+
return visit(input);
|
|
19634
|
+
}
|
|
19635
|
+
function gpt56ExplicitCacheEnabled(model) {
|
|
19636
|
+
if (parseBoolEnv(process.env.GH_ROUTER_DISABLE_GPT56_EXPLICIT_CACHE) === true) return false;
|
|
19637
|
+
return GPT56_EXPLICIT_CACHE_MODELS.has(model);
|
|
19638
|
+
}
|
|
19639
|
+
function responsesCacheKey(payload, opts, stablePrefix) {
|
|
19640
|
+
const digest = hash(JSON.stringify({
|
|
19641
|
+
namespace: CACHE_KEY_NAMESPACE,
|
|
19642
|
+
model: payload.model,
|
|
19643
|
+
workload: opts.workload,
|
|
19644
|
+
scope: opts.scope ?? "",
|
|
19645
|
+
stablePrefix,
|
|
19646
|
+
tools: payload.tools ?? []
|
|
19647
|
+
}));
|
|
19648
|
+
return `${CACHE_KEY_NAMESPACE}-${digest.slice(0, 48)}`;
|
|
19649
|
+
}
|
|
19650
|
+
/**
|
|
19651
|
+
* Add GPT-5.6 explicit caching only to router-owned REUSABLE-PREFIX payloads.
|
|
19652
|
+
* Public passthrough routes never call this helper, and existing caller
|
|
19653
|
+
* fields always win. Live shape acceptance is pinned by compatibility probe
|
|
19654
|
+
* `gpt56_explicit_cache_breakpoint`.
|
|
19655
|
+
*
|
|
19656
|
+
* **`"conversation"` is deliberately EXCLUDED and left untouched (a no-op),
|
|
19657
|
+
* same as `"passthrough"`/`"one-shot"`.** A live-verified regression: on a
|
|
19658
|
+
* growing multi-turn conversation (Claude Code's translated main loop and
|
|
19659
|
+
* the worker-agent loop, both of which pass `workload: "conversation"`),
|
|
19660
|
+
* marking only the stable SYSTEM block with an explicit breakpoint measured
|
|
19661
|
+
* substantially worse than leaving caching provider-managed and implicit.
|
|
19662
|
+
* Explicit mode is a distinct
|
|
19663
|
+
* caching strategy from Copilot's provider-managed automatic caching, not an
|
|
19664
|
+
* addition to it — turning it on for a request marks only the bytes an
|
|
19665
|
+
* explicit breakpoint names, and the REST of that request's prefix (here,
|
|
19666
|
+
* the entire un-marked growing message history) stops receiving automatic
|
|
19667
|
+
* prefix-growth caching too. Measured on `gpt-5.6-sol` with explicit mode
|
|
19668
|
+
* force-enabled for conversation workloads: turn 1 (cold)
|
|
19669
|
+
* `input_tokens=27038, cache_write=2031, cache_read=0`; turn 2
|
|
19670
|
+
* `input_tokens=27054, cache_read=2031`; turn 3 `input_tokens=27071,
|
|
19671
|
+
* cache_read=2031` — the ~2k-token system block cached once and never grew,
|
|
19672
|
+
* while the other ~25k tokens of accumulating history were recomputed from
|
|
19673
|
+
* scratch on every single turn. `"reusable-prefix"` calls (peer/advisor/
|
|
19674
|
+
* worker-tool/browser-compressor prefixes reused verbatim across many
|
|
19675
|
+
* DISCRETE calls, never a single request whose own history keeps growing)
|
|
19676
|
+
* do not have this failure mode and keep the explicit treatment below.
|
|
19677
|
+
*/
|
|
19678
|
+
function applyResponsesCachePolicy(payload, opts) {
|
|
19679
|
+
logCacheSignature({
|
|
19680
|
+
endpoint: "/responses",
|
|
19681
|
+
model: payload.model,
|
|
19682
|
+
workload: opts.workload,
|
|
19683
|
+
system: opts.stablePrefix ?? payload.instructions,
|
|
19684
|
+
tools: payload.tools,
|
|
19685
|
+
messages: payload.input
|
|
19686
|
+
});
|
|
19687
|
+
if (opts.workload !== "reusable-prefix" || !gpt56ExplicitCacheEnabled(payload.model) || payload.prompt_cache_key !== void 0 || payload.prompt_cache_options !== void 0 || hasResponsesBreakpoint(payload.input)) return payload;
|
|
19688
|
+
const stablePrefix = opts.stablePrefix ?? payload.instructions;
|
|
19689
|
+
const stableBytes = serializedBytes(stablePrefix ?? "") + serializedBytes(payload.tools ?? []);
|
|
19690
|
+
if (!stablePrefix || stableBytes < MIN_CACHEABLE_PREFIX_BYTES) return payload;
|
|
19691
|
+
const input = typeof payload.input === "string" ? [{
|
|
19692
|
+
role: "user",
|
|
19693
|
+
content: payload.input
|
|
19694
|
+
}] : [...payload.input];
|
|
19695
|
+
const stableContent = [{
|
|
19696
|
+
type: "input_text",
|
|
19697
|
+
text: stablePrefix,
|
|
19698
|
+
prompt_cache_breakpoint: { mode: "explicit" }
|
|
19699
|
+
}];
|
|
19700
|
+
let nextInput;
|
|
19701
|
+
let removeInstructions = false;
|
|
19702
|
+
if (payload.instructions === stablePrefix) {
|
|
19703
|
+
nextInput = [{
|
|
19704
|
+
role: "system",
|
|
19705
|
+
content: stableContent
|
|
19706
|
+
}, ...input];
|
|
19707
|
+
removeInstructions = true;
|
|
19708
|
+
} else {
|
|
19709
|
+
const stableSystemIndex = input.findIndex((item) => item.role === "system" && item.content === stablePrefix);
|
|
19710
|
+
if (stableSystemIndex < 0) return payload;
|
|
19711
|
+
nextInput = [...input];
|
|
19712
|
+
nextInput[stableSystemIndex] = {
|
|
19713
|
+
...nextInput[stableSystemIndex],
|
|
19714
|
+
content: stableContent
|
|
19715
|
+
};
|
|
19716
|
+
}
|
|
19717
|
+
const next = {
|
|
19718
|
+
...payload,
|
|
19719
|
+
input: nextInput,
|
|
19720
|
+
prompt_cache_key: responsesCacheKey(payload, opts, stablePrefix),
|
|
19721
|
+
prompt_cache_options: {
|
|
19722
|
+
mode: "explicit",
|
|
19723
|
+
ttl: "30m"
|
|
19724
|
+
}
|
|
19725
|
+
};
|
|
19726
|
+
if (removeInstructions) delete next.instructions;
|
|
19727
|
+
return next;
|
|
19728
|
+
}
|
|
19729
|
+
function itemHasCacheControl(value) {
|
|
19730
|
+
return !!value && typeof value === "object" && value.cache_control !== void 0;
|
|
19731
|
+
}
|
|
19732
|
+
function hasClaudeCacheControl(body) {
|
|
19733
|
+
if (itemHasCacheControl(body.system)) return true;
|
|
19734
|
+
if (Array.isArray(body.system) && body.system.some(itemHasCacheControl)) return true;
|
|
19735
|
+
if (Array.isArray(body.tools) && body.tools.some(itemHasCacheControl)) return true;
|
|
19736
|
+
if (!Array.isArray(body.messages)) return false;
|
|
19737
|
+
return body.messages.some((message) => {
|
|
19738
|
+
if (!message || typeof message !== "object") return false;
|
|
19739
|
+
const content = message.content;
|
|
19740
|
+
return itemHasCacheControl(content) || Array.isArray(content) && content.some(itemHasCacheControl);
|
|
19741
|
+
});
|
|
19742
|
+
}
|
|
19743
|
+
/**
|
|
19744
|
+
* UTF-8 byte length of the tools array alone — the eligibility floor for the
|
|
19745
|
+
* TOOL breakpoint. Marked on the last non-deferred tool, it caches only the
|
|
19746
|
+
* tools prefix (Claude's wire order is tools, then system, then messages), so
|
|
19747
|
+
* its own size — not the combined system+tools size — is what determines
|
|
19748
|
+
* whether that marker is worth spending. See `MIN_CACHEABLE_PREFIX_BYTES`.
|
|
19749
|
+
*/
|
|
19750
|
+
function claudeToolsPrefixBytes(body) {
|
|
19751
|
+
return serializedBytes(body.tools ?? []);
|
|
19752
|
+
}
|
|
19753
|
+
/**
|
|
19754
|
+
* UTF-8 byte length of tools + system combined — the eligibility floor for
|
|
19755
|
+
* the SYSTEM breakpoint. Marked on the last system text block, it caches
|
|
19756
|
+
* everything up to and including system (tools THEN system in wire order),
|
|
19757
|
+
* so the combined size is the right measure — checked SEPARATELY from the
|
|
19758
|
+
* tools-only floor above so a large system prompt behind tiny tools doesn't
|
|
19759
|
+
* smuggle a useless tools-only marker in under the combined total, and a
|
|
19760
|
+
* large tools array behind an empty system doesn't get double-counted as
|
|
19761
|
+
* "small" just because system alone is tiny.
|
|
19762
|
+
*/
|
|
19763
|
+
function claudeSystemPrefixBytes(body) {
|
|
19764
|
+
return claudeToolsPrefixBytes(body) + serializedBytes(body.system ?? "");
|
|
19765
|
+
}
|
|
19766
|
+
function markClaudeSystem(body) {
|
|
19767
|
+
if (typeof body.system === "string" && body.system.length > 0) {
|
|
19768
|
+
body.system = [{
|
|
19769
|
+
type: "text",
|
|
19770
|
+
text: body.system,
|
|
19771
|
+
cache_control: { type: "ephemeral" }
|
|
19772
|
+
}];
|
|
19773
|
+
return true;
|
|
19774
|
+
}
|
|
19775
|
+
if (!Array.isArray(body.system)) return false;
|
|
19776
|
+
for (let index = body.system.length - 1; index >= 0; index--) {
|
|
19777
|
+
const block = body.system[index];
|
|
19778
|
+
if (block && typeof block === "object" && block.type === "text" && typeof block.text === "string") {
|
|
19779
|
+
body.system[index] = {
|
|
19780
|
+
...block,
|
|
19781
|
+
cache_control: { type: "ephemeral" }
|
|
19782
|
+
};
|
|
19783
|
+
return true;
|
|
19784
|
+
}
|
|
19785
|
+
}
|
|
19786
|
+
return false;
|
|
19787
|
+
}
|
|
19788
|
+
function markClaudeTool(body) {
|
|
19789
|
+
if (!Array.isArray(body.tools)) return false;
|
|
19790
|
+
for (let index = body.tools.length - 1; index >= 0; index--) {
|
|
19791
|
+
const tool = body.tools[index];
|
|
19792
|
+
if (tool && typeof tool === "object" && tool.defer_loading !== true) {
|
|
19793
|
+
body.tools[index] = {
|
|
19794
|
+
...tool,
|
|
19795
|
+
cache_control: { type: "ephemeral" }
|
|
19796
|
+
};
|
|
19797
|
+
return true;
|
|
19798
|
+
}
|
|
19799
|
+
}
|
|
19800
|
+
return false;
|
|
19801
|
+
}
|
|
19802
|
+
/**
|
|
19803
|
+
* Apply the bounded Claude anchor policy to router-generated Messages bodies.
|
|
19804
|
+
* Caller-owned marker layouts are returned byte-for-byte unchanged.
|
|
19805
|
+
*
|
|
19806
|
+
* Marks at most TWO breakpoints — the last non-deferred tool and the stable
|
|
19807
|
+
* system boundary — each gated on its OWN eligibility check
|
|
19808
|
+
* (`claudeToolsPrefixBytes` / `claudeSystemPrefixBytes`) rather than one
|
|
19809
|
+
* combined check, so a large system prompt behind tiny tools doesn't also
|
|
19810
|
+
* mark a tools breakpoint too small to be worth a marker slot, and vice
|
|
19811
|
+
* versa. Anthropic's own hard ceiling is FOUR `cache_control` blocks per
|
|
19812
|
+
* request (probe `cache_control_marker_limit_5`); this policy only ever
|
|
19813
|
+
* spends up to two of them (`hasClaudeCacheControl` already refuses to run
|
|
19814
|
+
* at all once the caller has marked anything itself, so the two never
|
|
19815
|
+
* combine with a caller-owned marker to approach that ceiling).
|
|
19816
|
+
*
|
|
19817
|
+
* There used to be a third, message-level marking path gated on
|
|
19818
|
+
* `opts.workload === "conversation"`. It was removed as dead code: every
|
|
19819
|
+
* production call site of this function (`src/routes/mcp/handler.ts`,
|
|
19820
|
+
* `src/services/advisor/advisor.ts`) passes `workload: "reusable-prefix"`,
|
|
19821
|
+
* so the per-message branch never ran outside its own unit test, which gave
|
|
19822
|
+
* false confidence that production traffic exercised it. `CacheWorkload`
|
|
19823
|
+
* keeps `"conversation"` as a shared enum value — the Responses-side policy,
|
|
19824
|
+
* which has no per-message logic of its own, still uses it — so passing it
|
|
19825
|
+
* here remains type-valid; it now behaves identically to
|
|
19826
|
+
* `"reusable-prefix"`.
|
|
19827
|
+
*/
|
|
19828
|
+
function applyClaudeCachePolicy(rawBody, opts) {
|
|
19829
|
+
if (opts.workload === "passthrough" || opts.workload === "one-shot" || parseBoolEnv(process.env.GH_ROUTER_DISABLE_CLAUDE_CACHE_POLICY) === true) return rawBody;
|
|
19830
|
+
let parsed;
|
|
19831
|
+
try {
|
|
19832
|
+
parsed = JSON.parse(rawBody);
|
|
19833
|
+
} catch {
|
|
19834
|
+
return rawBody;
|
|
19835
|
+
}
|
|
19836
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return rawBody;
|
|
19837
|
+
const body = parsed;
|
|
19838
|
+
if (typeof body.model !== "string" || !body.model.startsWith("claude-") || hasClaudeCacheControl(body)) return rawBody;
|
|
19839
|
+
const toolsEligible = claudeToolsPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
|
|
19840
|
+
const systemEligible = claudeSystemPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
|
|
19841
|
+
if (!toolsEligible && !systemEligible) return rawBody;
|
|
19842
|
+
let markers = 0;
|
|
19843
|
+
if (toolsEligible && markClaudeTool(body)) markers++;
|
|
19844
|
+
if (systemEligible && markClaudeSystem(body)) markers++;
|
|
19845
|
+
if (markers === 0) return rawBody;
|
|
19846
|
+
logCacheSignature({
|
|
19847
|
+
endpoint: "/messages",
|
|
19848
|
+
model: body.model,
|
|
19849
|
+
workload: opts.workload,
|
|
19850
|
+
system: body.system,
|
|
19851
|
+
tools: body.tools,
|
|
19852
|
+
messages: body.messages
|
|
19853
|
+
});
|
|
19854
|
+
return JSON.stringify(body);
|
|
19855
|
+
}
|
|
19856
|
+
//#endregion
|
|
19500
19857
|
//#region src/lib/vision-preflight.ts
|
|
19501
19858
|
/**
|
|
19502
19859
|
* Outbound vision handling.
|
|
@@ -20421,7 +20778,7 @@ async function callViaChat(model, systemPrompt, userMessage, tool, signal) {
|
|
|
20421
20778
|
* Image parts use `input_image` (vs chat's `image_url`) — see
|
|
20422
20779
|
* `toResponsesContent`. */
|
|
20423
20780
|
async function callViaResponses(model, systemPrompt, userMessage, tool, signal) {
|
|
20424
|
-
const payload = {
|
|
20781
|
+
const payload = applyResponsesCachePolicy({
|
|
20425
20782
|
model,
|
|
20426
20783
|
stream: false,
|
|
20427
20784
|
input: [{
|
|
@@ -20441,7 +20798,10 @@ async function callViaResponses(model, systemPrompt, userMessage, tool, signal)
|
|
|
20441
20798
|
type: "function",
|
|
20442
20799
|
name: tool.name
|
|
20443
20800
|
}
|
|
20444
|
-
}
|
|
20801
|
+
}, {
|
|
20802
|
+
workload: "reusable-prefix",
|
|
20803
|
+
stablePrefix: systemPrompt
|
|
20804
|
+
});
|
|
20445
20805
|
const resp = await createResponses(payload, void 0, signal, true);
|
|
20446
20806
|
const output = Array.isArray(resp.output) ? resp.output : [];
|
|
20447
20807
|
for (const item of output) {
|
|
@@ -23897,8 +24257,8 @@ const FALLBACK_TOKEN_PRICES = Object.freeze({
|
|
|
23897
24257
|
out: 2500
|
|
23898
24258
|
},
|
|
23899
24259
|
"gpt-5.6-sol": {
|
|
23900
|
-
in:
|
|
23901
|
-
out:
|
|
24260
|
+
in: 200,
|
|
24261
|
+
out: 1e3
|
|
23902
24262
|
},
|
|
23903
24263
|
"grok-4.5": {
|
|
23904
24264
|
in: 200,
|
|
@@ -23909,8 +24269,8 @@ const FALLBACK_TOKEN_PRICES = Object.freeze({
|
|
|
23909
24269
|
out: 3e3
|
|
23910
24270
|
},
|
|
23911
24271
|
"gemini-3.6-flash": {
|
|
23912
|
-
in:
|
|
23913
|
-
out:
|
|
24272
|
+
in: 75,
|
|
24273
|
+
out: 375
|
|
23914
24274
|
},
|
|
23915
24275
|
"gemini-3.5-flash": {
|
|
23916
24276
|
in: 150,
|
|
@@ -24491,6 +24851,10 @@ function neutralToolsToResponses(tools) {
|
|
|
24491
24851
|
/** Assemble the full Responses payload from the neutral request shape. */
|
|
24492
24852
|
function assembleResponsesPayload(opts) {
|
|
24493
24853
|
const input = [];
|
|
24854
|
+
if (opts.dynamicInstructions) input.push({
|
|
24855
|
+
role: "system",
|
|
24856
|
+
content: opts.dynamicInstructions
|
|
24857
|
+
});
|
|
24494
24858
|
for (const m of opts.messages) for (const item of neutralMessageToResponsesInput(m)) input.push(item);
|
|
24495
24859
|
const payload = {
|
|
24496
24860
|
model: opts.model,
|
|
@@ -24507,7 +24871,10 @@ function assembleResponsesPayload(opts) {
|
|
|
24507
24871
|
if (typeof opts.maxOutputTokens === "number" && opts.maxOutputTokens > 0) payload.max_output_tokens = Math.max(opts.maxOutputTokens, RESPONSES_MIN_MAX_OUTPUT_TOKENS);
|
|
24508
24872
|
if (opts.stopSequences && opts.stopSequences.length > 0) payload.stop = [...opts.stopSequences];
|
|
24509
24873
|
if (opts.parallelToolCalls === false) payload.parallel_tool_calls = false;
|
|
24510
|
-
return payload
|
|
24874
|
+
return opts.cachePolicy ? applyResponsesCachePolicy(payload, {
|
|
24875
|
+
...opts.cachePolicy,
|
|
24876
|
+
stablePrefix: opts.cachePolicy.stablePrefix ?? opts.instructions
|
|
24877
|
+
}) : payload;
|
|
24511
24878
|
}
|
|
24512
24879
|
//#endregion
|
|
24513
24880
|
//#region src/lib/worker-agent/context-budget.ts
|
|
@@ -25042,7 +25409,10 @@ function mapResponsesUsage(u) {
|
|
|
25042
25409
|
prompt_tokens: u.input_tokens ?? 0,
|
|
25043
25410
|
completion_tokens: u.output_tokens ?? 0,
|
|
25044
25411
|
total_tokens: u.total_tokens ?? 0,
|
|
25045
|
-
prompt_tokens_details: u.input_tokens_details
|
|
25412
|
+
prompt_tokens_details: u.input_tokens_details != null ? {
|
|
25413
|
+
cached_tokens: u.input_tokens_details.cached_tokens ?? 0,
|
|
25414
|
+
cache_write_tokens: u.input_tokens_details.cache_write_tokens ?? u.input_tokens_details.cache_creation_tokens ?? 0
|
|
25415
|
+
} : void 0
|
|
25046
25416
|
};
|
|
25047
25417
|
}
|
|
25048
25418
|
/**
|
|
@@ -25291,6 +25661,7 @@ function buildResponsesPayload(context, resolved) {
|
|
|
25291
25661
|
messages,
|
|
25292
25662
|
tools: piToolsToNeutral(context.tools),
|
|
25293
25663
|
reasoningEffort: resolved.thinking,
|
|
25664
|
+
cachePolicy: { workload: "conversation" },
|
|
25294
25665
|
stream: true
|
|
25295
25666
|
});
|
|
25296
25667
|
}
|
|
@@ -25513,12 +25884,13 @@ function emptyUsage() {
|
|
|
25513
25884
|
}
|
|
25514
25885
|
function deriveUsage(u) {
|
|
25515
25886
|
if (!u) return emptyUsage();
|
|
25887
|
+
const normalized = normalizeOpenAIUsage(u);
|
|
25516
25888
|
return {
|
|
25517
|
-
input:
|
|
25518
|
-
output:
|
|
25519
|
-
cacheRead:
|
|
25520
|
-
cacheWrite:
|
|
25521
|
-
totalTokens:
|
|
25889
|
+
input: normalized.uncachedInput,
|
|
25890
|
+
output: normalized.output,
|
|
25891
|
+
cacheRead: normalized.cacheRead,
|
|
25892
|
+
cacheWrite: normalized.cacheWrite,
|
|
25893
|
+
totalTokens: normalized.totalTokens,
|
|
25522
25894
|
cost: {
|
|
25523
25895
|
input: 0,
|
|
25524
25896
|
output: 0,
|
|
@@ -26401,6 +26773,67 @@ function capToolResultText(content, capBytes) {
|
|
|
26401
26773
|
];
|
|
26402
26774
|
}
|
|
26403
26775
|
//#endregion
|
|
26776
|
+
//#region src/lib/launch-registry.ts
|
|
26777
|
+
/**
|
|
26778
|
+
* Register a new authenticated launch (a `github-router claude` process, or
|
|
26779
|
+
* `serve`'s per-repo session) in the keyed registry. Returns the stored
|
|
26780
|
+
* entry, including the generated `launchId` when the caller didn't supply
|
|
26781
|
+
* one.
|
|
26782
|
+
*
|
|
26783
|
+
* Callers are expected to mint `nonce` and `secret` as independent random
|
|
26784
|
+
* tokens (see `src/claude.ts` / `src/lib/serve/enhancements.ts`) — this
|
|
26785
|
+
* function stores whatever it's given without generating credentials
|
|
26786
|
+
* itself, so a caller cannot accidentally rely on it for randomness.
|
|
26787
|
+
*/
|
|
26788
|
+
function registerLaunch(params) {
|
|
26789
|
+
const entry = {
|
|
26790
|
+
launchId: params.launchId ?? randomUUID(),
|
|
26791
|
+
nonce: params.nonce,
|
|
26792
|
+
secret: params.secret,
|
|
26793
|
+
profileId: params.profileId,
|
|
26794
|
+
allowedGroups: params.allowedGroups,
|
|
26795
|
+
allowedPersonas: params.allowedPersonas,
|
|
26796
|
+
createdAt: Date.now()
|
|
26797
|
+
};
|
|
26798
|
+
state.launchRegistry.set(entry.launchId, entry);
|
|
26799
|
+
return entry;
|
|
26800
|
+
}
|
|
26801
|
+
/** Remove one launch's entry. Idempotent — removing an already-removed or
|
|
26802
|
+
* never-registered id is a no-op. Called from the launch's own cleanup
|
|
26803
|
+
* path so a torn-down session's credentials stop authenticating. */
|
|
26804
|
+
function unregisterLaunch(launchId) {
|
|
26805
|
+
state.launchRegistry.delete(launchId);
|
|
26806
|
+
}
|
|
26807
|
+
/**
|
|
26808
|
+
* Constant-time string compare. Per-launch credentials are random tokens,
|
|
26809
|
+
* not secrets an attacker gets many guesses at over the network within one
|
|
26810
|
+
* process lifetime, so timing attacks aren't a realistic concern here — but
|
|
26811
|
+
* it costs nothing and matches the prior nonce-compare's posture.
|
|
26812
|
+
*/
|
|
26813
|
+
function constantTimeStringEqual(a, b) {
|
|
26814
|
+
if (a.length !== b.length) return false;
|
|
26815
|
+
try {
|
|
26816
|
+
return timingSafeEqual(Buffer.from(a), Buffer.from(b));
|
|
26817
|
+
} catch {
|
|
26818
|
+
return false;
|
|
26819
|
+
}
|
|
26820
|
+
}
|
|
26821
|
+
/**
|
|
26822
|
+
* Find the launch whose `/mcp` bearer (`nonce`) matches. Linear scan over
|
|
26823
|
+
* `state.launchRegistry` — expected to hold a handful of entries at most
|
|
26824
|
+
* (one per concurrently running `claude`/`serve` session), so this is not a
|
|
26825
|
+
* hot-path concern. Returns undefined (never throws) when nothing matches,
|
|
26826
|
+
* including when the registry is empty (the "not enabled" case).
|
|
26827
|
+
*/
|
|
26828
|
+
function findLaunchByNonce(nonce) {
|
|
26829
|
+
for (const entry of state.launchRegistry.values()) if (constantTimeStringEqual(entry.nonce, nonce)) return entry;
|
|
26830
|
+
}
|
|
26831
|
+
/** Find the launch whose `/v1/messages` identity-preflight bearer
|
|
26832
|
+
* (`secret`) matches. Mirrors `findLaunchByNonce`. */
|
|
26833
|
+
function findLaunchBySecret(secret) {
|
|
26834
|
+
for (const entry of state.launchRegistry.values()) if (constantTimeStringEqual(entry.secret, secret)) return entry;
|
|
26835
|
+
}
|
|
26836
|
+
//#endregion
|
|
26404
26837
|
//#region src/lib/peer-attachments.ts
|
|
26405
26838
|
/**
|
|
26406
26839
|
* Server-side image loading for peer-critic attachments (`imagePaths`).
|
|
@@ -27268,6 +27701,67 @@ function generalPurposeFastModel() {
|
|
|
27268
27701
|
minContextTokens: ONE_M_TOKENS
|
|
27269
27702
|
});
|
|
27270
27703
|
}
|
|
27704
|
+
const FAST_SCOUT_MODEL = "gpt-5.6-luna";
|
|
27705
|
+
const FAST_IMPLEMENTER_MODEL = "gpt-5.6-luna";
|
|
27706
|
+
/** Grok 4.6 advertises 500K total context / 372K max prompt, so it remains bare
|
|
27707
|
+
* and is gated by max_prompt_tokens rather than the 1M floor. */
|
|
27708
|
+
const FAST_REVIEWER_MODEL = "grok-4.6";
|
|
27709
|
+
const FAST_PLANNER_MODEL = "gpt-5.6-sol";
|
|
27710
|
+
const FAST_ORACLE_MODEL = "claude-opus-5";
|
|
27711
|
+
/** Fixed effort pins for the fast profile. */
|
|
27712
|
+
const FAST_SCOUT_EFFORT = "high";
|
|
27713
|
+
const FAST_REVIEWER_EFFORT = "medium";
|
|
27714
|
+
const FAST_PLANNER_EFFORT = "high";
|
|
27715
|
+
function fastScoutModel() {
|
|
27716
|
+
return firstPresentInCatalog([FAST_SCOUT_MODEL], {
|
|
27717
|
+
requireToolCalls: true,
|
|
27718
|
+
minContextTokens: ONE_M_TOKENS
|
|
27719
|
+
});
|
|
27720
|
+
}
|
|
27721
|
+
function fastImplementerModel() {
|
|
27722
|
+
return firstPresentInCatalog([FAST_IMPLEMENTER_MODEL], {
|
|
27723
|
+
requireToolCalls: true,
|
|
27724
|
+
minContextTokens: ONE_M_TOKENS
|
|
27725
|
+
});
|
|
27726
|
+
}
|
|
27727
|
+
function fastPlannerModel() {
|
|
27728
|
+
const id = firstPresentInCatalog([FAST_PLANNER_MODEL], {
|
|
27729
|
+
requireToolCalls: true,
|
|
27730
|
+
minContextTokens: ONE_M_TOKENS
|
|
27731
|
+
});
|
|
27732
|
+
if (!id) return void 0;
|
|
27733
|
+
const found = state.models?.data.find((m) => m.id === id);
|
|
27734
|
+
const efforts = found?.capabilities?.supports?.reasoning_effort;
|
|
27735
|
+
if (!Array.isArray(efforts) || !efforts.includes("high")) return void 0;
|
|
27736
|
+
if (pickEndpoint(found) !== "responses") return void 0;
|
|
27737
|
+
return id;
|
|
27738
|
+
}
|
|
27739
|
+
/** Gate Grok on the prompt limit that actually constrains pasted review input. */
|
|
27740
|
+
function fastReviewerModel() {
|
|
27741
|
+
const models = state.models?.data;
|
|
27742
|
+
if (!models) return void 0;
|
|
27743
|
+
const found = models.find((m) => m.id === FAST_REVIEWER_MODEL);
|
|
27744
|
+
if (!found) return void 0;
|
|
27745
|
+
if (found.capabilities?.supports?.tool_calls !== true) return void 0;
|
|
27746
|
+
const efforts = found.capabilities?.supports?.reasoning_effort;
|
|
27747
|
+
if (!Array.isArray(efforts) || !efforts.includes("medium")) return void 0;
|
|
27748
|
+
if ((found.capabilities?.limits?.max_prompt_tokens ?? 0) < 2e5) return void 0;
|
|
27749
|
+
if (pickEndpoint(found) !== "responses") return void 0;
|
|
27750
|
+
return FAST_REVIEWER_MODEL;
|
|
27751
|
+
}
|
|
27752
|
+
/** Exact Opus 5 only: the fast Oracle never inherits standard opus_critic's
|
|
27753
|
+
* older-family fallback. */
|
|
27754
|
+
function fastOracleModel() {
|
|
27755
|
+
const found = state.models?.data.find((m) => m.id === FAST_ORACLE_MODEL);
|
|
27756
|
+
if (!found) return void 0;
|
|
27757
|
+
if ((found.capabilities?.limits?.max_context_window_tokens ?? 0) < 1e6) return void 0;
|
|
27758
|
+
if ((found.capabilities?.limits?.max_prompt_tokens ?? 0) <= 0) return void 0;
|
|
27759
|
+
const efforts = found.capabilities?.supports?.reasoning_effort;
|
|
27760
|
+
if (!Array.isArray(efforts) || !efforts.includes("high")) return void 0;
|
|
27761
|
+
if (found.capabilities?.supports?.adaptive_thinking !== true) return void 0;
|
|
27762
|
+
if (!(found.supported_endpoints ?? []).some((endpoint) => endpoint === "/messages" || endpoint === "/v1/messages")) return void 0;
|
|
27763
|
+
return FAST_ORACLE_MODEL;
|
|
27764
|
+
}
|
|
27271
27765
|
/**
|
|
27272
27766
|
* Gate for the worker tools (`explore`, `review`, `implement`).
|
|
27273
27767
|
*
|
|
@@ -27492,40 +27986,29 @@ function isLoopbackHost(host) {
|
|
|
27492
27986
|
const hostname = idx >= 0 ? host.slice(0, idx) : host;
|
|
27493
27987
|
return hostname === "127.0.0.1" || hostname === "localhost";
|
|
27494
27988
|
}
|
|
27495
|
-
/**
|
|
27496
|
-
* Constant-time bearer compare. Random per-launch nonces aren't really
|
|
27497
|
-
* timing-attackable in practice, but this costs nothing.
|
|
27498
|
-
*/
|
|
27499
|
-
function nonceMatches(provided, expected) {
|
|
27500
|
-
if (provided.length !== expected.length) return false;
|
|
27501
|
-
const a = Buffer.from(provided);
|
|
27502
|
-
const b = Buffer.from(expected);
|
|
27503
|
-
try {
|
|
27504
|
-
return timingSafeEqual(a, b);
|
|
27505
|
-
} catch {
|
|
27506
|
-
return false;
|
|
27507
|
-
}
|
|
27508
|
-
}
|
|
27509
27989
|
function checkAuth(c) {
|
|
27510
27990
|
if (!isLoopbackHost(c.req.header("host"))) return {
|
|
27511
27991
|
ok: false,
|
|
27512
27992
|
status: 403,
|
|
27513
27993
|
reason: "non-loopback Host header rejected"
|
|
27514
27994
|
};
|
|
27515
|
-
|
|
27516
|
-
if (!expected) return {
|
|
27995
|
+
if (state.launchRegistry.size === 0) return {
|
|
27517
27996
|
ok: false,
|
|
27518
27997
|
status: 401,
|
|
27519
27998
|
reason: "/mcp not enabled in this proxy session"
|
|
27520
27999
|
};
|
|
27521
28000
|
const auth = c.req.header("authorization") ?? "";
|
|
27522
28001
|
const m = /^Bearer\s+(.+)$/i.exec(auth);
|
|
27523
|
-
|
|
28002
|
+
const launch = m ? findLaunchByNonce(m[1]) : void 0;
|
|
28003
|
+
if (!launch) return {
|
|
27524
28004
|
ok: false,
|
|
27525
28005
|
status: 401,
|
|
27526
28006
|
reason: "missing or invalid Authorization bearer"
|
|
27527
28007
|
};
|
|
27528
|
-
return {
|
|
28008
|
+
return {
|
|
28009
|
+
ok: true,
|
|
28010
|
+
launch
|
|
28011
|
+
};
|
|
27529
28012
|
}
|
|
27530
28013
|
/**
|
|
27531
28014
|
* opus_critic's effective model, resolved against the live catalog.
|
|
@@ -27566,8 +28049,54 @@ function activePersonas() {
|
|
|
27566
28049
|
};
|
|
27567
28050
|
});
|
|
27568
28051
|
}
|
|
27569
|
-
function
|
|
27570
|
-
|
|
28052
|
+
function oracleToolEntry() {
|
|
28053
|
+
return {
|
|
28054
|
+
name: "oracle",
|
|
28055
|
+
description: "Fast-profile last-resort guidance from exact Opus 5 (1M context) at high effort. Stateless and tool-less: pass complete context plus one precise query only after the primary Luna path, Advisor, and the relevant reviewer/planner path remain stuck. It can advise or request missing information; it cannot inspect the repo, execute, merge, or authorize actions.",
|
|
28056
|
+
inputSchema: {
|
|
28057
|
+
type: "object",
|
|
28058
|
+
required: ["query", "context"],
|
|
28059
|
+
additionalProperties: false,
|
|
28060
|
+
properties: {
|
|
28061
|
+
query: {
|
|
28062
|
+
type: "string",
|
|
28063
|
+
description: "One precise unresolved question."
|
|
28064
|
+
},
|
|
28065
|
+
context: {
|
|
28066
|
+
type: "string",
|
|
28067
|
+
description: "Complete evidence and constraints needed to answer cold-start."
|
|
28068
|
+
}
|
|
28069
|
+
}
|
|
28070
|
+
}
|
|
28071
|
+
};
|
|
28072
|
+
}
|
|
28073
|
+
function toolEntries(scope, launch) {
|
|
28074
|
+
if (launch.profileId === "fast") {
|
|
28075
|
+
const entries = [];
|
|
28076
|
+
if ((scope === "all" || scope === "peers") && launch.allowedGroups?.has("peers") && launch.allowedPersonas?.has("oracle") && fastOracleModel()) entries.push(oracleToolEntry());
|
|
28077
|
+
for (const tool of NON_PERSONA_MCP_TOOLS) {
|
|
28078
|
+
if (scope !== "all" && tool.group !== scope) continue;
|
|
28079
|
+
if (!launch.allowedGroups?.has(tool.group)) continue;
|
|
28080
|
+
if (tool.group === "search") {
|
|
28081
|
+
entries.push({
|
|
28082
|
+
name: tool.toolNameHttp,
|
|
28083
|
+
description: tool.description,
|
|
28084
|
+
inputSchema: tool.inputSchema
|
|
28085
|
+
});
|
|
28086
|
+
continue;
|
|
28087
|
+
}
|
|
28088
|
+
if (tool.group !== "browser" || !browserToolsEnabled()) continue;
|
|
28089
|
+
if (tool.capability === "browser_compound" && !browserCompoundToolsEnabled()) continue;
|
|
28090
|
+
if (tool.capability === "browser_power" && !browserPowerToolsEnabled()) continue;
|
|
28091
|
+
entries.push({
|
|
28092
|
+
name: tool.toolNameHttp,
|
|
28093
|
+
description: tool.description,
|
|
28094
|
+
inputSchema: tool.inputSchema
|
|
28095
|
+
});
|
|
28096
|
+
}
|
|
28097
|
+
return entries;
|
|
28098
|
+
}
|
|
28099
|
+
const personaEntries = (!launch.allowedGroups || launch.allowedGroups.has("peers")) && (scope === "all" || scope === "peers") ? activePersonas().filter((p) => !launch.allowedPersonas || launch.allowedPersonas.has(p.toolNameHttp)).map((p) => ({
|
|
27571
28100
|
name: p.toolNameHttp,
|
|
27572
28101
|
description: p.description,
|
|
27573
28102
|
inputSchema: {
|
|
@@ -27598,6 +28127,7 @@ function toolEntries(scope) {
|
|
|
27598
28127
|
})) : [];
|
|
27599
28128
|
const nonPersonaEntries = NON_PERSONA_MCP_TOOLS.filter((t) => {
|
|
27600
28129
|
if (scope !== "all" && t.group !== scope) return false;
|
|
28130
|
+
if (launch.allowedGroups && !launch.allowedGroups.has(t.group)) return false;
|
|
27601
28131
|
if (t.capability === "worker") return workerToolsEnabled();
|
|
27602
28132
|
if (t.capability === "browse_agent") return browseAgentEnabled();
|
|
27603
28133
|
if (t.capability === "stand_in") return standInToolEnabled();
|
|
@@ -27839,7 +28369,7 @@ function jsonPathPreflightCap(body, scope) {
|
|
|
27839
28369
|
async function dispatchModelCall(args) {
|
|
27840
28370
|
const resolvedModel = resolveModel(args.model);
|
|
27841
28371
|
if (args.endpoint === "/v1/responses") {
|
|
27842
|
-
const payload = {
|
|
28372
|
+
const payload = applyResponsesCachePolicy({
|
|
27843
28373
|
model: resolvedModel,
|
|
27844
28374
|
instructions: args.instructions,
|
|
27845
28375
|
input: [{
|
|
@@ -27854,7 +28384,7 @@ async function dispatchModelCall(args) {
|
|
|
27854
28384
|
}],
|
|
27855
28385
|
stream: false,
|
|
27856
28386
|
reasoning: { effort: args.effort }
|
|
27857
|
-
};
|
|
28387
|
+
}, { workload: "reusable-prefix" });
|
|
27858
28388
|
return extractResponsesText(await withTransientRetry(() => createResponses(payload, void 0, args.signal), {
|
|
27859
28389
|
signal: args.signal,
|
|
27860
28390
|
label: resolvedModel
|
|
@@ -27862,7 +28392,7 @@ async function dispatchModelCall(args) {
|
|
|
27862
28392
|
}
|
|
27863
28393
|
if (args.endpoint === "/v1/messages") {
|
|
27864
28394
|
const maxTokens = args.effort === "low" ? 4096 : args.effort === "medium" ? 8192 : args.effort === "high" ? 16384 : 32768;
|
|
27865
|
-
const body = JSON.stringify({
|
|
28395
|
+
const body = applyClaudeCachePolicy(JSON.stringify({
|
|
27866
28396
|
model: resolvedModel,
|
|
27867
28397
|
max_tokens: maxTokens,
|
|
27868
28398
|
system: args.instructions,
|
|
@@ -27885,7 +28415,7 @@ async function dispatchModelCall(args) {
|
|
|
27885
28415
|
role: "user",
|
|
27886
28416
|
content: args.userText
|
|
27887
28417
|
}]
|
|
27888
|
-
});
|
|
28418
|
+
}), { workload: "reusable-prefix" });
|
|
27889
28419
|
return extractMessagesText(await (await withTransientRetry(() => createMessages(body, void 0, args.signal), {
|
|
27890
28420
|
signal: args.signal,
|
|
27891
28421
|
label: resolvedModel
|
|
@@ -27970,16 +28500,87 @@ function applySessionWorkspace(args, sessionWorkspace, tool) {
|
|
|
27970
28500
|
}
|
|
27971
28501
|
return "absent";
|
|
27972
28502
|
}
|
|
27973
|
-
async function handleToolsCall(body, scope, sessionWorkspace) {
|
|
28503
|
+
async function handleToolsCall(body, scope, launch, sessionWorkspace) {
|
|
27974
28504
|
const params = body.params ?? {};
|
|
27975
28505
|
const name = typeof params.name === "string" ? params.name : "";
|
|
27976
28506
|
const args = params.arguments ?? {};
|
|
27977
28507
|
if (!name) return rpcError(body.id, RPC_INVALID_PARAMS, "tools/call missing name");
|
|
28508
|
+
if (launch.profileId === "fast" && name === "oracle") {
|
|
28509
|
+
if (scope !== "all" && scope !== "peers" || !launch.allowedGroups?.has("peers") || !launch.allowedPersonas?.has("oracle") || !fastOracleModel()) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
28510
|
+
const query = typeof args.query === "string" ? args.query.trim() : "";
|
|
28511
|
+
const context = typeof args.context === "string" ? args.context.trim() : "";
|
|
28512
|
+
if (!query || !context) return rpcError(body.id, RPC_INVALID_PARAMS, "tools/call: oracle requires non-empty arguments.query and arguments.context");
|
|
28513
|
+
const MAX_ORACLE_INPUT_BYTES = 262144;
|
|
28514
|
+
const oracleInput = `Query:\n${query}\n\nContext:\n${context}`;
|
|
28515
|
+
const inputBytes = Buffer.byteLength(oracleInput, "utf8");
|
|
28516
|
+
if (inputBytes > MAX_ORACLE_INPUT_BYTES) return rpcResult(body.id, toolError(`pre-flight rejected: oracle input is ${inputBytes} bytes, over the ${MAX_ORACLE_INPUT_BYTES}-byte cap; narrow the context without silently truncating it`));
|
|
28517
|
+
const oraclePersona = {
|
|
28518
|
+
agentName: "oracle",
|
|
28519
|
+
toolNameHttp: "oracle",
|
|
28520
|
+
model: "claude-opus-5",
|
|
28521
|
+
endpoint: "/v1/messages",
|
|
28522
|
+
description: "Fast-profile Oracle",
|
|
28523
|
+
baseInstructions: "You are Oracle, a stateless last-resort consultant. You have no tools or repository access. Answer only from the supplied context. Give focused guidance or ask for the exact missing information. Never claim to execute, approve, merge, or authorize an action.",
|
|
28524
|
+
agentPrompt: "",
|
|
28525
|
+
writeCapable: false,
|
|
28526
|
+
requiresHttp: true,
|
|
28527
|
+
allowedEfforts: ["high"],
|
|
28528
|
+
defaultEffort: "high"
|
|
28529
|
+
};
|
|
28530
|
+
const overflow = await predictedWindowOverflow(oraclePersona, oracleInput, void 0);
|
|
28531
|
+
if (overflow) return rpcResult(body.id, toolError(overflow));
|
|
28532
|
+
const release = acquireInFlightSlot();
|
|
28533
|
+
if (!release) return rpcResult(body.id, toolError(`Peer MCP queue full (${MAX_INFLIGHT_TOOLS_CALL} in-flight). Retry shortly.`));
|
|
28534
|
+
const startedAt = Date.now();
|
|
28535
|
+
const abortKey = body.id !== void 0 && body.id !== null ? body.id : void 0;
|
|
28536
|
+
const aborter = new AbortController();
|
|
28537
|
+
const inflightEntry = {
|
|
28538
|
+
aborter,
|
|
28539
|
+
release
|
|
28540
|
+
};
|
|
28541
|
+
if (abortKey !== void 0) inflightAborts.set(abortKey, inflightEntry);
|
|
28542
|
+
try {
|
|
28543
|
+
const text = await dispatchModelCall({
|
|
28544
|
+
model: "claude-opus-5",
|
|
28545
|
+
endpoint: "/v1/messages",
|
|
28546
|
+
instructions: oraclePersona.baseInstructions,
|
|
28547
|
+
userText: oracleInput,
|
|
28548
|
+
effort: "high",
|
|
28549
|
+
signal: aborter.signal
|
|
28550
|
+
});
|
|
28551
|
+
logTelemetry({
|
|
28552
|
+
name: "oracle",
|
|
28553
|
+
model: "claude-opus-5",
|
|
28554
|
+
durationMs: Date.now() - startedAt,
|
|
28555
|
+
result: "ok"
|
|
28556
|
+
});
|
|
28557
|
+
return rpcResult(body.id, { content: [{
|
|
28558
|
+
type: "text",
|
|
28559
|
+
text
|
|
28560
|
+
}] });
|
|
28561
|
+
} catch (err) {
|
|
28562
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
28563
|
+
logTelemetry({
|
|
28564
|
+
name: "oracle",
|
|
28565
|
+
model: "claude-opus-5",
|
|
28566
|
+
durationMs: Date.now() - startedAt,
|
|
28567
|
+
result: "exception",
|
|
28568
|
+
errorMessage: message
|
|
28569
|
+
});
|
|
28570
|
+
return rpcResult(body.id, toolError(`oracle failed: ${message}`));
|
|
28571
|
+
} finally {
|
|
28572
|
+
if (abortKey !== void 0 && inflightAborts.get(abortKey) === inflightEntry) inflightAborts.delete(abortKey);
|
|
28573
|
+
release();
|
|
28574
|
+
}
|
|
28575
|
+
}
|
|
28576
|
+
if (launch.profileId === "fast" && PERSONAS_READ.some((p) => p.toolNameHttp === name)) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
27978
28577
|
const persona = activePersonas().find((p) => p.toolNameHttp === name);
|
|
27979
28578
|
const nonPersonaTool = persona ? void 0 : NON_PERSONA_MCP_TOOLS.find((t) => t.toolNameHttp === name);
|
|
27980
28579
|
if (!persona && !nonPersonaTool) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
27981
28580
|
const toolGroup = persona ? "peers" : nonPersonaTool.group;
|
|
27982
28581
|
if (scope !== "all" && toolGroup !== scope) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
28582
|
+
if (launch.allowedGroups && !launch.allowedGroups.has(toolGroup)) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
28583
|
+
if (persona && launch.allowedPersonas && !launch.allowedPersonas.has(persona.toolNameHttp)) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
27983
28584
|
if (nonPersonaTool && nonPersonaTool.capability === "worker" && !workerToolsEnabled()) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
27984
28585
|
if (nonPersonaTool && nonPersonaTool.capability === "browse_agent" && !browseAgentEnabled()) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
27985
28586
|
if (nonPersonaTool && nonPersonaTool.capability === "stand_in" && !standInToolEnabled()) return rpcError(body.id, RPC_METHOD_NOT_FOUND, `tools/call: unknown tool "${name}"`);
|
|
@@ -28087,7 +28688,7 @@ function handleCancelledNotification(body) {
|
|
|
28087
28688
|
}
|
|
28088
28689
|
cancelInflight(requestId, "client requested cancellation");
|
|
28089
28690
|
}
|
|
28090
|
-
async function handleRpc(_c, body, scope, sessionWorkspace) {
|
|
28691
|
+
async function handleRpc(_c, body, scope, launch, sessionWorkspace) {
|
|
28091
28692
|
if (body === null || typeof body !== "object" || Array.isArray(body)) return {
|
|
28092
28693
|
status: 200,
|
|
28093
28694
|
body: rpcError(null, RPC_INVALID_REQUEST, "jsonrpc 2.0 envelope required")
|
|
@@ -28129,7 +28730,7 @@ async function handleRpc(_c, body, scope, sessionWorkspace) {
|
|
|
28129
28730
|
};
|
|
28130
28731
|
return {
|
|
28131
28732
|
status: 200,
|
|
28132
|
-
body: rpcResult(body.id, { tools: toolEntries(scope) })
|
|
28733
|
+
body: rpcResult(body.id, { tools: toolEntries(scope, launch) })
|
|
28133
28734
|
};
|
|
28134
28735
|
case "tools/call":
|
|
28135
28736
|
if (isNotification) return {
|
|
@@ -28138,7 +28739,7 @@ async function handleRpc(_c, body, scope, sessionWorkspace) {
|
|
|
28138
28739
|
};
|
|
28139
28740
|
return {
|
|
28140
28741
|
status: 200,
|
|
28141
|
-
body: await handleToolsCall(body, scope, sessionWorkspace)
|
|
28742
|
+
body: await handleToolsCall(body, scope, launch, sessionWorkspace)
|
|
28142
28743
|
};
|
|
28143
28744
|
case "resources/list":
|
|
28144
28745
|
if (isNotification) return {
|
|
@@ -28218,6 +28819,7 @@ async function handleRpc(_c, body, scope, sessionWorkspace) {
|
|
|
28218
28819
|
async function handleMcpPost(c, scopeArg = "all") {
|
|
28219
28820
|
const auth = checkAuth(c);
|
|
28220
28821
|
if (!auth.ok) return c.json(rpcError(null, RPC_INVALID_REQUEST, auth.reason), auth.status);
|
|
28822
|
+
const { launch } = auth;
|
|
28221
28823
|
let scope;
|
|
28222
28824
|
if (scopeArg === "all") scope = "all";
|
|
28223
28825
|
else if (isMcpGroup(scopeArg)) scope = scopeArg;
|
|
@@ -28234,13 +28836,13 @@ async function handleMcpPost(c, scopeArg = "all") {
|
|
|
28234
28836
|
const nm = typeof body.params?.name === "string" ? body.params.name : "?";
|
|
28235
28837
|
process.stderr.write(`[peer-mcp] recv t=${Date.now()} name=${nm} scope=${scope} inflight=${currentInFlight()}\n`);
|
|
28236
28838
|
}
|
|
28237
|
-
if (typeof body === "object" && body !== null && !Array.isArray(body) && body.method === "tools/call" && acceptsEventStream(c.req.header("accept"))) return handleToolsCallSSE(body, scope, sessionWorkspace);
|
|
28839
|
+
if (typeof body === "object" && body !== null && !Array.isArray(body) && body.method === "tools/call" && acceptsEventStream(c.req.header("accept"))) return handleToolsCallSSE(body, scope, launch, sessionWorkspace);
|
|
28238
28840
|
if (typeof body === "object" && body !== null && !Array.isArray(body) && body.method === "tools/call") {
|
|
28239
28841
|
const preflight = jsonPathPreflightCap(body, scope);
|
|
28240
28842
|
if (preflight) return c.json(preflight, 200);
|
|
28241
28843
|
}
|
|
28242
28844
|
try {
|
|
28243
|
-
const { status, body: respBody } = await handleRpc(c, body, scope, sessionWorkspace);
|
|
28845
|
+
const { status, body: respBody } = await handleRpc(c, body, scope, launch, sessionWorkspace);
|
|
28244
28846
|
if (respBody === null) return c.body(null, status);
|
|
28245
28847
|
return c.json(respBody, status);
|
|
28246
28848
|
} catch (err) {
|
|
@@ -28293,9 +28895,9 @@ function acceptsEventStream(accept) {
|
|
|
28293
28895
|
* "Invalid state: Controller is already closed" race without warning.
|
|
28294
28896
|
*/
|
|
28295
28897
|
const SSE_HEARTBEAT_INTERVAL_MS = 5e3;
|
|
28296
|
-
async function handleToolsCallSSE(body, scope, sessionWorkspace) {
|
|
28898
|
+
async function handleToolsCallSSE(body, scope, launch, sessionWorkspace) {
|
|
28297
28899
|
const encoder = new TextEncoder();
|
|
28298
|
-
const callPromise = handleToolsCall(body, scope, sessionWorkspace);
|
|
28900
|
+
const callPromise = handleToolsCall(body, scope, launch, sessionWorkspace);
|
|
28299
28901
|
let heartbeatHandle;
|
|
28300
28902
|
const stream = new ReadableStream({
|
|
28301
28903
|
async start(controller) {
|
|
@@ -28972,13 +29574,18 @@ const ADVISOR_INTERNAL_TOOL_NAME = "__anthropic_advisor";
|
|
|
28972
29574
|
const ADVISOR_CLIENT_TOOL_NAME = "advisor";
|
|
28973
29575
|
/** Default advisor model + reasoning effort. Per gemini-critic + user
|
|
28974
29576
|
* direction: hardcode to a cross-lab model (gpt-5.6-sol — Copilot's
|
|
28975
|
-
* /responses-only flagship)
|
|
28976
|
-
*
|
|
28977
|
-
*
|
|
28978
|
-
*
|
|
28979
|
-
*
|
|
29577
|
+
* /responses-only flagship). The cross-lab choice gives a true "second set
|
|
29578
|
+
* of eyes" instead of the main model reviewing itself.
|
|
29579
|
+
*
|
|
29580
|
+
* Effort default is `high`, not the historical `xhigh`: `resolveAdvisorEffort`
|
|
29581
|
+
* no longer floors the picked effort (see that function), so `high` is both
|
|
29582
|
+
* the default AND the lowest the advisor will ever think at when the picker
|
|
29583
|
+
* expresses no preference — a deliberate, user-approved cost/depth trade
|
|
29584
|
+
* applied uniformly across every advisor target (Sol, the Opus escalation,
|
|
29585
|
+
* and the fast-profile Gemini advisor below all read this same constant). */
|
|
28980
29586
|
const ADVISOR_DEFAULT_MODEL = "gpt-5.6-sol";
|
|
28981
29587
|
const ADVISOR_DEFAULT_EFFORT = "xhigh";
|
|
29588
|
+
const ADVISOR_MIN_EFFORT = "high";
|
|
28982
29589
|
/** The Anthropic frontier model the advisor escalates to when the LEAD is a
|
|
28983
29590
|
* lighter Claude tier (sonnet, haiku).
|
|
28984
29591
|
*
|
|
@@ -28995,16 +29602,32 @@ const ADVISOR_DEFAULT_EFFORT = "xhigh";
|
|
|
28995
29602
|
* the decorrelation instrument and are untouched. `GH_ROUTER_ADVISOR_MODEL`
|
|
28996
29603
|
* keeps a cross-lab advisor one env var away for anyone who wants it back. */
|
|
28997
29604
|
const ADVISOR_ESCALATION_MODEL = "claude-opus-5";
|
|
28998
|
-
/**
|
|
28999
|
-
*
|
|
29000
|
-
*
|
|
29001
|
-
*
|
|
29002
|
-
*
|
|
29003
|
-
*
|
|
29004
|
-
*
|
|
29005
|
-
*
|
|
29006
|
-
|
|
29007
|
-
|
|
29605
|
+
/** The Advisor model for the fast Luna profile. Gemini 3.7 Flash is a
|
|
29606
|
+
* different lab from BOTH the Luna lead (OpenAI) and the fast profile's
|
|
29607
|
+
* `gemini-critic` persona shares this same model — see
|
|
29608
|
+
* `docs/default-models.md` "Fast launch profile" for the roster this
|
|
29609
|
+
* belongs to. Kept distinct from `ADVISOR_DEFAULT_MODEL` so the two never
|
|
29610
|
+
* have to agree; `resolveAdvisorModel` picks between them purely on lead
|
|
29611
|
+
* identity, never model availability heuristics beyond a live-catalog
|
|
29612
|
+
* presence check (mirrors `shouldEscalateAdvisor`'s pattern). */
|
|
29613
|
+
const ADVISOR_FAST_PROFILE_MODEL = "gemini-3.7-flash";
|
|
29614
|
+
/**
|
|
29615
|
+
* True when `leadModel` names the fast-profile Luna lead (bare, or with the
|
|
29616
|
+
* `[1m]` context decoration `withOneMSuffixForLead` applies to it).
|
|
29617
|
+
*/
|
|
29618
|
+
function isFastProfileLead(leadModel) {
|
|
29619
|
+
if (!leadModel) return false;
|
|
29620
|
+
const bare = leadModel.replace(/\[1m\]$/, "").trim();
|
|
29621
|
+
const lastSegment = bare.slice(bare.lastIndexOf("/") + 1);
|
|
29622
|
+
return bare === "gpt-5.6-luna" || lastSegment === "gpt-5.6-luna";
|
|
29623
|
+
}
|
|
29624
|
+
/** True when the live catalog actually carries `ADVISOR_FAST_PROFILE_MODEL`.
|
|
29625
|
+
* Mirrors `shouldEscalateAdvisor`'s catalog probe: never advertise a model
|
|
29626
|
+
* the account cannot reach, and fall back to the cross-lab default instead
|
|
29627
|
+
* of a hard failure when it's absent. */
|
|
29628
|
+
function fastProfileAdvisorAvailable() {
|
|
29629
|
+
return state.models?.data?.some((m) => m.id === "gemini-3.7-flash") ?? false;
|
|
29630
|
+
}
|
|
29008
29631
|
/** Output cap for the Anthropic-branch advisor call when the catalog carries no
|
|
29009
29632
|
* limits for the resolved model. The value the branch used unconditionally
|
|
29010
29633
|
* before it became reachable, kept so a catalog-less path is no worse off. */
|
|
@@ -29036,6 +29659,28 @@ function advisorUsesResponses(resolvedAdvisorModel) {
|
|
|
29036
29659
|
if (endpoints && endpoints.length > 0) return endpoints.some((e) => ADVISOR_RESPONSES_ENDPOINTS.has(e));
|
|
29037
29660
|
return /^(gpt-|o\d|.*codex)/i.test(bare);
|
|
29038
29661
|
}
|
|
29662
|
+
/**
|
|
29663
|
+
* Decide `advisorTransport` for a resolved advisor model id.
|
|
29664
|
+
*
|
|
29665
|
+
* Order matters: Claude identity is checked FIRST and wins even though
|
|
29666
|
+
* `claude-opus-5` also advertises `/chat/completions` in the live catalog —
|
|
29667
|
+
* the historical branch never sent Claude to chat, and this preserves that
|
|
29668
|
+
* byte-for-byte (reuses the SAME classifier `classifyMessagesRoute` uses for
|
|
29669
|
+
* the main `/v1/messages` shim fork, so the two surfaces cannot disagree
|
|
29670
|
+
* about what counts as a Claude model). Responses is checked next
|
|
29671
|
+
* (`advisorUsesResponses`, unchanged — catalog-first, name-regex fallback,
|
|
29672
|
+
* still exported and directly tested on its own). Anything else defaults to
|
|
29673
|
+
* chat, mirroring `pickEndpoint`'s "omits supported_endpoints => chat-eligible"
|
|
29674
|
+
* convention — the same convention `classifyMessagesRoute` relies on for a
|
|
29675
|
+
* lead model, applied here to the advisor's OWN model instead.
|
|
29676
|
+
*/
|
|
29677
|
+
function advisorTransport(resolvedAdvisorModel) {
|
|
29678
|
+
const bare = resolvedAdvisorModel.slice(resolvedAdvisorModel.lastIndexOf("/") + 1);
|
|
29679
|
+
const entry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel || m.id === bare);
|
|
29680
|
+
if (isClaudeModel(resolvedAdvisorModel, entry)) return "messages";
|
|
29681
|
+
if (advisorUsesResponses(resolvedAdvisorModel)) return "responses";
|
|
29682
|
+
return "chat";
|
|
29683
|
+
}
|
|
29039
29684
|
/** True when the model advertises a usable reasoning-effort ladder. */
|
|
29040
29685
|
function advertisedEffortLadder(resolvedAdvisorModel) {
|
|
29041
29686
|
const supported = state.models?.data?.find((m) => m.id === resolvedAdvisorModel)?.capabilities?.supports?.reasoning_effort;
|
|
@@ -29072,10 +29717,16 @@ function shouldEscalateAdvisor(leadModel) {
|
|
|
29072
29717
|
* Precedence:
|
|
29073
29718
|
* 1. `GH_ROUTER_ADVISOR_MODEL` (trimmed) — the operator pin, checked first so
|
|
29074
29719
|
* it works on every lead.
|
|
29075
|
-
* 2.
|
|
29076
|
-
*
|
|
29720
|
+
* 2. An authenticated fast launch whose current lead is Luna, with
|
|
29721
|
+
* `ADVISOR_FAST_PROFILE_MODEL` present in the live catalog.
|
|
29722
|
+
* 3. A lighter Claude lead with the escalation model in the catalog.
|
|
29723
|
+
* 4. `ADVISOR_DEFAULT_MODEL`.
|
|
29724
|
+
*
|
|
29725
|
+
* Steps 2 and 3 are mutually exclusive lead families (non-Claude Luna vs. a
|
|
29726
|
+
* lighter Claude tier) so their relative order does not matter functionally;
|
|
29727
|
+
* fast-profile is checked first only because it is the more specific match.
|
|
29077
29728
|
*
|
|
29078
|
-
* Step
|
|
29729
|
+
* Step 4 returns the LITERAL constant rather than walking the OpenAI frontier
|
|
29079
29730
|
* chain. An Opus lead must resolve to exactly what it resolves to today, and a
|
|
29080
29731
|
* frontier walk could yield `gpt-5.5` on a catalog missing `gpt-5.6-sol` —
|
|
29081
29732
|
* a silent change to the one path that is required not to move.
|
|
@@ -29104,19 +29755,27 @@ function normalizeAdvisorPin(pinned) {
|
|
|
29104
29755
|
const bare = pinned.slice(pinned.lastIndexOf("/") + 1);
|
|
29105
29756
|
return bare !== pinned && models.some((m) => m.id === bare) ? bare : pinned;
|
|
29106
29757
|
}
|
|
29107
|
-
function resolveAdvisorModel(leadModel) {
|
|
29758
|
+
function resolveAdvisorModel(leadModel, fastProfile = false) {
|
|
29108
29759
|
const pinned = process.env.GH_ROUTER_ADVISOR_MODEL?.trim();
|
|
29109
29760
|
if (pinned) return {
|
|
29110
29761
|
model: normalizeAdvisorPin(pinned),
|
|
29111
|
-
escalated: false
|
|
29762
|
+
escalated: false,
|
|
29763
|
+
fastProfile: false
|
|
29764
|
+
};
|
|
29765
|
+
if (fastProfile && leadModel && isFastProfileLead(leadModel) && fastProfileAdvisorAvailable()) return {
|
|
29766
|
+
model: ADVISOR_FAST_PROFILE_MODEL,
|
|
29767
|
+
escalated: false,
|
|
29768
|
+
fastProfile: true
|
|
29112
29769
|
};
|
|
29113
29770
|
if (leadModel && shouldEscalateAdvisor(leadModel)) return {
|
|
29114
29771
|
model: ADVISOR_ESCALATION_MODEL,
|
|
29115
|
-
escalated: true
|
|
29772
|
+
escalated: true,
|
|
29773
|
+
fastProfile: false
|
|
29116
29774
|
};
|
|
29117
29775
|
return {
|
|
29118
29776
|
model: ADVISOR_DEFAULT_MODEL,
|
|
29119
|
-
escalated: false
|
|
29777
|
+
escalated: false,
|
|
29778
|
+
fastProfile: false
|
|
29120
29779
|
};
|
|
29121
29780
|
}
|
|
29122
29781
|
/**
|
|
@@ -29138,13 +29797,16 @@ function resolveAdvisorModel(leadModel) {
|
|
|
29138
29797
|
* 3. `ADVISOR_DEFAULT_EFFORT` — a request expressing no preference behaves
|
|
29139
29798
|
* exactly as it did before the picker was honored at all.
|
|
29140
29799
|
*
|
|
29141
|
-
*
|
|
29142
|
-
*
|
|
29143
|
-
*
|
|
29144
|
-
*
|
|
29800
|
+
* There is deliberately NO floor anymore (removed per the user-approved
|
|
29801
|
+
* "default high, no floor" change): the advisor follows the picker all the way
|
|
29802
|
+
* down as well as up, so an explicit `none`/`low` pick is honored rather than
|
|
29803
|
+
* clamped up to a minimum. The only remaining adjustment is the CEILING clamp
|
|
29804
|
+
* against the resolved advisor's own live `reasoning_effort` allowlist — a
|
|
29805
|
+
* model whose ladder tops out below the requested tier still needs to receive
|
|
29806
|
+
* something it accepts.
|
|
29145
29807
|
*/
|
|
29146
|
-
function resolveAdvisorEffort(rawRequestBody, advisorModel) {
|
|
29147
|
-
let requested = ADVISOR_DEFAULT_EFFORT;
|
|
29808
|
+
function resolveAdvisorEffort(rawRequestBody, advisorModel, fastProfile = false) {
|
|
29809
|
+
let requested = fastProfile ? "high" : ADVISOR_DEFAULT_EFFORT;
|
|
29148
29810
|
if (rawRequestBody) try {
|
|
29149
29811
|
const body = JSON.parse(rawRequestBody);
|
|
29150
29812
|
const oc = body.output_config;
|
|
@@ -29153,10 +29815,11 @@ function resolveAdvisorEffort(rawRequestBody, advisorModel) {
|
|
|
29153
29815
|
if (typeof explicit === "string" && EFFORT_ORDER.includes(explicit)) requested = explicit;
|
|
29154
29816
|
else if (thinking && typeof thinking === "object" && thinking.type === "enabled") requested = bucketEffort(thinking.budget_tokens);
|
|
29155
29817
|
} catch {}
|
|
29156
|
-
|
|
29818
|
+
if (fastProfile) requested = "high";
|
|
29819
|
+
else if (EFFORT_ORDER.indexOf(requested) < EFFORT_ORDER.indexOf(ADVISOR_MIN_EFFORT)) requested = ADVISOR_MIN_EFFORT;
|
|
29157
29820
|
const supported = state.models?.data?.find((m) => m.id === resolveModel(advisorModel))?.capabilities?.supports?.reasoning_effort;
|
|
29158
|
-
if (!Array.isArray(supported) || supported.length === 0) return
|
|
29159
|
-
return clampEffort(
|
|
29821
|
+
if (!Array.isArray(supported) || supported.length === 0) return requested;
|
|
29822
|
+
return clampEffort(requested, supported);
|
|
29160
29823
|
}
|
|
29161
29824
|
/** ADVISOR_TOOL_INSTRUCTIONS verbatim from cc-backup
|
|
29162
29825
|
* src/utils/advisor.ts — describes when the model should invoke
|
|
@@ -29386,8 +30049,9 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29386
30049
|
maxUnits = ADVISOR_MAX_CONVERSATION_CHARS;
|
|
29387
30050
|
}
|
|
29388
30051
|
const conversationText = renderConversationAsText(conversation, maxUnits, measure);
|
|
29389
|
-
|
|
29390
|
-
|
|
30052
|
+
const transport = advisorTransport(resolvedAdvisorModel);
|
|
30053
|
+
if (transport === "responses") {
|
|
30054
|
+
const payload = applyResponsesCachePolicy({
|
|
29391
30055
|
model: resolvedAdvisorModel,
|
|
29392
30056
|
instructions: advisorSystem,
|
|
29393
30057
|
input: [{
|
|
@@ -29399,7 +30063,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29399
30063
|
}],
|
|
29400
30064
|
stream: false,
|
|
29401
30065
|
reasoning: { effort: advisorEffort }
|
|
29402
|
-
};
|
|
30066
|
+
}, { workload: "reusable-prefix" });
|
|
29403
30067
|
const response = await withTransientRetry(() => createResponses(payload, void 0, signal), {
|
|
29404
30068
|
signal,
|
|
29405
30069
|
label: resolvedAdvisorModel
|
|
@@ -29421,10 +30085,31 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29421
30085
|
if (!text) throw new Error(`Advisor model ${resolvedAdvisorModel} returned empty assistant output`);
|
|
29422
30086
|
return text;
|
|
29423
30087
|
}
|
|
30088
|
+
if (transport === "chat") {
|
|
30089
|
+
const ladder = advertisedEffortLadder(resolvedAdvisorModel);
|
|
30090
|
+
const chatPayload = {
|
|
30091
|
+
model: resolvedAdvisorModel,
|
|
30092
|
+
messages: [{
|
|
30093
|
+
role: "system",
|
|
30094
|
+
content: advisorSystem
|
|
30095
|
+
}, {
|
|
30096
|
+
role: "user",
|
|
30097
|
+
content: conversationText
|
|
30098
|
+
}],
|
|
30099
|
+
stream: false,
|
|
30100
|
+
...ladder ? { reasoning_effort: advisorEffort } : {}
|
|
30101
|
+
};
|
|
30102
|
+
const text = (await withTransientRetry(() => createChatCompletions(chatPayload, void 0, signal), {
|
|
30103
|
+
signal,
|
|
30104
|
+
label: resolvedAdvisorModel
|
|
30105
|
+
})).choices?.[0]?.message?.content;
|
|
30106
|
+
if (typeof text !== "string" || text.length === 0) throw new Error(`Advisor model ${resolvedAdvisorModel} returned empty response`);
|
|
30107
|
+
return text;
|
|
30108
|
+
}
|
|
29424
30109
|
const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
|
|
29425
30110
|
const limits = advisorEntry?.capabilities?.limits;
|
|
29426
30111
|
const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
|
|
29427
|
-
const advisorBody = JSON.stringify({
|
|
30112
|
+
const advisorBody = applyClaudeCachePolicy(JSON.stringify({
|
|
29428
30113
|
model: resolvedAdvisorModel,
|
|
29429
30114
|
max_tokens: maxTokens,
|
|
29430
30115
|
system: advisorSystem,
|
|
@@ -29437,7 +30122,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29437
30122
|
thinking: { type: "adaptive" },
|
|
29438
30123
|
...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
|
|
29439
30124
|
} : {}
|
|
29440
|
-
});
|
|
30125
|
+
}), { workload: "reusable-prefix" });
|
|
29441
30126
|
const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
|
|
29442
30127
|
signal,
|
|
29443
30128
|
label: resolvedAdvisorModel
|
|
@@ -29449,20 +30134,51 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29449
30134
|
/**
|
|
29450
30135
|
* Derive a spec-compliant `srvtoolu_*` id for a client-facing
|
|
29451
30136
|
* `server_tool_use` (and matching `advisor_tool_result.tool_use_id`)
|
|
29452
|
-
* from the upstream model's
|
|
29453
|
-
*
|
|
29454
|
-
*
|
|
29455
|
-
*
|
|
29456
|
-
*
|
|
29457
|
-
*
|
|
29458
|
-
*
|
|
29459
|
-
*
|
|
29460
|
-
|
|
29461
|
-
|
|
29462
|
-
|
|
29463
|
-
|
|
29464
|
-
|
|
29465
|
-
|
|
30137
|
+
* from the upstream model's tool-call id.
|
|
30138
|
+
*
|
|
30139
|
+
* TOTAL — never throws. Two paths:
|
|
30140
|
+
*
|
|
30141
|
+
* 1. A real Anthropic `toolu_*` id whose suffix is already in the
|
|
30142
|
+
* `^[a-zA-Z0-9_]+$` charset: `srvtoolu_<suffix>`, byte-for-byte
|
|
30143
|
+
* identical to the historical (Claude-lead) behavior.
|
|
30144
|
+
* 2. Anything else — a Responses `call_*` id (the fast Luna profile's
|
|
30145
|
+
* lead, once its `tool_use{__anthropic_advisor}` block is synthesized
|
|
30146
|
+
* by the anthropic-translate shim from a Copilot `/responses` tool
|
|
30147
|
+
* call), a hyphenated or otherwise non-conforming id, an empty string,
|
|
30148
|
+
* unicode, or a corrupt id — sanitize to the Anthropic charset and
|
|
30149
|
+
* prefix with `fallbackIndex` (the caller's per-block synthetic stream
|
|
30150
|
+
* index, unique within one `buildAdvisorStream` run) so two different
|
|
30151
|
+
* raw ids that happen to sanitize to the same string can never
|
|
30152
|
+
* collide. `fallbackIndex` is REQUIRED for this path's uniqueness
|
|
30153
|
+
* guarantee — callers must pass a value that is unique per call within
|
|
30154
|
+
* one advisor stream (every call site does: `myIndex` from the
|
|
30155
|
+
* turn processor's monotonic `nextSyntheticIndex`).
|
|
30156
|
+
*
|
|
30157
|
+
* This function ONLY has to produce a valid, deterministic, collision-free
|
|
30158
|
+
* LABEL — the original raw id is preserved separately for Copilot replay
|
|
30159
|
+
* (`CapturedBlock.advisorReplay.id`), never reconstructed from the derived
|
|
30160
|
+
* client id. That is what makes totality safe: there is no bijective-decode
|
|
30161
|
+
* requirement on this function itself, only on the (id, clientId) pairing a
|
|
30162
|
+
* caller keeps alongside it.
|
|
30163
|
+
*
|
|
30164
|
+
* Historically this threw "advisor tool_use id is not round-trippable" for
|
|
30165
|
+
* any non-`toolu_` shape. That was correct for a Claude-only advisor lead —
|
|
30166
|
+
* Copilot's native `/v1/messages` never emits anything else — but became a
|
|
30167
|
+
* live defect once the advisor loop could run on a non-Claude (Luna) lead
|
|
30168
|
+
* shimmed through `/responses`: `responses-egress.ts` forwards a Responses
|
|
30169
|
+
* `call_*` id VERBATIM as the synthesized `tool_use.id` (see
|
|
30170
|
+
* `makeToolUseId` — it only synthesizes a `toolu_*` id when the upstream id
|
|
30171
|
+
* is EMPTY), so the advisor's `tool_use{__anthropic_advisor}` block on that
|
|
30172
|
+
* lead legitimately carries a `call_*` id and the throw fired on every
|
|
30173
|
+
* single advisor call.
|
|
30174
|
+
*/
|
|
30175
|
+
function toClientServerToolUseId(id, fallbackIndex) {
|
|
30176
|
+
if (id.startsWith("toolu_")) {
|
|
30177
|
+
const suffix = id.slice(6);
|
|
30178
|
+
if (/^[a-zA-Z0-9_]+$/.test(suffix)) return `srvtoolu_${suffix}`;
|
|
30179
|
+
}
|
|
30180
|
+
const sanitized = id.replace(/[^a-zA-Z0-9_]/g, "_");
|
|
30181
|
+
return `srvtoolu_gen${fallbackIndex}${sanitized.length > 0 ? `_${sanitized}` : ""}`;
|
|
29466
30182
|
}
|
|
29467
30183
|
/**
|
|
29468
30184
|
* Build an SSE event line in the canonical Anthropic shape:
|
|
@@ -29474,6 +30190,35 @@ function sseEvent(type, data) {
|
|
|
29474
30190
|
return `event: ${type}\ndata: ${JSON.stringify(data)}\n\n`;
|
|
29475
30191
|
}
|
|
29476
30192
|
/**
|
|
30193
|
+
* The default `continueTurn` for `buildAdvisorStream`: native Claude
|
|
30194
|
+
* passthrough (`createMessages`) plus signed-thinking-history repair-and-retry.
|
|
30195
|
+
* Extracted verbatim from the loop body so the behavior is byte-identical to
|
|
30196
|
+
* before `continueTurn` became injectable, and so a non-Claude
|
|
30197
|
+
* `continueTurn` (the fast Luna profile's shim-backed one) can omit this
|
|
30198
|
+
* Claude-only repair path entirely rather than inherit dead code that would
|
|
30199
|
+
* never fire for it.
|
|
30200
|
+
*/
|
|
30201
|
+
async function defaultContinueTurn(body, signal, requestHeaders) {
|
|
30202
|
+
let continuationSend = JSON.stringify(body);
|
|
30203
|
+
const knownRepair = repairKnownThinkingHistory(continuationSend);
|
|
30204
|
+
if (knownRepair) continuationSend = knownRepair.body;
|
|
30205
|
+
try {
|
|
30206
|
+
return await createMessages(continuationSend, requestHeaders, signal, true);
|
|
30207
|
+
} catch (continuationError) {
|
|
30208
|
+
if (!(continuationError instanceof HTTPError)) throw continuationError;
|
|
30209
|
+
const errorBody = await continuationError.response.clone().text().catch(() => "");
|
|
30210
|
+
const outcome = repairRejectedThinkingHistory(continuationSend, errorBody);
|
|
30211
|
+
if (!outcome.ok) {
|
|
30212
|
+
consola.warn(`Advisor continuation thinking-history repair declined: ${formatThinkingRepairDecline(outcome.decline)}`);
|
|
30213
|
+
throw continuationError;
|
|
30214
|
+
}
|
|
30215
|
+
consola.warn(`Advisor continuation: retrying without rejected thinking blocks: message=${outcome.repair.messageIndex} removed_blocks=${outcome.repair.removedBlocks}`);
|
|
30216
|
+
const response = await createMessages(outcome.repair.body, requestHeaders, signal, true);
|
|
30217
|
+
rememberThinkingHistoryRepair(outcome.repair.fingerprint);
|
|
30218
|
+
return response;
|
|
30219
|
+
}
|
|
30220
|
+
}
|
|
30221
|
+
/**
|
|
29477
30222
|
* The streaming translate-loop. Returns a ReadableStream<Uint8Array>
|
|
29478
30223
|
* suitable to wrap with Hono's c.body() / new Response().
|
|
29479
30224
|
*
|
|
@@ -29492,6 +30237,7 @@ function buildAdvisorStream(opts) {
|
|
|
29492
30237
|
const advisorModel = opts.advisorModel ?? "gpt-5.6-sol";
|
|
29493
30238
|
const advisorEffort = opts.advisorEffort ?? "xhigh";
|
|
29494
30239
|
const advisorEscalated = opts.advisorEscalated ?? false;
|
|
30240
|
+
const continueTurn = opts.continueTurn ?? ((body, signal) => defaultContinueTurn(body, signal, opts.requestHeaders));
|
|
29495
30241
|
const aborter = opts.externalAborter ?? new AbortController();
|
|
29496
30242
|
let conversation = [...opts.initialConversation];
|
|
29497
30243
|
return new ReadableStream({
|
|
@@ -29789,27 +30535,11 @@ function buildAdvisorStream(opts) {
|
|
|
29789
30535
|
}))
|
|
29790
30536
|
});
|
|
29791
30537
|
if (aborter.signal.aborted) return;
|
|
29792
|
-
|
|
30538
|
+
response = await continueTurn({
|
|
29793
30539
|
...opts.baseBody,
|
|
29794
30540
|
messages: conversation,
|
|
29795
30541
|
stream: true
|
|
29796
|
-
});
|
|
29797
|
-
const knownRepair = repairKnownThinkingHistory(continuationSend);
|
|
29798
|
-
if (knownRepair) continuationSend = knownRepair.body;
|
|
29799
|
-
try {
|
|
29800
|
-
response = await createMessages(continuationSend, opts.requestHeaders, aborter.signal, true);
|
|
29801
|
-
} catch (continuationError) {
|
|
29802
|
-
if (!(continuationError instanceof HTTPError)) throw continuationError;
|
|
29803
|
-
const errorBody = await continuationError.response.clone().text().catch(() => "");
|
|
29804
|
-
const outcome = repairRejectedThinkingHistory(continuationSend, errorBody);
|
|
29805
|
-
if (!outcome.ok) {
|
|
29806
|
-
consola.warn(`Advisor continuation thinking-history repair declined: ${formatThinkingRepairDecline(outcome.decline)}`);
|
|
29807
|
-
throw continuationError;
|
|
29808
|
-
}
|
|
29809
|
-
consola.warn(`Advisor continuation: retrying without rejected thinking blocks: message=${outcome.repair.messageIndex} removed_blocks=${outcome.repair.removedBlocks}`);
|
|
29810
|
-
response = await createMessages(outcome.repair.body, opts.requestHeaders, aborter.signal, true);
|
|
29811
|
-
rememberThinkingHistoryRepair(outcome.repair.fingerprint);
|
|
29812
|
-
}
|
|
30542
|
+
}, aborter.signal);
|
|
29813
30543
|
}
|
|
29814
30544
|
if (aborter.signal.aborted) return;
|
|
29815
30545
|
const finalIndex = nextSyntheticIndex++;
|
|
@@ -31625,6 +32355,17 @@ const ADVISOR_PARAMS = Type$1.Object({ concern: Type$1.String({
|
|
|
31625
32355
|
description: "What you want a second pair of eyes on — your current approach, the blocker you're stuck on, or the decision you're about to commit. Required: the advisor needs a focal point.",
|
|
31626
32356
|
minLength: 1
|
|
31627
32357
|
}) });
|
|
32358
|
+
/**
|
|
32359
|
+
* Fixed reasoning effort for the WORKER'S own `advisor` tool call (distinct
|
|
32360
|
+
* from the server-side ADVISOR mechanism's `ADVISOR_DEFAULT_EFFORT` in
|
|
32361
|
+
* `src/services/advisor/advisor.ts`, which now follows the Claude Code
|
|
32362
|
+
* effort picker and defaults to `high`). This tool has no picker to follow —
|
|
32363
|
+
* it is a single fixed-effort consultation a worker triggers explicitly, not
|
|
32364
|
+
* a per-request value derived from client input — so it keeps its own
|
|
32365
|
+
* historical constant rather than sharing one that started varying for an
|
|
32366
|
+
* unrelated reason.
|
|
32367
|
+
*/
|
|
32368
|
+
const WORKER_ADVISOR_EFFORT = "xhigh";
|
|
31628
32369
|
/** Advisor transcript budget — leaves headroom in the advisor's
|
|
31629
32370
|
* context window after the system prompt + concern + reasoning
|
|
31630
32371
|
* overhead. Truncate-from-front so the most recent turn (where the
|
|
@@ -31715,7 +32456,7 @@ function advisorTool(getMessages) {
|
|
|
31715
32456
|
const release = acquireInFlightSlot();
|
|
31716
32457
|
if (!release) throw new Error(`advisor: MCP in-flight cap (${MAX_INFLIGHT_TOOLS_CALL}) saturated; retry shortly`);
|
|
31717
32458
|
try {
|
|
31718
|
-
const
|
|
32459
|
+
const payload = applyResponsesCachePolicy({
|
|
31719
32460
|
model: resolvedModel,
|
|
31720
32461
|
instructions: advisorSystem,
|
|
31721
32462
|
input: [{
|
|
@@ -31726,8 +32467,9 @@ function advisorTool(getMessages) {
|
|
|
31726
32467
|
}]
|
|
31727
32468
|
}],
|
|
31728
32469
|
stream: false,
|
|
31729
|
-
reasoning: { effort:
|
|
31730
|
-
},
|
|
32470
|
+
reasoning: { effort: WORKER_ADVISOR_EFFORT }
|
|
32471
|
+
}, { workload: "reusable-prefix" });
|
|
32472
|
+
const text = extractResponsesText(await createResponses(payload, void 0, signal, true));
|
|
31731
32473
|
if (!text) throw new Error("advisor returned empty output");
|
|
31732
32474
|
return textResult(text);
|
|
31733
32475
|
} finally {
|
|
@@ -34808,6 +35550,17 @@ function buildAgentPrompt(persona, opts) {
|
|
|
34808
35550
|
*/
|
|
34809
35551
|
function buildPeerAwarenessSnippet(opts) {
|
|
34810
35552
|
const key = (g) => opts.groupKeys?.[g] ?? GROUP_META[g].preferredKey;
|
|
35553
|
+
if (opts.profile === "fast") {
|
|
35554
|
+
const fastPeersKey = key("peers");
|
|
35555
|
+
const fastSearchKey = key("search");
|
|
35556
|
+
return [
|
|
35557
|
+
"## Peer review and advisor",
|
|
35558
|
+
"",
|
|
35559
|
+
`This is the fast launch profile. \`mcp__${fastPeersKey}__oracle\` is exact Opus 5 (1M/high), a stateless last-resort consultant after the primary Luna path, Advisor, and reviewer/planner remain stuck. Advisor is the transcript-aware brainstorming, sounding-board, fresh-look, uncertainty, and stuck path.`,
|
|
35560
|
+
"",
|
|
35561
|
+
`\`mcp__${fastSearchKey}__code\` is semantic-first code search and \`mcp__${fastSearchKey}__web\` surfaces citable sources. Native Task roster: \`scout\` (broad discovery), \`implementer\` (mechanical implementation), \`reviewer\` (repo-aware verification/reproduction), and \`planner\` (Sol plan consultant/approver after Luna's draft). Before implementation obtain planner approval; before declaring done run relevant tests and ask reviewer to verify.${opts.browseAvailable ? ` \`mcp__${key("browser")}__*\` is the opt-in browser surface.` : ""}`
|
|
35562
|
+
].join("\n");
|
|
35563
|
+
}
|
|
34811
35564
|
const peersKey = key("peers");
|
|
34812
35565
|
const searchKey = key("search");
|
|
34813
35566
|
const workersKey = key("workers");
|
|
@@ -34861,6 +35614,12 @@ function buildPeerAwarenessSnippet(opts) {
|
|
|
34861
35614
|
*/
|
|
34862
35615
|
function buildPeerAwarenessSummary(opts) {
|
|
34863
35616
|
const key = (g) => opts.groupKeys?.[g] ?? GROUP_META[g].preferredKey;
|
|
35617
|
+
if (opts.profile === "fast") return [
|
|
35618
|
+
"## Injected capabilities (summary)",
|
|
35619
|
+
"",
|
|
35620
|
+
"Fast launch profile. Task roster: `scout`, `implementer`, `reviewer`, `planner`. Luna investigates and drafts; `planner` must approve before implementation. Before declaring done, run relevant tests and ask `reviewer` to verify.",
|
|
35621
|
+
`Advisor is the transcript-aware brainstorming/sounding-board/fresh-look path. \`mcp__${key("peers")}__oracle\` is exact Opus 5 (1M/high), stateless and last resort. \`mcp__${key("search")}__code\` and \`mcp__${key("search")}__web\` provide search.${opts.browseAvailable ? ` \`mcp__${key("browser")}__*\` provides the opt-in browser.` : ""}`
|
|
35622
|
+
].join("\n");
|
|
34864
35623
|
const renderNative = (name) => {
|
|
34865
35624
|
const modelId = opts.nativeAgentModels?.[name];
|
|
34866
35625
|
if (!modelId) return `\`${name}\``;
|
|
@@ -34895,6 +35654,22 @@ function buildPeerAwarenessSummary(opts) {
|
|
|
34895
35654
|
return lines.join("\n");
|
|
34896
35655
|
}
|
|
34897
35656
|
/**
|
|
35657
|
+
* Translate a `toolNameHttp`-keyed persona allowlist (the currency
|
|
35658
|
+
* `LaunchProfileDescriptor.personaAllowlist` uses, since that is what the MCP
|
|
35659
|
+
* boundary's `tools/call` narrowing filters on) into the `agentName`-keyed
|
|
35660
|
+
* allowlist `personasFor`'s `agentAllowlist` consumes (since that is the key
|
|
35661
|
+
* `buildPeerAgentDefinitions` uses to build subagent `.md` files). The two
|
|
35662
|
+
* identifiers differ (`gemini_critic` vs `gemini-critic`), so a caller wiring
|
|
35663
|
+
* a launch profile's persona restriction into subagent generation needs this
|
|
35664
|
+
* translation rather than assuming the sets are interchangeable.
|
|
35665
|
+
*/
|
|
35666
|
+
function agentNamesForToolAllowlist(toolAllowlist) {
|
|
35667
|
+
const allow = toolAllowlist instanceof Set ? toolAllowlist : new Set(toolAllowlist);
|
|
35668
|
+
const names = /* @__PURE__ */ new Set();
|
|
35669
|
+
for (const p of [...PERSONAS_READ, ...PERSONAS_WRITE]) if (allow.has(p.toolNameHttp)) names.add(p.agentName);
|
|
35670
|
+
return names;
|
|
35671
|
+
}
|
|
35672
|
+
/**
|
|
34898
35673
|
* Applies the resolved Gemini review model to a persona requiring the Gemini
|
|
34899
35674
|
* catalog: swaps `.model` and rewrites every literal occurrence of the
|
|
34900
35675
|
* default id in `.description` so the two never disagree about which model
|
|
@@ -34919,8 +35694,10 @@ function resolveGeminiPersona(p, geminiModel) {
|
|
|
34919
35694
|
}
|
|
34920
35695
|
/** Convenience: every persona that should be registered for the given mode. */
|
|
34921
35696
|
function personasFor(opts) {
|
|
35697
|
+
const allow = opts.agentAllowlist == null ? void 0 : opts.agentAllowlist instanceof Set ? opts.agentAllowlist : new Set(opts.agentAllowlist);
|
|
34922
35698
|
const result = [];
|
|
34923
35699
|
for (const p of PERSONAS_READ) {
|
|
35700
|
+
if (allow && !allow.has(p.agentName)) continue;
|
|
34924
35701
|
if (p.requiresGeminiCatalog) {
|
|
34925
35702
|
if (!opts.geminiAvailable) continue;
|
|
34926
35703
|
result.push(resolveGeminiPersona(p, opts.geminiModel));
|
|
@@ -34928,7 +35705,10 @@ function personasFor(opts) {
|
|
|
34928
35705
|
}
|
|
34929
35706
|
result.push(p);
|
|
34930
35707
|
}
|
|
34931
|
-
if (opts.codexCli) for (const p of PERSONAS_WRITE)
|
|
35708
|
+
if (opts.codexCli) for (const p of PERSONAS_WRITE) {
|
|
35709
|
+
if (allow && !allow.has(p.agentName)) continue;
|
|
35710
|
+
result.push(p);
|
|
35711
|
+
}
|
|
34932
35712
|
return result;
|
|
34933
35713
|
}
|
|
34934
35714
|
const WEB_SEARCH_DESCRIPTION = "Web search via GitHub Copilot's MCP that returns answer text plus source URLs the caller can cite. It accepts a natural-language `query`; the upstream provider rewrites for the search index and the handler formats any references as markdown links. Use for current external information such as API documentation, error-message diagnosis, upstream issue searches, and claims that need web sources. Not for local repository discovery or code navigation, use code, Read, Grep, or Glob for workspace content. Prefer it over the built-in WebSearch when source URLs are needed or the built-in surface is geographically constrained.";
|
|
@@ -36111,6 +36891,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
|
|
|
36111
36891
|
return [...new Set(names)];
|
|
36112
36892
|
}
|
|
36113
36893
|
//#endregion
|
|
36114
|
-
export {
|
|
36894
|
+
export { bucketEffort as $, CONDENSED_OPERATING_SEQUENCE as $t, assetFor as A, countTokens as At, resolveAdvisorModel as B, createChatCompletions as Bt, buildEnv as C, withOneMSuffixForLead as Cn, reviewerFastModel as Ct, toolbeltSkipSet as D, standInToolEnabled as Dt, toolbeltEnabled as E, scribeModel as Et, buildAdvisorStream as F, unregisterLaunch as Ft, buildAnthropicErrorEvent as G, provisionBrowserAssets as Gt, rememberThinkingHistoryRepair as H, readResponseBodyCapped as Ht, injectAdvisorTool as I, assembleResponsesPayload as It, logStreamError as J, provisionAndIndexColbert as Jt, buildOpenAIErrorEvent as K, hasSupportedBrowserInstalled as Kt, isAdvisorRequested as L, warnOnTokenPriceDrift as Lt, searchWeb as M, getTokenCount as Mt, ADVISOR_INTERNAL_TOOL_NAME as N, findLaunchBySecret as Nt, vscodeRipgrepPath as O, workerToolsEnabled as Ot, ADVISOR_TOOL_INSTRUCTIONS as P, registerLaunch as Pt, UNKNOWN_EFFORT_ANCHOR as Q, provisionTreeSitterAssets as Qt, isFastProfileLead as R, resolveMcpToolTimeoutMs as Rt, runWorkerAgent as S, withOneMSuffix as Sn, resolveGeminiReviewModel as St, buildToolbeltAwareness as T, scoutModel as Tt, repairKnownThinkingHistory as U, parseJsonOrDiagnose as Ut, formatThinkingRepairDecline as V, MAX_RESPONSE_BODY_BYTES as Vt, repairRejectedThinkingHistory as W, normalizeOpenAIUsage as Wt, relayAnthropicStream as X, extractZipMember as Xt, readIteratorWithTimeout as Y, extractTarGzMember as Yt, EFFORT_ORDER as Z, warmTreeSitterPool as Zt, TEST_DEFAULT_MODEL as _, upstreamMaxConnections as _n, fleetToolsEnabled as _t, buildAgentPrompt as a, BUDGET_SMALL_FAST_SLUG as an, FAST_SCOUT_EFFORT as at, resolveModeDefaults as b, catalogAdvertises1M as bn, implementerFastModel as bt, enumerateInjectedMcpToolNames as c, DEFAULT_CODEX_MODEL_FALLBACKS as cn, brainstormModel as ct, DEFAULT_MODEL_CHAIN as d, UPSTREAM_INACTIVITY_TIMEOUT_MS as dn, browserToolsEnabled as dt, DEFINITION_OF_GREATNESS as en, clampEffort as et, EXPLORE_DEFAULT_MODEL as f, generateRandomPort as fn, fastImplementerModel as ft, REVIEW_DEFAULT_MODEL as g, upstreamAllowH2 as gn, fastScoutModel as gt, PLAN_DEFAULT_MODEL as h, resolveLeadSlugArg as hn, fastReviewerModel as ht, assertMcpToolSurfaceConsistent as i, BUDGET_SMALL_FAST_CATALOG_ID as in, FAST_REVIEWER_EFFORT as it, satisfiesMinVersion as j, createMessages as jt, TOOLBELT_TOOLS$1 as k, shimDefaultsToXhigh as kt, personasFor as l, DEFAULT_PORT as ln, browseAgentEnabled as lt, IMPLEMENT_DEFAULT_MODEL as m, pickClaudeDefault as mn, fastPlannerModel as mt, MCP_GROUPS as n, collapsePathKeys as nn, handleMcpPost as nt, buildPeerAwarenessSnippet as o, DEFAULT_CLAUDE_MODEL_FALLBACKS as on, agentToolsEnabled as ot, EXPLORE_DEFAULT_THINKING as p, isBudgetClaudeLead as pn, fastOracleModel as pt, isControllerClosedError as q, colbertDegradedWarning as qt, agentNamesForToolAllowlist as r, toolbeltPathOverride as rn, FAST_PLANNER_EFFORT as rt, buildPeerAwarenessSummary as s, DEFAULT_CODEX_MODEL as sn, artifactToolsEnabled as st, GROUP_META as t, shouldUseInsecureTls as tn, handleMcpDelete as tt, BROWSE_DEFAULT_MODEL as u, UPSTREAM_FETCH_TIMEOUT_MS as un, browserCompoundToolsEnabled as ut, appendPlanReminder as v, classifyMessagesRoute as vn, geminiAvailable as vt, availableToolCommands as w, withInstallLock as wn, reviewerModel as wt, resolveWorkerRunOpts as x, oneMContextDisabled as xn, nativeSubagentModel as xt, resolveDefaultModel as y, pickEndpoint as yn, generalPurposeFastModel as yt, resolveAdvisorEffort as z, createResponses as zt };
|
|
36115
36895
|
|
|
36116
|
-
//# sourceMappingURL=peer-mcp-personas-
|
|
36896
|
+
//# sourceMappingURL=peer-mcp-personas-CHbl6MwM.js.map
|