github-router 0.3.285 → 0.3.289
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -5
- package/dist/{attribution-settings-B8M2fvhz.js → attribution-settings-Cmz2jt7P.js} +44 -17
- package/dist/attribution-settings-Cmz2jt7P.js.map +1 -0
- package/dist/{auth-VUL2Zxvw.js → auth-DG4vh8-F.js} +3 -3
- package/dist/{auth-VUL2Zxvw.js.map → auth-DG4vh8-F.js.map} +1 -1
- package/dist/browser-ext/manifest.json +1 -1
- package/dist/{check-usage-CcLdFPGr.js → check-usage-BqN7mBYv.js} +4 -4
- package/dist/{check-usage-CcLdFPGr.js.map → check-usage-BqN7mBYv.js.map} +1 -1
- package/dist/{claude-e-JIQGhR.js → claude-C-9xFI4b.js} +23 -17
- package/dist/claude-C-9xFI4b.js.map +1 -0
- package/dist/{codex-DSnVZ9oK.js → codex-DRW0yb8x.js} +5 -5
- package/dist/{codex-DSnVZ9oK.js.map → codex-DRW0yb8x.js.map} +1 -1
- package/dist/{debug-CtzYWxpJ.js → debug-B5TjPTTH.js} +2 -2
- package/dist/{debug-CtzYWxpJ.js.map → debug-B5TjPTTH.js.map} +1 -1
- package/dist/engine-iEqGdx6T.js +2 -0
- package/dist/{gate-discovery-BCFwLm0q.js → gate-discovery-Cz6kwIVG.js} +5 -5
- package/dist/{gate-discovery-BCFwLm0q.js.map → gate-discovery-Cz6kwIVG.js.map} +1 -1
- package/dist/{get-copilot-usage-B-EDAqQb.js → get-copilot-usage-BjA0nyGR.js} +2 -2
- package/dist/{get-copilot-usage-B-EDAqQb.js.map → get-copilot-usage-BjA0nyGR.js.map} +1 -1
- package/dist/{internal-artifact-open-Dj5Nlq0L.js → internal-artifact-open-BskmUpnb.js} +2 -2
- package/dist/{internal-artifact-open-Dj5Nlq0L.js.map → internal-artifact-open-BskmUpnb.js.map} +1 -1
- package/dist/{internal-first-mate-guard-DOgFVki5.js → internal-first-mate-guard-5XEHMaqy.js} +3 -3
- package/dist/{internal-first-mate-guard-DOgFVki5.js.map → internal-first-mate-guard-5XEHMaqy.js.map} +1 -1
- package/dist/{internal-first-mate-guard-B7p4NttK.js → internal-first-mate-guard-DHDQ6hFz.js} +1 -1
- package/dist/{internal-plan-review-DRFHET_B.js → internal-plan-review-BUuMx4ku.js} +3 -3
- package/dist/{internal-plan-review-DRFHET_B.js.map → internal-plan-review-BUuMx4ku.js.map} +1 -1
- package/dist/{internal-prompt-submit-f1LR2P2G.js → internal-prompt-submit-CQQ15xdO.js} +4 -4
- package/dist/{internal-prompt-submit-f1LR2P2G.js.map → internal-prompt-submit-CQQ15xdO.js.map} +1 -1
- package/dist/{internal-session-bind-DhZhxJ3T.js → internal-session-bind-D04W2yWI.js} +2 -2
- package/dist/{internal-session-bind-DhZhxJ3T.js.map → internal-session-bind-D04W2yWI.js.map} +1 -1
- package/dist/{internal-stop-hook-Cf-7w3RH.js → internal-stop-hook-Dvkppwo7.js} +5 -5
- package/dist/{internal-stop-hook-Cf-7w3RH.js.map → internal-stop-hook-Dvkppwo7.js.map} +1 -1
- package/dist/{internal-stop-review-BBsLcbPG.js → internal-stop-review-CdByyJLc.js} +2 -2
- package/dist/{internal-stop-review-BBsLcbPG.js.map → internal-stop-review-CdByyJLc.js.map} +1 -1
- package/dist/{internal-worker-guard-CKgYFiFO.js → internal-worker-guard-BIPN6Rv9.js} +2 -2
- package/dist/{internal-worker-guard-CKgYFiFO.js.map → internal-worker-guard-BIPN6Rv9.js.map} +1 -1
- package/dist/{internal-workspace-header-OYgHEnFt.js → internal-workspace-header-BKqejstG.js} +2 -2
- package/dist/{internal-workspace-header-OYgHEnFt.js.map → internal-workspace-header-BKqejstG.js.map} +1 -1
- package/dist/lifecycle-BTodQvn4.js +2 -0
- package/dist/lifecycle-C7JYNz-F.js +2 -0
- package/dist/{lifecycle-B7CHqKlF.js → lifecycle-DbM29FLK.js} +2 -2
- package/dist/{lifecycle-B7CHqKlF.js.map → lifecycle-DbM29FLK.js.map} +1 -1
- package/dist/{lifecycle-D-rL81tT.js → lifecycle-SXaWssN9.js} +2 -2
- package/dist/{lifecycle-D-rL81tT.js.map → lifecycle-SXaWssN9.js.map} +1 -1
- package/dist/main.js +17 -17
- package/dist/{mcp-workspace-header-ucs2SDST.js → mcp-workspace-header-DRCCWlOi.js} +2 -2
- package/dist/{mcp-workspace-header-ucs2SDST.js.map → mcp-workspace-header-DRCCWlOi.js.map} +1 -1
- package/dist/{models-C16mBK2M.js → models-Dz8d_SnI.js} +3 -3
- package/dist/{models-C16mBK2M.js.map → models-Dz8d_SnI.js.map} +1 -1
- package/dist/{orchestration-CtM6FYNx.js → orchestration-BrJwZxMN.js} +2 -2
- package/dist/{orchestration-CtM6FYNx.js.map → orchestration-BrJwZxMN.js.map} +1 -1
- package/dist/paths-CTr59UC6.js +2 -0
- package/dist/{paths-wLC0InjX.js → paths-D7_SAaIQ.js} +4 -4
- package/dist/{paths-wLC0InjX.js.map → paths-D7_SAaIQ.js.map} +1 -1
- package/dist/{peer-mcp-personas-V6stFvpq.js → peer-mcp-personas-Bd56EmiO.js} +533 -55
- package/dist/peer-mcp-personas-Bd56EmiO.js.map +1 -0
- package/dist/{plan-review-hook-Lf9ISdF6.js → plan-review-hook-CVZsG9MZ.js} +3 -3
- package/dist/{plan-review-hook-Lf9ISdF6.js.map → plan-review-hook-CVZsG9MZ.js.map} +1 -1
- package/dist/{prompt-submit-hook-BlijaOn7.js → prompt-submit-hook-BW92FX2D.js} +3 -3
- package/dist/{prompt-submit-hook-BlijaOn7.js.map → prompt-submit-hook-BW92FX2D.js.map} +1 -1
- package/dist/{provision-BpL6gZIt.js → provision-B53wbHwa.js} +4 -4
- package/dist/{provision-BpL6gZIt.js.map → provision-B53wbHwa.js.map} +1 -1
- package/dist/{self-invocation-CKMjcA5F.js → self-invocation-CP_SOkrr.js} +2 -2
- package/dist/{self-invocation-CKMjcA5F.js.map → self-invocation-CP_SOkrr.js.map} +1 -1
- package/dist/{serve-CXUf7RtJ.js → serve-aZCEYFe5.js} +37 -24
- package/dist/serve-aZCEYFe5.js.map +1 -0
- package/dist/{server-setup-Bppjt9xO.js → server-setup-DlztZAGT.js} +258 -99
- package/dist/server-setup-DlztZAGT.js.map +1 -0
- package/dist/{start-C1-jrHfU.js → start-DwNiXv5N.js} +3 -3
- package/dist/{start-C1-jrHfU.js.map → start-DwNiXv5N.js.map} +1 -1
- package/dist/{stop-gate-hook-DgJ6sW8N.js → stop-gate-hook-DriRc9xN.js} +3 -3
- package/dist/{stop-gate-hook-DgJ6sW8N.js.map → stop-gate-hook-DriRc9xN.js.map} +1 -1
- package/dist/{stop-gate-policy-DG5yYWGn.js → stop-gate-policy-DMPanpoR.js} +2 -2
- package/dist/{stop-gate-policy-DG5yYWGn.js.map → stop-gate-policy-DMPanpoR.js.map} +1 -1
- package/dist/{token-CnlB0884.js → token-BGCjZwtj.js} +2 -2
- package/dist/token-BGCjZwtj.js.map +1 -0
- package/dist/{worker-dispatch-zW8Zi69V.js → worker-dispatch-BCTMyNE-.js} +2 -2
- package/dist/{worker-dispatch-zW8Zi69V.js.map → worker-dispatch-BCTMyNE-.js.map} +1 -1
- package/package.json +2 -1
- package/dist/attribution-settings-B8M2fvhz.js.map +0 -1
- package/dist/claude-e-JIQGhR.js.map +0 -1
- package/dist/engine-DMveEa9x.js +0 -2
- package/dist/lifecycle-CbmHMMSD.js +0 -2
- package/dist/lifecycle-DTcZwa7U.js +0 -2
- package/dist/paths-CV9K7Xqm.js +0 -2
- package/dist/peer-mcp-personas-V6stFvpq.js.map +0 -1
- package/dist/serve-CXUf7RtJ.js.map +0 -1
- package/dist/server-setup-Bppjt9xO.js.map +0 -1
- package/dist/token-CnlB0884.js.map +0 -1
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
|
|
2
2
|
import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
|
|
3
|
-
import { t as PATHS } from "./paths-
|
|
4
|
-
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-
|
|
3
|
+
import { t as PATHS } from "./paths-D7_SAaIQ.js";
|
|
4
|
+
import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-BGCjZwtj.js";
|
|
5
5
|
import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
|
|
6
|
-
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-
|
|
7
|
-
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-
|
|
6
|
+
import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-DbM29FLK.js";
|
|
7
|
+
import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-DRCCWlOi.js";
|
|
8
8
|
import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
|
|
9
|
-
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-
|
|
10
|
-
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-
|
|
11
|
-
import { t as liveExec } from "./orchestration-
|
|
9
|
+
import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-SXaWssN9.js";
|
|
10
|
+
import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DriRc9xN.js";
|
|
11
|
+
import { t as liveExec } from "./orchestration-BrJwZxMN.js";
|
|
12
12
|
import { createRequire } from "node:module";
|
|
13
13
|
import consola from "consola";
|
|
14
14
|
import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
|
|
@@ -995,6 +995,9 @@ function enumProp$1(values, description) {
|
|
|
995
995
|
};
|
|
996
996
|
}
|
|
997
997
|
//#endregion
|
|
998
|
+
//#region src/lib/gemini-review-model.ts
|
|
999
|
+
const GEMINI_REVIEW_DEFAULT_MODEL = "gemini-3.1-pro-preview";
|
|
1000
|
+
//#endregion
|
|
998
1001
|
//#region src/lib/fleet/mesh-egress-agent.ts
|
|
999
1002
|
/**
|
|
1000
1003
|
* Runtime-aware "route this fetch through the mesh loopback egress proxy" for a
|
|
@@ -18914,7 +18917,7 @@ function logAudit$1(record) {
|
|
|
18914
18917
|
try {
|
|
18915
18918
|
const fs = await import("node:fs/promises");
|
|
18916
18919
|
const path = await import("node:path");
|
|
18917
|
-
const { PATHS } = await import("./paths-
|
|
18920
|
+
const { PATHS } = await import("./paths-CTr59UC6.js");
|
|
18918
18921
|
const dir = path.join(PATHS.APP_DIR, "browser-mcp");
|
|
18919
18922
|
await fs.mkdir(dir, { recursive: true });
|
|
18920
18923
|
const line = JSON.stringify({
|
|
@@ -19494,6 +19497,354 @@ function currentInFlight() {
|
|
|
19494
19497
|
return inFlight$2;
|
|
19495
19498
|
}
|
|
19496
19499
|
//#endregion
|
|
19500
|
+
//#region src/lib/prompt-cache.ts
|
|
19501
|
+
/**
|
|
19502
|
+
* Conservative eligibility floor, in UTF-8 BYTES — never `.length`, which
|
|
19503
|
+
* counts UTF-16 code units and undercounts anything outside the BMP (an
|
|
19504
|
+
* emoji is 2 code units but 4 bytes). This is a proxy for "the prefix is
|
|
19505
|
+
* obviously large enough that marking it as a cache breakpoint is worth one
|
|
19506
|
+
* of the scarce marker slots," and it deliberately does NOT claim to be a
|
|
19507
|
+
* token count: byte-to-token density varies by tokenizer and content — CJK
|
|
19508
|
+
* text carries MORE tokens per byte than ASCII prose (undercounting risk is
|
|
19509
|
+
* the SAFE direction: we'd skip a marker that might have qualified), while a
|
|
19510
|
+
* long run of a repeated character or repeated whitespace carries FEWER
|
|
19511
|
+
* tokens per byte than either, since BPE merges long runs into very few
|
|
19512
|
+
* tokens (overcounting risk: a byte count clearing the floor doesn't
|
|
19513
|
+
* guarantee the real token count clears Anthropic's or Copilot's per-model
|
|
19514
|
+
* minimum). No fixed byte threshold can bound that adversarial case; this
|
|
19515
|
+
* value is chosen so ordinary Claude Code system prompts and tool schemas
|
|
19516
|
+
* (natural-language / JSON, not deliberately repetitive) reliably qualify,
|
|
19517
|
+
* while genuinely small prefixes never burn a marker for no benefit.
|
|
19518
|
+
*/
|
|
19519
|
+
const MIN_CACHEABLE_PREFIX_BYTES = 4096;
|
|
19520
|
+
const CACHE_KEY_NAMESPACE = "ghr-cache-v1";
|
|
19521
|
+
const CACHE_DIAGNOSTIC_LIMIT = 128;
|
|
19522
|
+
const GPT56_EXPLICIT_CACHE_MODELS = /* @__PURE__ */ new Set([
|
|
19523
|
+
"gpt-5.6-sol",
|
|
19524
|
+
"gpt-5.6-terra",
|
|
19525
|
+
"gpt-5.6-luna"
|
|
19526
|
+
]);
|
|
19527
|
+
const priorSignatures = /* @__PURE__ */ new Map();
|
|
19528
|
+
function nonNegativeInt(value) {
|
|
19529
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return 0;
|
|
19530
|
+
return Math.floor(value);
|
|
19531
|
+
}
|
|
19532
|
+
/**
|
|
19533
|
+
* Pick the first genuinely POSITIVE numeric candidate from an ordered,
|
|
19534
|
+
* priority-ranked list of usage-shape fields. `??`-chaining these fields is
|
|
19535
|
+
* wrong: it stops at the first field that is merely PRESENT, and a provider
|
|
19536
|
+
* surface that always populates a nested detail object with `0` as a
|
|
19537
|
+
* placeholder (while the real, positive count is reported only in a
|
|
19538
|
+
* lower-priority field, e.g. the top-level one) would have its explicit zero
|
|
19539
|
+
* silently shadow that populated count. Falling through zeros to find a real
|
|
19540
|
+
* positive value fixes that; when every candidate is zero, absent, or
|
|
19541
|
+
* non-numeric, this returns `0` — a genuine all-zero reading, never
|
|
19542
|
+
* `undefined` — so downstream `nonNegativeInt` always has a countable value.
|
|
19543
|
+
*/
|
|
19544
|
+
function firstPositive(...candidates) {
|
|
19545
|
+
for (const c of candidates) if (typeof c === "number" && Number.isFinite(c) && c > 0) return c;
|
|
19546
|
+
return 0;
|
|
19547
|
+
}
|
|
19548
|
+
function usageDetails(usage) {
|
|
19549
|
+
const input = usage.input_tokens_details ?? {};
|
|
19550
|
+
const prompt = usage.prompt_tokens_details ?? {};
|
|
19551
|
+
return {
|
|
19552
|
+
cached_tokens: firstPositive(input.cached_tokens, prompt.cached_tokens, usage.cache_read_input_tokens),
|
|
19553
|
+
cache_write_tokens: firstPositive(input.cache_write_tokens, input.cache_creation_tokens, prompt.cache_write_tokens, prompt.cache_creation_tokens, usage.cache_write_tokens, usage.cache_creation_input_tokens),
|
|
19554
|
+
cache_ttl_seconds: firstPositive(input.cache_ttl_seconds, prompt.cache_ttl_seconds, usage.cache_ttl_seconds)
|
|
19555
|
+
};
|
|
19556
|
+
}
|
|
19557
|
+
/**
|
|
19558
|
+
* OpenAI totals INCLUDE cached and cache-write tokens. Normalize them into
|
|
19559
|
+
* mutually exclusive buckets so Anthropic and Pi consumers do not count the
|
|
19560
|
+
* same input twice.
|
|
19561
|
+
*/
|
|
19562
|
+
function normalizeOpenAIUsage(usage) {
|
|
19563
|
+
if (!usage) return {
|
|
19564
|
+
totalInput: 0,
|
|
19565
|
+
uncachedInput: 0,
|
|
19566
|
+
output: 0,
|
|
19567
|
+
cacheRead: 0,
|
|
19568
|
+
cacheWrite: 0,
|
|
19569
|
+
totalTokens: 0
|
|
19570
|
+
};
|
|
19571
|
+
const totalInput = nonNegativeInt(usage.input_tokens ?? usage.prompt_tokens);
|
|
19572
|
+
const output = nonNegativeInt(usage.output_tokens ?? usage.completion_tokens);
|
|
19573
|
+
const details = usageDetails(usage);
|
|
19574
|
+
const cacheRead = Math.min(totalInput, nonNegativeInt(details.cached_tokens));
|
|
19575
|
+
const remaining = Math.max(0, totalInput - cacheRead);
|
|
19576
|
+
const cacheWrite = Math.min(remaining, nonNegativeInt(details.cache_write_tokens ?? details.cache_creation_tokens));
|
|
19577
|
+
const uncachedInput = Math.max(0, totalInput - cacheRead - cacheWrite);
|
|
19578
|
+
const reportedTotal = nonNegativeInt(usage.total_tokens);
|
|
19579
|
+
const cacheTtlSeconds = nonNegativeInt(details.cache_ttl_seconds);
|
|
19580
|
+
return {
|
|
19581
|
+
totalInput,
|
|
19582
|
+
uncachedInput,
|
|
19583
|
+
output,
|
|
19584
|
+
cacheRead,
|
|
19585
|
+
cacheWrite,
|
|
19586
|
+
totalTokens: Math.max(reportedTotal, totalInput + output),
|
|
19587
|
+
...cacheTtlSeconds > 0 ? { cacheTtlSeconds } : {}
|
|
19588
|
+
};
|
|
19589
|
+
}
|
|
19590
|
+
function hash(value) {
|
|
19591
|
+
return createHash("sha256").update(value).digest("hex");
|
|
19592
|
+
}
|
|
19593
|
+
function signatureFor(value) {
|
|
19594
|
+
return hash(typeof value === "string" ? value : JSON.stringify(value ?? null));
|
|
19595
|
+
}
|
|
19596
|
+
function serializedBytes(value) {
|
|
19597
|
+
return Buffer.byteLength(typeof value === "string" ? value : JSON.stringify(value ?? null));
|
|
19598
|
+
}
|
|
19599
|
+
function logCacheSignature(args) {
|
|
19600
|
+
if (parseBoolEnv(process.env.GH_ROUTER_LOG_CACHE) !== true) return;
|
|
19601
|
+
const key = `${args.endpoint}:${args.model}:${args.workload}`;
|
|
19602
|
+
const current = {
|
|
19603
|
+
system: signatureFor(args.system),
|
|
19604
|
+
tools: signatureFor(args.tools),
|
|
19605
|
+
messages: signatureFor(args.messages)
|
|
19606
|
+
};
|
|
19607
|
+
const previous = priorSignatures.get(key);
|
|
19608
|
+
let changed = "cold";
|
|
19609
|
+
if (previous) changed = previous.system !== current.system ? "system" : previous.tools !== current.tools ? "tools" : previous.messages !== current.messages ? "messages" : "none";
|
|
19610
|
+
priorSignatures.set(key, current);
|
|
19611
|
+
if (priorSignatures.size > CACHE_DIAGNOSTIC_LIMIT) {
|
|
19612
|
+
const oldest = priorSignatures.keys().next().value;
|
|
19613
|
+
if (oldest !== void 0) priorSignatures.delete(oldest);
|
|
19614
|
+
}
|
|
19615
|
+
consola.info(`cache-signature endpoint=${args.endpoint} model=${args.model} workload=${args.workload} changed=${changed} system_bytes=${serializedBytes(args.system ?? "")} tools_bytes=${serializedBytes(args.tools ?? [])} messages_bytes=${serializedBytes(args.messages ?? [])}`);
|
|
19616
|
+
}
|
|
19617
|
+
function hasResponsesBreakpoint(input) {
|
|
19618
|
+
const visit = (value) => {
|
|
19619
|
+
if (!value || typeof value !== "object") return false;
|
|
19620
|
+
if (Array.isArray(value)) return value.some(visit);
|
|
19621
|
+
const record = value;
|
|
19622
|
+
return record.prompt_cache_breakpoint !== void 0 || Object.values(record).some(visit);
|
|
19623
|
+
};
|
|
19624
|
+
return visit(input);
|
|
19625
|
+
}
|
|
19626
|
+
function gpt56ExplicitCacheEnabled(model) {
|
|
19627
|
+
if (parseBoolEnv(process.env.GH_ROUTER_DISABLE_GPT56_EXPLICIT_CACHE) === true) return false;
|
|
19628
|
+
return GPT56_EXPLICIT_CACHE_MODELS.has(model);
|
|
19629
|
+
}
|
|
19630
|
+
function responsesCacheKey(payload, opts, stablePrefix) {
|
|
19631
|
+
const digest = hash(JSON.stringify({
|
|
19632
|
+
namespace: CACHE_KEY_NAMESPACE,
|
|
19633
|
+
model: payload.model,
|
|
19634
|
+
workload: opts.workload,
|
|
19635
|
+
scope: opts.scope ?? "",
|
|
19636
|
+
stablePrefix,
|
|
19637
|
+
tools: payload.tools ?? []
|
|
19638
|
+
}));
|
|
19639
|
+
return `${CACHE_KEY_NAMESPACE}-${digest.slice(0, 48)}`;
|
|
19640
|
+
}
|
|
19641
|
+
/**
|
|
19642
|
+
* Add GPT-5.6 explicit caching only to router-owned REUSABLE-PREFIX payloads.
|
|
19643
|
+
* Public passthrough routes never call this helper, and existing caller
|
|
19644
|
+
* fields always win. Live shape acceptance is pinned by compatibility probe
|
|
19645
|
+
* `gpt56_explicit_cache_breakpoint`.
|
|
19646
|
+
*
|
|
19647
|
+
* **`"conversation"` is deliberately EXCLUDED and left untouched (a no-op),
|
|
19648
|
+
* same as `"passthrough"`/`"one-shot"`.** A live-verified regression: on a
|
|
19649
|
+
* growing multi-turn conversation (Claude Code's translated main loop and
|
|
19650
|
+
* the worker-agent loop, both of which pass `workload: "conversation"`),
|
|
19651
|
+
* marking only the stable SYSTEM block with an explicit breakpoint measured
|
|
19652
|
+
* substantially worse than leaving caching provider-managed and implicit.
|
|
19653
|
+
* Explicit mode is a distinct
|
|
19654
|
+
* caching strategy from Copilot's provider-managed automatic caching, not an
|
|
19655
|
+
* addition to it — turning it on for a request marks only the bytes an
|
|
19656
|
+
* explicit breakpoint names, and the REST of that request's prefix (here,
|
|
19657
|
+
* the entire un-marked growing message history) stops receiving automatic
|
|
19658
|
+
* prefix-growth caching too. Measured on `gpt-5.6-sol` with explicit mode
|
|
19659
|
+
* force-enabled for conversation workloads: turn 1 (cold)
|
|
19660
|
+
* `input_tokens=27038, cache_write=2031, cache_read=0`; turn 2
|
|
19661
|
+
* `input_tokens=27054, cache_read=2031`; turn 3 `input_tokens=27071,
|
|
19662
|
+
* cache_read=2031` — the ~2k-token system block cached once and never grew,
|
|
19663
|
+
* while the other ~25k tokens of accumulating history were recomputed from
|
|
19664
|
+
* scratch on every single turn. `"reusable-prefix"` calls (peer/advisor/
|
|
19665
|
+
* worker-tool/browser-compressor prefixes reused verbatim across many
|
|
19666
|
+
* DISCRETE calls, never a single request whose own history keeps growing)
|
|
19667
|
+
* do not have this failure mode and keep the explicit treatment below.
|
|
19668
|
+
*/
|
|
19669
|
+
function applyResponsesCachePolicy(payload, opts) {
|
|
19670
|
+
logCacheSignature({
|
|
19671
|
+
endpoint: "/responses",
|
|
19672
|
+
model: payload.model,
|
|
19673
|
+
workload: opts.workload,
|
|
19674
|
+
system: opts.stablePrefix ?? payload.instructions,
|
|
19675
|
+
tools: payload.tools,
|
|
19676
|
+
messages: payload.input
|
|
19677
|
+
});
|
|
19678
|
+
if (opts.workload !== "reusable-prefix" || !gpt56ExplicitCacheEnabled(payload.model) || payload.prompt_cache_key !== void 0 || payload.prompt_cache_options !== void 0 || hasResponsesBreakpoint(payload.input)) return payload;
|
|
19679
|
+
const stablePrefix = opts.stablePrefix ?? payload.instructions;
|
|
19680
|
+
const stableBytes = serializedBytes(stablePrefix ?? "") + serializedBytes(payload.tools ?? []);
|
|
19681
|
+
if (!stablePrefix || stableBytes < MIN_CACHEABLE_PREFIX_BYTES) return payload;
|
|
19682
|
+
const input = typeof payload.input === "string" ? [{
|
|
19683
|
+
role: "user",
|
|
19684
|
+
content: payload.input
|
|
19685
|
+
}] : [...payload.input];
|
|
19686
|
+
const stableContent = [{
|
|
19687
|
+
type: "input_text",
|
|
19688
|
+
text: stablePrefix,
|
|
19689
|
+
prompt_cache_breakpoint: { mode: "explicit" }
|
|
19690
|
+
}];
|
|
19691
|
+
let nextInput;
|
|
19692
|
+
let removeInstructions = false;
|
|
19693
|
+
if (payload.instructions === stablePrefix) {
|
|
19694
|
+
nextInput = [{
|
|
19695
|
+
role: "system",
|
|
19696
|
+
content: stableContent
|
|
19697
|
+
}, ...input];
|
|
19698
|
+
removeInstructions = true;
|
|
19699
|
+
} else {
|
|
19700
|
+
const stableSystemIndex = input.findIndex((item) => item.role === "system" && item.content === stablePrefix);
|
|
19701
|
+
if (stableSystemIndex < 0) return payload;
|
|
19702
|
+
nextInput = [...input];
|
|
19703
|
+
nextInput[stableSystemIndex] = {
|
|
19704
|
+
...nextInput[stableSystemIndex],
|
|
19705
|
+
content: stableContent
|
|
19706
|
+
};
|
|
19707
|
+
}
|
|
19708
|
+
const next = {
|
|
19709
|
+
...payload,
|
|
19710
|
+
input: nextInput,
|
|
19711
|
+
prompt_cache_key: responsesCacheKey(payload, opts, stablePrefix),
|
|
19712
|
+
prompt_cache_options: {
|
|
19713
|
+
mode: "explicit",
|
|
19714
|
+
ttl: "30m"
|
|
19715
|
+
}
|
|
19716
|
+
};
|
|
19717
|
+
if (removeInstructions) delete next.instructions;
|
|
19718
|
+
return next;
|
|
19719
|
+
}
|
|
19720
|
+
function itemHasCacheControl(value) {
|
|
19721
|
+
return !!value && typeof value === "object" && value.cache_control !== void 0;
|
|
19722
|
+
}
|
|
19723
|
+
function hasClaudeCacheControl(body) {
|
|
19724
|
+
if (itemHasCacheControl(body.system)) return true;
|
|
19725
|
+
if (Array.isArray(body.system) && body.system.some(itemHasCacheControl)) return true;
|
|
19726
|
+
if (Array.isArray(body.tools) && body.tools.some(itemHasCacheControl)) return true;
|
|
19727
|
+
if (!Array.isArray(body.messages)) return false;
|
|
19728
|
+
return body.messages.some((message) => {
|
|
19729
|
+
if (!message || typeof message !== "object") return false;
|
|
19730
|
+
const content = message.content;
|
|
19731
|
+
return itemHasCacheControl(content) || Array.isArray(content) && content.some(itemHasCacheControl);
|
|
19732
|
+
});
|
|
19733
|
+
}
|
|
19734
|
+
/**
|
|
19735
|
+
* UTF-8 byte length of the tools array alone — the eligibility floor for the
|
|
19736
|
+
* TOOL breakpoint. Marked on the last non-deferred tool, it caches only the
|
|
19737
|
+
* tools prefix (Claude's wire order is tools, then system, then messages), so
|
|
19738
|
+
* its own size — not the combined system+tools size — is what determines
|
|
19739
|
+
* whether that marker is worth spending. See `MIN_CACHEABLE_PREFIX_BYTES`.
|
|
19740
|
+
*/
|
|
19741
|
+
function claudeToolsPrefixBytes(body) {
|
|
19742
|
+
return serializedBytes(body.tools ?? []);
|
|
19743
|
+
}
|
|
19744
|
+
/**
|
|
19745
|
+
* UTF-8 byte length of tools + system combined — the eligibility floor for
|
|
19746
|
+
* the SYSTEM breakpoint. Marked on the last system text block, it caches
|
|
19747
|
+
* everything up to and including system (tools THEN system in wire order),
|
|
19748
|
+
* so the combined size is the right measure — checked SEPARATELY from the
|
|
19749
|
+
* tools-only floor above so a large system prompt behind tiny tools doesn't
|
|
19750
|
+
* smuggle a useless tools-only marker in under the combined total, and a
|
|
19751
|
+
* large tools array behind an empty system doesn't get double-counted as
|
|
19752
|
+
* "small" just because system alone is tiny.
|
|
19753
|
+
*/
|
|
19754
|
+
function claudeSystemPrefixBytes(body) {
|
|
19755
|
+
return claudeToolsPrefixBytes(body) + serializedBytes(body.system ?? "");
|
|
19756
|
+
}
|
|
19757
|
+
function markClaudeSystem(body) {
|
|
19758
|
+
if (typeof body.system === "string" && body.system.length > 0) {
|
|
19759
|
+
body.system = [{
|
|
19760
|
+
type: "text",
|
|
19761
|
+
text: body.system,
|
|
19762
|
+
cache_control: { type: "ephemeral" }
|
|
19763
|
+
}];
|
|
19764
|
+
return true;
|
|
19765
|
+
}
|
|
19766
|
+
if (!Array.isArray(body.system)) return false;
|
|
19767
|
+
for (let index = body.system.length - 1; index >= 0; index--) {
|
|
19768
|
+
const block = body.system[index];
|
|
19769
|
+
if (block && typeof block === "object" && block.type === "text" && typeof block.text === "string") {
|
|
19770
|
+
body.system[index] = {
|
|
19771
|
+
...block,
|
|
19772
|
+
cache_control: { type: "ephemeral" }
|
|
19773
|
+
};
|
|
19774
|
+
return true;
|
|
19775
|
+
}
|
|
19776
|
+
}
|
|
19777
|
+
return false;
|
|
19778
|
+
}
|
|
19779
|
+
function markClaudeTool(body) {
|
|
19780
|
+
if (!Array.isArray(body.tools)) return false;
|
|
19781
|
+
for (let index = body.tools.length - 1; index >= 0; index--) {
|
|
19782
|
+
const tool = body.tools[index];
|
|
19783
|
+
if (tool && typeof tool === "object" && tool.defer_loading !== true) {
|
|
19784
|
+
body.tools[index] = {
|
|
19785
|
+
...tool,
|
|
19786
|
+
cache_control: { type: "ephemeral" }
|
|
19787
|
+
};
|
|
19788
|
+
return true;
|
|
19789
|
+
}
|
|
19790
|
+
}
|
|
19791
|
+
return false;
|
|
19792
|
+
}
|
|
19793
|
+
/**
|
|
19794
|
+
* Apply the bounded Claude anchor policy to router-generated Messages bodies.
|
|
19795
|
+
* Caller-owned marker layouts are returned byte-for-byte unchanged.
|
|
19796
|
+
*
|
|
19797
|
+
* Marks at most TWO breakpoints — the last non-deferred tool and the stable
|
|
19798
|
+
* system boundary — each gated on its OWN eligibility check
|
|
19799
|
+
* (`claudeToolsPrefixBytes` / `claudeSystemPrefixBytes`) rather than one
|
|
19800
|
+
* combined check, so a large system prompt behind tiny tools doesn't also
|
|
19801
|
+
* mark a tools breakpoint too small to be worth a marker slot, and vice
|
|
19802
|
+
* versa. Anthropic's own hard ceiling is FOUR `cache_control` blocks per
|
|
19803
|
+
* request (probe `cache_control_marker_limit_5`); this policy only ever
|
|
19804
|
+
* spends up to two of them (`hasClaudeCacheControl` already refuses to run
|
|
19805
|
+
* at all once the caller has marked anything itself, so the two never
|
|
19806
|
+
* combine with a caller-owned marker to approach that ceiling).
|
|
19807
|
+
*
|
|
19808
|
+
* There used to be a third, message-level marking path gated on
|
|
19809
|
+
* `opts.workload === "conversation"`. It was removed as dead code: every
|
|
19810
|
+
* production call site of this function (`src/routes/mcp/handler.ts`,
|
|
19811
|
+
* `src/services/advisor/advisor.ts`) passes `workload: "reusable-prefix"`,
|
|
19812
|
+
* so the per-message branch never ran outside its own unit test, which gave
|
|
19813
|
+
* false confidence that production traffic exercised it. `CacheWorkload`
|
|
19814
|
+
* keeps `"conversation"` as a shared enum value — the Responses-side policy,
|
|
19815
|
+
* which has no per-message logic of its own, still uses it — so passing it
|
|
19816
|
+
* here remains type-valid; it now behaves identically to
|
|
19817
|
+
* `"reusable-prefix"`.
|
|
19818
|
+
*/
|
|
19819
|
+
function applyClaudeCachePolicy(rawBody, opts) {
|
|
19820
|
+
if (opts.workload === "passthrough" || opts.workload === "one-shot" || parseBoolEnv(process.env.GH_ROUTER_DISABLE_CLAUDE_CACHE_POLICY) === true) return rawBody;
|
|
19821
|
+
let parsed;
|
|
19822
|
+
try {
|
|
19823
|
+
parsed = JSON.parse(rawBody);
|
|
19824
|
+
} catch {
|
|
19825
|
+
return rawBody;
|
|
19826
|
+
}
|
|
19827
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return rawBody;
|
|
19828
|
+
const body = parsed;
|
|
19829
|
+
if (typeof body.model !== "string" || !body.model.startsWith("claude-") || hasClaudeCacheControl(body)) return rawBody;
|
|
19830
|
+
const toolsEligible = claudeToolsPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
|
|
19831
|
+
const systemEligible = claudeSystemPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
|
|
19832
|
+
if (!toolsEligible && !systemEligible) return rawBody;
|
|
19833
|
+
let markers = 0;
|
|
19834
|
+
if (toolsEligible && markClaudeTool(body)) markers++;
|
|
19835
|
+
if (systemEligible && markClaudeSystem(body)) markers++;
|
|
19836
|
+
if (markers === 0) return rawBody;
|
|
19837
|
+
logCacheSignature({
|
|
19838
|
+
endpoint: "/messages",
|
|
19839
|
+
model: body.model,
|
|
19840
|
+
workload: opts.workload,
|
|
19841
|
+
system: body.system,
|
|
19842
|
+
tools: body.tools,
|
|
19843
|
+
messages: body.messages
|
|
19844
|
+
});
|
|
19845
|
+
return JSON.stringify(body);
|
|
19846
|
+
}
|
|
19847
|
+
//#endregion
|
|
19497
19848
|
//#region src/lib/vision-preflight.ts
|
|
19498
19849
|
/**
|
|
19499
19850
|
* Outbound vision handling.
|
|
@@ -20418,7 +20769,7 @@ async function callViaChat(model, systemPrompt, userMessage, tool, signal) {
|
|
|
20418
20769
|
* Image parts use `input_image` (vs chat's `image_url`) — see
|
|
20419
20770
|
* `toResponsesContent`. */
|
|
20420
20771
|
async function callViaResponses(model, systemPrompt, userMessage, tool, signal) {
|
|
20421
|
-
const payload = {
|
|
20772
|
+
const payload = applyResponsesCachePolicy({
|
|
20422
20773
|
model,
|
|
20423
20774
|
stream: false,
|
|
20424
20775
|
input: [{
|
|
@@ -20438,7 +20789,10 @@ async function callViaResponses(model, systemPrompt, userMessage, tool, signal)
|
|
|
20438
20789
|
type: "function",
|
|
20439
20790
|
name: tool.name
|
|
20440
20791
|
}
|
|
20441
|
-
}
|
|
20792
|
+
}, {
|
|
20793
|
+
workload: "reusable-prefix",
|
|
20794
|
+
stablePrefix: systemPrompt
|
|
20795
|
+
});
|
|
20442
20796
|
const resp = await createResponses(payload, void 0, signal, true);
|
|
20443
20797
|
const output = Array.isArray(resp.output) ? resp.output : [];
|
|
20444
20798
|
for (const item of output) {
|
|
@@ -23916,6 +24270,10 @@ const FALLBACK_TOKEN_PRICES = Object.freeze({
|
|
|
23916
24270
|
"gemini-3.1-pro-preview": {
|
|
23917
24271
|
in: 200,
|
|
23918
24272
|
out: 1200
|
|
24273
|
+
},
|
|
24274
|
+
"gemini-3.7-flash": {
|
|
24275
|
+
in: 75,
|
|
24276
|
+
out: 375
|
|
23919
24277
|
}
|
|
23920
24278
|
});
|
|
23921
24279
|
/**
|
|
@@ -23976,6 +24334,22 @@ function catalogTokenPrices(modelId) {
|
|
|
23976
24334
|
* takes ~0.9s. That figure is deliberately NOT surfaced to the model, because a
|
|
23977
24335
|
* second speed axis invites optimising a routing choice that policy already
|
|
23978
24336
|
* settles (see the decorrelation note below).
|
|
24337
|
+
*
|
|
24338
|
+
* `gemini-3.7-flash` and the re-measured `gemini-3.1-pro-preview` (2026-08-13,
|
|
24339
|
+
* n=5, one untimed warmup rep) used the script's newer `GH_ROUTER_BENCH_STREAM=1`
|
|
24340
|
+
* mode, which splits wall-clock into TTFT and a decode-phase rate excluding the
|
|
24341
|
+
* first delta's own time+tokens. This row still records the SAME metric family
|
|
24342
|
+
* as every other row here (total tokens / total wall-clock, i.e. the script's
|
|
24343
|
+
* "total tok/s" column) — NOT the new decode-phase figure — so the table stays
|
|
24344
|
+
* internally comparable across rows measured at different times. The decode
|
|
24345
|
+
* figure is materially different and worth knowing for routing decisions that
|
|
24346
|
+
* care about steady-state throughput specifically: at matched effort,
|
|
24347
|
+
* `gemini-3.7-flash` decodes at ~275-315 tok/s (Google's own ~340 tok/s
|
|
24348
|
+
* Artificial Analysis figure, roughly confirmed once TTFT is excluded) against
|
|
24349
|
+
* `gpt-5.6-terra`'s ~135-190 tok/s decode — gemini-3.7-flash is the faster
|
|
24350
|
+
* decoder, but its ~1.1-1.7s TTFT (vs terra's ~0.9-1.0s) drags its TOTAL rate
|
|
24351
|
+
* below terra's on short responses, which is exactly why this row uses total,
|
|
24352
|
+
* not decode, to stay consistent with its neighbors.
|
|
23979
24353
|
*/
|
|
23980
24354
|
const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
|
|
23981
24355
|
"gpt-5.6-luna": 120,
|
|
@@ -23988,9 +24362,10 @@ const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
|
|
|
23988
24362
|
"gpt-5.6-sol": 75,
|
|
23989
24363
|
"grok-4.5": 70,
|
|
23990
24364
|
"gpt-5.5": 65,
|
|
24365
|
+
"gemini-3.7-flash": 91,
|
|
23991
24366
|
"gemini-3.6-flash": 45,
|
|
23992
24367
|
"gemini-3.5-flash": 40,
|
|
23993
|
-
"gemini-3.1-pro-preview":
|
|
24368
|
+
"gemini-3.1-pro-preview": 33
|
|
23994
24369
|
});
|
|
23995
24370
|
/** Returns the approximate, indicative output speed when it was measured. */
|
|
23996
24371
|
function indicativeTokensPerSecond(modelId) {
|
|
@@ -24467,6 +24842,10 @@ function neutralToolsToResponses(tools) {
|
|
|
24467
24842
|
/** Assemble the full Responses payload from the neutral request shape. */
|
|
24468
24843
|
function assembleResponsesPayload(opts) {
|
|
24469
24844
|
const input = [];
|
|
24845
|
+
if (opts.dynamicInstructions) input.push({
|
|
24846
|
+
role: "system",
|
|
24847
|
+
content: opts.dynamicInstructions
|
|
24848
|
+
});
|
|
24470
24849
|
for (const m of opts.messages) for (const item of neutralMessageToResponsesInput(m)) input.push(item);
|
|
24471
24850
|
const payload = {
|
|
24472
24851
|
model: opts.model,
|
|
@@ -24483,7 +24862,10 @@ function assembleResponsesPayload(opts) {
|
|
|
24483
24862
|
if (typeof opts.maxOutputTokens === "number" && opts.maxOutputTokens > 0) payload.max_output_tokens = Math.max(opts.maxOutputTokens, RESPONSES_MIN_MAX_OUTPUT_TOKENS);
|
|
24484
24863
|
if (opts.stopSequences && opts.stopSequences.length > 0) payload.stop = [...opts.stopSequences];
|
|
24485
24864
|
if (opts.parallelToolCalls === false) payload.parallel_tool_calls = false;
|
|
24486
|
-
return payload
|
|
24865
|
+
return opts.cachePolicy ? applyResponsesCachePolicy(payload, {
|
|
24866
|
+
...opts.cachePolicy,
|
|
24867
|
+
stablePrefix: opts.cachePolicy.stablePrefix ?? opts.instructions
|
|
24868
|
+
}) : payload;
|
|
24487
24869
|
}
|
|
24488
24870
|
//#endregion
|
|
24489
24871
|
//#region src/lib/worker-agent/context-budget.ts
|
|
@@ -25018,7 +25400,10 @@ function mapResponsesUsage(u) {
|
|
|
25018
25400
|
prompt_tokens: u.input_tokens ?? 0,
|
|
25019
25401
|
completion_tokens: u.output_tokens ?? 0,
|
|
25020
25402
|
total_tokens: u.total_tokens ?? 0,
|
|
25021
|
-
prompt_tokens_details: u.input_tokens_details
|
|
25403
|
+
prompt_tokens_details: u.input_tokens_details != null ? {
|
|
25404
|
+
cached_tokens: u.input_tokens_details.cached_tokens ?? 0,
|
|
25405
|
+
cache_write_tokens: u.input_tokens_details.cache_write_tokens ?? u.input_tokens_details.cache_creation_tokens ?? 0
|
|
25406
|
+
} : void 0
|
|
25022
25407
|
};
|
|
25023
25408
|
}
|
|
25024
25409
|
/**
|
|
@@ -25267,6 +25652,7 @@ function buildResponsesPayload(context, resolved) {
|
|
|
25267
25652
|
messages,
|
|
25268
25653
|
tools: piToolsToNeutral(context.tools),
|
|
25269
25654
|
reasoningEffort: resolved.thinking,
|
|
25655
|
+
cachePolicy: { workload: "conversation" },
|
|
25270
25656
|
stream: true
|
|
25271
25657
|
});
|
|
25272
25658
|
}
|
|
@@ -25489,12 +25875,13 @@ function emptyUsage() {
|
|
|
25489
25875
|
}
|
|
25490
25876
|
function deriveUsage(u) {
|
|
25491
25877
|
if (!u) return emptyUsage();
|
|
25878
|
+
const normalized = normalizeOpenAIUsage(u);
|
|
25492
25879
|
return {
|
|
25493
|
-
input:
|
|
25494
|
-
output:
|
|
25495
|
-
cacheRead:
|
|
25496
|
-
cacheWrite:
|
|
25497
|
-
totalTokens:
|
|
25880
|
+
input: normalized.uncachedInput,
|
|
25881
|
+
output: normalized.output,
|
|
25882
|
+
cacheRead: normalized.cacheRead,
|
|
25883
|
+
cacheWrite: normalized.cacheWrite,
|
|
25884
|
+
totalTokens: normalized.totalTokens,
|
|
25498
25885
|
cost: {
|
|
25499
25886
|
input: 0,
|
|
25500
25887
|
output: 0,
|
|
@@ -27013,6 +27400,7 @@ function shimDefaultsToXhigh(id) {
|
|
|
27013
27400
|
* live tool list would be a silent regression (the snippet would name
|
|
27014
27401
|
* a tool the live catalog doesn't expose).
|
|
27015
27402
|
*/
|
|
27403
|
+
const REVIEW_FAST_DEFAULT_MODEL = "gemini-3.7-flash";
|
|
27016
27404
|
/**
|
|
27017
27405
|
* Gate for the `stand_in` tool.
|
|
27018
27406
|
*
|
|
@@ -27021,9 +27409,8 @@ function shimDefaultsToXhigh(id) {
|
|
|
27021
27409
|
* - an OpenAI frontier model (`gpt-5.6-sol`, else `gpt-5.5` — see
|
|
27022
27410
|
* `resolveOpenAiFrontier`)
|
|
27023
27411
|
* - `claude-opus-5` (stand_in's Anthropic slot)
|
|
27024
|
-
* -
|
|
27025
|
-
*
|
|
27026
|
-
* the GA slug renames `gemini-3.1-pro-preview` → `gemini-3.1-pro`)
|
|
27412
|
+
* - the preferred Gemini reviewer model (`gemini-3.1-pro-preview`, falling
|
|
27413
|
+
* back to `gemini-3.7-flash` after the preview's 2026-09-01 removal)
|
|
27027
27414
|
*
|
|
27028
27415
|
* If any one is missing, `stand_in` is dropped from `tools/list` AND
|
|
27029
27416
|
* fails `tools/call` with -32601 (mirroring the `worker` capability's
|
|
@@ -27032,10 +27419,47 @@ function shimDefaultsToXhigh(id) {
|
|
|
27032
27419
|
* `claude-opus-5` is a single-segment slug (dotted == dashed), so the
|
|
27033
27420
|
* catalog probe matches Copilot's actual id shape directly.
|
|
27034
27421
|
*/
|
|
27035
|
-
|
|
27422
|
+
/**
|
|
27423
|
+
* Any live-catalog model matching Google's `gemini-3.x-pro` family, excluding
|
|
27424
|
+
* the two known literals — catches a GA rename of the preview slug (e.g.
|
|
27425
|
+
* `gemini-3.1-pro-preview` -> `gemini-3.1-pro`) so a vendor rename doesn't
|
|
27426
|
+
* silently downgrade every Gemini-gated resolver to the flash fallback while a
|
|
27427
|
+
* real pro-tier successor is actually present in the catalog. This is the
|
|
27428
|
+
* same regex the removed `geminiAvailable()` used, for the same reason —
|
|
27429
|
+
* losing it here was a real regression caught in review, not a deliberate
|
|
27430
|
+
* simplification.
|
|
27431
|
+
*/
|
|
27432
|
+
function findGeminiProGaRename(models) {
|
|
27433
|
+
return models.find((m) => /^gemini-3\..*pro/i.test(m.id) && m.id !== "gemini-3.1-pro-preview" && m.id !== "gemini-3.7-flash")?.id;
|
|
27434
|
+
}
|
|
27435
|
+
function resolveGeminiReviewModel(source = state) {
|
|
27036
27436
|
const models = source.models?.data;
|
|
27037
|
-
if (!models) return
|
|
27038
|
-
|
|
27437
|
+
if (!models) return void 0;
|
|
27438
|
+
if (models.some((m) => m.id === "gemini-3.1-pro-preview")) return GEMINI_REVIEW_DEFAULT_MODEL;
|
|
27439
|
+
const gaRename = findGeminiProGaRename(models);
|
|
27440
|
+
if (gaRename) return gaRename;
|
|
27441
|
+
if (models.some((m) => m.id === "gemini-3.7-flash")) return REVIEW_FAST_DEFAULT_MODEL;
|
|
27442
|
+
}
|
|
27443
|
+
/**
|
|
27444
|
+
* Gemini review candidates in preference order, for resolvers that ALSO need
|
|
27445
|
+
* `firstPresentInCatalog`'s `requireToolCalls`/`minContextTokens` enforcement
|
|
27446
|
+
* (`resolveGeminiReviewModel()` only checks id presence, not those capability
|
|
27447
|
+
* flags). Mirrors `resolveGeminiReviewModel()`'s own preference order: the
|
|
27448
|
+
* known preview id, then a GA rename of it, then the flash fallback — kept as
|
|
27449
|
+
* a shared helper so `reviewerModel()`/`brainstormModel()` can't drift from
|
|
27450
|
+
* `resolveGeminiReviewModel()`'s GA-rename handling the way the hardcoded
|
|
27451
|
+
* per-resolver chains did before this was extracted.
|
|
27452
|
+
*/
|
|
27453
|
+
function geminiReviewChainCandidates() {
|
|
27454
|
+
const gaRename = findGeminiProGaRename(state.models?.data ?? []);
|
|
27455
|
+
return [
|
|
27456
|
+
GEMINI_REVIEW_DEFAULT_MODEL,
|
|
27457
|
+
...gaRename ? [gaRename] : [],
|
|
27458
|
+
REVIEW_FAST_DEFAULT_MODEL
|
|
27459
|
+
];
|
|
27460
|
+
}
|
|
27461
|
+
function geminiAvailable(source = state) {
|
|
27462
|
+
return resolveGeminiReviewModel(source) != null;
|
|
27039
27463
|
}
|
|
27040
27464
|
/**
|
|
27041
27465
|
* First id in `chain` that is present in the live catalog. With
|
|
@@ -27107,7 +27531,7 @@ function nativeSubagentModel() {
|
|
|
27107
27531
|
* was one model checking its own output. Not merely the same lab: the same
|
|
27108
27532
|
* model. Two independent blind audits flagged it, and the repo already applies
|
|
27109
27533
|
* the opposite rule one layer down, where `worker-review` runs
|
|
27110
|
-
* `
|
|
27534
|
+
* `GEMINI_REVIEW_DEFAULT_MODEL` precisely so the reviewer's lab is decorrelated from the
|
|
27111
27535
|
* producer's.
|
|
27112
27536
|
*
|
|
27113
27537
|
* The Anthropic lead and the OpenAI-frontier `implementer` are the two producers
|
|
@@ -27115,7 +27539,7 @@ function nativeSubagentModel() {
|
|
|
27115
27539
|
* frontier remains the fallback: a same-lab reviewer still beats no reviewer.
|
|
27116
27540
|
*/
|
|
27117
27541
|
function reviewerModel() {
|
|
27118
|
-
return firstPresentInCatalog([
|
|
27542
|
+
return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
|
|
27119
27543
|
}
|
|
27120
27544
|
/** Model for `brainstorm`. Absent → inherits the lead's model.
|
|
27121
27545
|
*
|
|
@@ -27124,7 +27548,7 @@ function reviewerModel() {
|
|
|
27124
27548
|
* `implementer`/`reviewer`, so a same-lab brainstormer would mostly restate
|
|
27125
27549
|
* what the lead already thought of. */
|
|
27126
27550
|
function brainstormModel() {
|
|
27127
|
-
return firstPresentInCatalog([
|
|
27551
|
+
return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
|
|
27128
27552
|
}
|
|
27129
27553
|
/** Model for `scribe`. Absent → inherits the lead's model.
|
|
27130
27554
|
*
|
|
@@ -27146,18 +27570,24 @@ function scribeModel() {
|
|
|
27146
27570
|
* impostor wearing the cheap agent's name.
|
|
27147
27571
|
*
|
|
27148
27572
|
* `gpt-5.6-luna` leads because it is the cheapest 1M-context model in the
|
|
27149
|
-
* catalog; `gemini-3.
|
|
27573
|
+
* catalog; `gemini-3.7-flash` remains the cross-vendor fallback so an OpenAI-side
|
|
27150
27574
|
* outage does not remove the scout. Both entries must continue advertising at
|
|
27151
27575
|
* least 1M context so Claude Code's `[1m]` accounting remains honest if an
|
|
27152
27576
|
* upstream catalog entry shrinks.
|
|
27153
27577
|
*
|
|
27578
|
+
* The fallback moved off `gemini-3.6-flash` on 2026-08-13: `gemini-3.7-flash`
|
|
27579
|
+
* is strictly better on every axis this chain cares about — half the price
|
|
27580
|
+
* (75/375 vs 150/750 per 1M), materially faster (measured tool-call p50 ~1.2s
|
|
27581
|
+
* against 3.6's ~2.6s), same 1M window, same vendor, so the cross-vendor
|
|
27582
|
+
* property the fallback exists for is preserved.
|
|
27583
|
+
*
|
|
27154
27584
|
* This chain deliberately uses literal ids rather than `EXPLORE_DEFAULT_MODEL`:
|
|
27155
27585
|
* the explore worker default and scout's cross-vendor fallback are independent
|
|
27156
27586
|
* policies, so retuning one must not silently collapse the other. There is no
|
|
27157
27587
|
* 400K last resort. On a catalog carrying neither chain member, `scout` is
|
|
27158
27588
|
* dropped rather than inheriting the lead or presenting a narrower-context agent.
|
|
27159
27589
|
*/
|
|
27160
|
-
const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.
|
|
27590
|
+
const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.7-flash"]);
|
|
27161
27591
|
function scoutModel() {
|
|
27162
27592
|
return firstPresentInCatalog(SCOUT_MODEL_CHAIN, {
|
|
27163
27593
|
requireToolCalls: true,
|
|
@@ -27172,7 +27602,17 @@ function scoutModel() {
|
|
|
27172
27602
|
* well-specified, mechanical changes at a lower tier. Both entries are 1M+;
|
|
27173
27603
|
* their different speed and effort properties stay out of shared claims. */
|
|
27174
27604
|
function implementerFastModel() {
|
|
27175
|
-
return firstPresentInCatalog(["gpt-5.6-terra",
|
|
27605
|
+
return firstPresentInCatalog(["gpt-5.6-terra", GEMINI_REVIEW_DEFAULT_MODEL], {
|
|
27606
|
+
requireToolCalls: true,
|
|
27607
|
+
minContextTokens: ONE_M_TOKENS
|
|
27608
|
+
});
|
|
27609
|
+
}
|
|
27610
|
+
/** Model for `reviewer-fast` — the cheaper Google review tier. Absent → the
|
|
27611
|
+
* agent is dropped. Single-entry by design: inheriting the lead or falling
|
|
27612
|
+
* across labs would defeat both its cost purpose and its decorrelation from
|
|
27613
|
+
* the OpenAI-backed implementer. */
|
|
27614
|
+
function reviewerFastModel() {
|
|
27615
|
+
return firstPresentInCatalog([REVIEW_FAST_DEFAULT_MODEL], {
|
|
27176
27616
|
requireToolCalls: true,
|
|
27177
27617
|
minContextTokens: ONE_M_TOKENS
|
|
27178
27618
|
});
|
|
@@ -27473,6 +27913,7 @@ function resolveOpusCriticModel() {
|
|
|
27473
27913
|
}
|
|
27474
27914
|
function activePersonas() {
|
|
27475
27915
|
return PERSONAS_READ.filter((p) => !p.requiresGeminiCatalog || geminiAvailable()).map((p) => {
|
|
27916
|
+
if (p.requiresGeminiCatalog) return resolveGeminiPersona(p, resolveGeminiReviewModel());
|
|
27476
27917
|
if (p.toolNameHttp !== "opus_critic") return p;
|
|
27477
27918
|
const model = resolveOpusCriticModel();
|
|
27478
27919
|
const allowedEfforts = model === "claude-opus-5" ? [
|
|
@@ -27761,7 +28202,7 @@ function jsonPathPreflightCap(body, scope) {
|
|
|
27761
28202
|
async function dispatchModelCall(args) {
|
|
27762
28203
|
const resolvedModel = resolveModel(args.model);
|
|
27763
28204
|
if (args.endpoint === "/v1/responses") {
|
|
27764
|
-
const payload = {
|
|
28205
|
+
const payload = applyResponsesCachePolicy({
|
|
27765
28206
|
model: resolvedModel,
|
|
27766
28207
|
instructions: args.instructions,
|
|
27767
28208
|
input: [{
|
|
@@ -27776,7 +28217,7 @@ async function dispatchModelCall(args) {
|
|
|
27776
28217
|
}],
|
|
27777
28218
|
stream: false,
|
|
27778
28219
|
reasoning: { effort: args.effort }
|
|
27779
|
-
};
|
|
28220
|
+
}, { workload: "reusable-prefix" });
|
|
27780
28221
|
return extractResponsesText(await withTransientRetry(() => createResponses(payload, void 0, args.signal), {
|
|
27781
28222
|
signal: args.signal,
|
|
27782
28223
|
label: resolvedModel
|
|
@@ -27784,7 +28225,7 @@ async function dispatchModelCall(args) {
|
|
|
27784
28225
|
}
|
|
27785
28226
|
if (args.endpoint === "/v1/messages") {
|
|
27786
28227
|
const maxTokens = args.effort === "low" ? 4096 : args.effort === "medium" ? 8192 : args.effort === "high" ? 16384 : 32768;
|
|
27787
|
-
const body = JSON.stringify({
|
|
28228
|
+
const body = applyClaudeCachePolicy(JSON.stringify({
|
|
27788
28229
|
model: resolvedModel,
|
|
27789
28230
|
max_tokens: maxTokens,
|
|
27790
28231
|
system: args.instructions,
|
|
@@ -27807,7 +28248,7 @@ async function dispatchModelCall(args) {
|
|
|
27807
28248
|
role: "user",
|
|
27808
28249
|
content: args.userText
|
|
27809
28250
|
}]
|
|
27810
|
-
});
|
|
28251
|
+
}), { workload: "reusable-prefix" });
|
|
27811
28252
|
return extractMessagesText(await (await withTransientRetry(() => createMessages(body, void 0, args.signal), {
|
|
27812
28253
|
signal: args.signal,
|
|
27813
28254
|
label: resolvedModel
|
|
@@ -29309,7 +29750,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29309
29750
|
}
|
|
29310
29751
|
const conversationText = renderConversationAsText(conversation, maxUnits, measure);
|
|
29311
29752
|
if (advisorUsesResponses(resolvedAdvisorModel)) {
|
|
29312
|
-
const payload = {
|
|
29753
|
+
const payload = applyResponsesCachePolicy({
|
|
29313
29754
|
model: resolvedAdvisorModel,
|
|
29314
29755
|
instructions: advisorSystem,
|
|
29315
29756
|
input: [{
|
|
@@ -29321,7 +29762,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29321
29762
|
}],
|
|
29322
29763
|
stream: false,
|
|
29323
29764
|
reasoning: { effort: advisorEffort }
|
|
29324
|
-
};
|
|
29765
|
+
}, { workload: "reusable-prefix" });
|
|
29325
29766
|
const response = await withTransientRetry(() => createResponses(payload, void 0, signal), {
|
|
29326
29767
|
signal,
|
|
29327
29768
|
label: resolvedAdvisorModel
|
|
@@ -29346,7 +29787,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29346
29787
|
const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
|
|
29347
29788
|
const limits = advisorEntry?.capabilities?.limits;
|
|
29348
29789
|
const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
|
|
29349
|
-
const advisorBody = JSON.stringify({
|
|
29790
|
+
const advisorBody = applyClaudeCachePolicy(JSON.stringify({
|
|
29350
29791
|
model: resolvedAdvisorModel,
|
|
29351
29792
|
max_tokens: maxTokens,
|
|
29352
29793
|
system: advisorSystem,
|
|
@@ -29359,7 +29800,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
|
|
|
29359
29800
|
thinking: { type: "adaptive" },
|
|
29360
29801
|
...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
|
|
29361
29802
|
} : {}
|
|
29362
|
-
});
|
|
29803
|
+
}), { workload: "reusable-prefix" });
|
|
29363
29804
|
const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
|
|
29364
29805
|
signal,
|
|
29365
29806
|
label: resolvedAdvisorModel
|
|
@@ -31637,7 +32078,7 @@ function advisorTool(getMessages) {
|
|
|
31637
32078
|
const release = acquireInFlightSlot();
|
|
31638
32079
|
if (!release) throw new Error(`advisor: MCP in-flight cap (${MAX_INFLIGHT_TOOLS_CALL}) saturated; retry shortly`);
|
|
31639
32080
|
try {
|
|
31640
|
-
const
|
|
32081
|
+
const payload = applyResponsesCachePolicy({
|
|
31641
32082
|
model: resolvedModel,
|
|
31642
32083
|
instructions: advisorSystem,
|
|
31643
32084
|
input: [{
|
|
@@ -31649,7 +32090,8 @@ function advisorTool(getMessages) {
|
|
|
31649
32090
|
}],
|
|
31650
32091
|
stream: false,
|
|
31651
32092
|
reasoning: { effort: ADVISOR_DEFAULT_EFFORT }
|
|
31652
|
-
},
|
|
32093
|
+
}, { workload: "reusable-prefix" });
|
|
32094
|
+
const text = extractResponsesText(await createResponses(payload, void 0, signal, true));
|
|
31653
32095
|
if (!text) throw new Error("advisor returned empty output");
|
|
31654
32096
|
return textResult(text);
|
|
31655
32097
|
} finally {
|
|
@@ -32927,7 +33369,7 @@ function appendPlanReminder(messages, planState) {
|
|
|
32927
33369
|
* gemini-3.1-pro-preview is pinned to `high` because the model rejects
|
|
32928
33370
|
* `xhigh` at the wire with a Copilot 400. `high` is the realistic ceiling.
|
|
32929
33371
|
*/
|
|
32930
|
-
const
|
|
33372
|
+
const STAND_IN_MODELS_BASE = Object.freeze([
|
|
32931
33373
|
{
|
|
32932
33374
|
key: "gpt-5.6-sol",
|
|
32933
33375
|
model: "gpt-5.6-sol",
|
|
@@ -32947,6 +33389,13 @@ const STAND_IN_MODELS = Object.freeze([
|
|
|
32947
33389
|
effort: "high"
|
|
32948
33390
|
}
|
|
32949
33391
|
]);
|
|
33392
|
+
function standInModels() {
|
|
33393
|
+
const geminiModel = resolveGeminiReviewModel();
|
|
33394
|
+
return STAND_IN_MODELS_BASE.map((config) => config.key === "gemini-3.1-pro-preview" ? {
|
|
33395
|
+
...config,
|
|
33396
|
+
model: geminiModel ?? "gemini-3.7-flash"
|
|
33397
|
+
} : config);
|
|
33398
|
+
}
|
|
32950
33399
|
const SYSTEM_PROMPT_R1 = `You are one of three frontier reasoning models the user has authorized to stand in for them on a bounded decision while they are unavailable. Your task: pick the best option from those provided.
|
|
32951
33400
|
|
|
32952
33401
|
Respond with ONLY a single JSON object — no prose, no markdown fences, no preamble. Schema:
|
|
@@ -32998,7 +33447,7 @@ const RETRY_PROMPT_SUFFIX = `\n\nYour previous response was not valid JSON match
|
|
|
32998
33447
|
async function runStandIn(input, signal) {
|
|
32999
33448
|
const validIds = new Set(input.options.map((o) => o.id));
|
|
33000
33449
|
const r1UserText = buildRound1UserText(input);
|
|
33001
|
-
const r1 = await Promise.all(
|
|
33450
|
+
const r1 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R1, r1UserText, validIds, signal)));
|
|
33002
33451
|
const successfulR1 = r1.filter((r) => isVote(r.vote));
|
|
33003
33452
|
const nmiR1 = gapAbstainVerdict(successfulR1, r1, null);
|
|
33004
33453
|
if (nmiR1) return nmiR1;
|
|
@@ -33018,7 +33467,7 @@ async function runStandIn(input, signal) {
|
|
|
33018
33467
|
notes: `Only ${successfulR1.length} of 3 models returned a parseable round-1 vote; insufficient signal to run round 2.`
|
|
33019
33468
|
}, r1, null);
|
|
33020
33469
|
const r2UserTextBase = buildRound2UserTextBase(input, r1);
|
|
33021
|
-
const r2 = await Promise.all(
|
|
33470
|
+
const r2 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R2, r2UserTextBase + `\n\nYou are ${cfg.key}. Reconsider and vote.`, validIds, signal)));
|
|
33022
33471
|
const successfulR2 = r2.filter((r) => isVote(r.vote));
|
|
33023
33472
|
if (successfulR2.length < 2) return withDerivedNotes({
|
|
33024
33473
|
verdict: "no_consensus",
|
|
@@ -33195,7 +33644,7 @@ function aggregateVotes(results) {
|
|
|
33195
33644
|
topCount = count;
|
|
33196
33645
|
topSumConfidence = sumConfidence;
|
|
33197
33646
|
}
|
|
33198
|
-
const total =
|
|
33647
|
+
const total = standInModels().length;
|
|
33199
33648
|
if (topChoice && topCount === total) return {
|
|
33200
33649
|
verdict: "consensus",
|
|
33201
33650
|
winner: topChoice,
|
|
@@ -33246,7 +33695,7 @@ function gapAbstainVerdict(successful, r1, r2) {
|
|
|
33246
33695
|
const gapVotes = successful.filter((r) => r.vote.choice === null && r.vote.needMoreInfo);
|
|
33247
33696
|
if (gapVotes.length < 2) return null;
|
|
33248
33697
|
const gaps = gapVotes.map((r) => `- ${r.key}: ${r.vote.needMoreInfo}`).join("\n");
|
|
33249
|
-
const header = gapVotes.length ===
|
|
33698
|
+
const header = gapVotes.length === standInModels().length ? "All three models reported they need more context to decide:" : `${gapVotes.length} of 3 models reported they need more context to decide:`;
|
|
33250
33699
|
return withDerivedNotes({
|
|
33251
33700
|
verdict: "need_more_info",
|
|
33252
33701
|
recommendation: null,
|
|
@@ -33262,7 +33711,7 @@ function gapAbstainVerdict(successful, r1, r2) {
|
|
|
33262
33711
|
*/
|
|
33263
33712
|
function freshestVotes(r1, r2) {
|
|
33264
33713
|
const out = [];
|
|
33265
|
-
for (const cfg of
|
|
33714
|
+
for (const cfg of standInModels()) {
|
|
33266
33715
|
const r2Entry = r2?.find((r) => r.key === cfg.key);
|
|
33267
33716
|
const r1Entry = r1.find((r) => r.key === cfg.key);
|
|
33268
33717
|
const vote = r2Entry && isVote(r2Entry.vote) ? r2Entry.vote : r1Entry && isVote(r1Entry.vote) ? r1Entry.vote : null;
|
|
@@ -33300,7 +33749,7 @@ function withDerivedNotes(result, r1, r2) {
|
|
|
33300
33749
|
}
|
|
33301
33750
|
function voteRecord(r1, r2) {
|
|
33302
33751
|
const record = {};
|
|
33303
|
-
for (const cfg of
|
|
33752
|
+
for (const cfg of standInModels()) {
|
|
33304
33753
|
const r1Entry = r1.find((r) => r.key === cfg.key);
|
|
33305
33754
|
const r2Entry = r2?.find((r) => r.key === cfg.key) ?? null;
|
|
33306
33755
|
record[cfg.key] = {
|
|
@@ -34529,7 +34978,7 @@ const PERSONAS_READ = Object.freeze([
|
|
|
34529
34978
|
{
|
|
34530
34979
|
agentName: "gemini-critic",
|
|
34531
34980
|
toolNameHttp: "gemini_critic",
|
|
34532
|
-
model:
|
|
34981
|
+
model: GEMINI_REVIEW_DEFAULT_MODEL,
|
|
34533
34982
|
endpoint: "/v1/chat/completions",
|
|
34534
34983
|
description: "Adversarial third-lab critic backed by gemini-3.1-pro-preview (Google), strong on formal reasoning, invariants, proofs, and cross-checking another critic's conclusion. It reviews plans, designs, mathematical arguments, and large artifacts for assumption gaps or invariant failures, then returns a focused critique or no-material-objection style verdict. Use when codex_critic's result needs an independent lab check or when the artifact hinges on formal correctness. Not for line-level diff review, use gemini_reviewer or codex_reviewer; pass the artifact and constraints verbatim.",
|
|
34535
34984
|
baseInstructions: GEMINI_CRITIC_BASE,
|
|
@@ -34565,7 +35014,7 @@ const PERSONAS_READ = Object.freeze([
|
|
|
34565
35014
|
{
|
|
34566
35015
|
agentName: "gemini-reviewer",
|
|
34567
35016
|
toolNameHttp: "gemini_reviewer",
|
|
34568
|
-
model:
|
|
35017
|
+
model: GEMINI_REVIEW_DEFAULT_MODEL,
|
|
34569
35018
|
endpoint: "/v1/chat/completions",
|
|
34570
35019
|
description: "Line-level code reviewer backed by gemini-3.1-pro-preview (Google), providing second-lab coverage that catches a different slice of concrete-code defects than codex_reviewer. It reviews diffs, files, or function bodies and returns severity-ranked findings with file:line citations and suggested fixes. Use alongside codex_reviewer when a non-trivial diff benefits from cross-lab code-review coverage, especially around invariants or edge cases. Not for architecture or product-design review, use codex_critic or gemini_critic; pass the code artifact verbatim.",
|
|
34571
35020
|
baseInstructions: GEMINI_REVIEWER_BASE,
|
|
@@ -34734,15 +35183,16 @@ function buildPeerAwarenessSnippet(opts) {
|
|
|
34734
35183
|
const powerBrowseAvailable = opts.browseAvailable && opts.powerBrowseAvailable === true;
|
|
34735
35184
|
const criticList = ["`codex_critic` (gpt-5.6-sol)", "`codex_reviewer` (gpt-5.3-codex)"];
|
|
34736
35185
|
if (opts.geminiAvailable) {
|
|
34737
|
-
|
|
34738
|
-
criticList.push(
|
|
35186
|
+
const geminiModel = opts.geminiModel ?? "gemini-3.1-pro-preview";
|
|
35187
|
+
criticList.push(`\`gemini_reviewer\` (${geminiModel}, line-level code review)`);
|
|
35188
|
+
criticList.push(`\`gemini_critic\` (${geminiModel})`);
|
|
34739
35189
|
}
|
|
34740
35190
|
criticList.push("`opus_critic` (Opus 5)");
|
|
34741
35191
|
const codexCliClause = opts.codexCli ? " `mcp__codex-cli__codex` dispatches to `codex-implementer` (gpt-5.3-codex with workspace-write) for end-to-end coding tasks." : "";
|
|
34742
35192
|
const para2Parts = [`\`mcp__${searchKey}__code\` is the one-stop code search (no extra model call). Its DEFAULT mode (or \`mode:"semantic"\`) ranks by MEANING via ColBERT over a per-workspace index, the first thing to reach for on intent/concept questions ("where is retry/backoff handled", "how does auth work"); when that index isn't ready it transparently falls back to lexical (the response \`source\` says which engine ran). Forced modes cover the rest: \`lexical\` (BM25F-ranked + tree-sitter, best for exact symbols), \`exact\`, \`regex\`, \`complete\` (exhaustive set), \`ast_pattern\`+\`ast_lang\` for multi-line AST shapes, \`scan\` for a whole-workspace symbol outline, \`multiline\` for cross-line regex. Multiple queries can run in a single turn. The index covers code-shaped files; for unstructured files (logs, \`.csv\`, \`.env*\`, config-only wiring), \`grep\`/\`glob\` still apply.`];
|
|
34743
35193
|
if (opts.workerToolsAvailable) para2Parts.push(`\`worker-*\` are background Agent subagents (subagent_type) that run the matching worker in its own context and deliver the result as a completion notification, so a long run never blocks the turn: \`worker-explore\` (read-only research), \`worker-review\` (reads the code to verify a change or claim), \`worker-plan\` (ordered implementation plan), \`worker-implement\` (edit/write/bash; ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file; for in-place edits use the \`implementer\` subagent), \`worker-test\` (independent test author; also always worktree-isolated)${opts.browseAvailable ? ", `worker-browse` (autonomous browser agent driving a real browser)" : ""}. The raw \`mcp__${workersKey}__*\` tools they call are guarded (a direct main-thread call is redirected to the matching agent); Workers themselves have \`code_search\`.`);
|
|
34744
35194
|
const catchAllClause = opts.generalPurposeFastAvailable === false ? "" : " Catch-all on a fast, economical non-lead model for work no specialist fits: `general-purpose-fast`.";
|
|
34745
|
-
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
35195
|
+
para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure)${opts.reviewerFastAvailable === false ? "" : ", `reviewer-fast` (lower-stakes assessment on a cheaper cross-lab model)"}, \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
|
|
34746
35196
|
if (opts.workerToolsAvailable) para2Parts.push(`For a bounded, well-scoped implementation, prefer the \`implementer\` subagent over \`worker-implement\`; reach for \`worker-implement\` only when you specifically need git-worktree isolation, parallel variants, or a throwaway experiment.`);
|
|
34747
35197
|
if (opts.workerToolsAvailable) para2Parts.push(`\`mcp__${orchestrateKey}__decompose\` composes an open-ended ask into a typed, VERIFIED workflow IR (a strong driver decorrelated by a cross-lab critic, so the decompose step isn't a single point of failure), and \`mcp__${orchestrateKey}__run_workflow\` executes that IR through a frozen kernel delivering max(orchestrated, baseline) over a sealed executable gate, so it never ships worse than a plain single-model run. \`mcp__${orchestrateKey}__verify_workflow\` checks an IR's floor invariants before you run it, and \`mcp__${orchestrateKey}__attest_step\` audits that a finished run's producers were each checked by a different lab. They suit non-trivial, role-separated asks; a trivial ask does not need them.`);
|
|
34748
35198
|
else para2Parts.push(`\`mcp__${orchestrateKey}__verify_workflow\` statically checks a workflow IR's floor invariants and \`mcp__${orchestrateKey}__attest_step\` audits a run's cross-lab lineage (the \`decompose\`/\`run_workflow\` composer + kernel need the worker backend, unavailable here).`);
|
|
@@ -34787,6 +35237,7 @@ function buildPeerAwarenessSummary(opts) {
|
|
|
34787
35237
|
renderNative("implementer"),
|
|
34788
35238
|
opts.implementerFastAvailable === false ? void 0 : renderNative("implementer-fast"),
|
|
34789
35239
|
renderNative("reviewer"),
|
|
35240
|
+
opts.reviewerFastAvailable === false ? void 0 : renderNative("reviewer-fast"),
|
|
34790
35241
|
renderNative("brainstorm"),
|
|
34791
35242
|
opts.scoutAvailable === false ? void 0 : renderNative("scout"),
|
|
34792
35243
|
renderNative("scribe"),
|
|
@@ -34807,11 +35258,38 @@ function buildPeerAwarenessSummary(opts) {
|
|
|
34807
35258
|
lines.push(`Each tool's own description carries when to use it and when not. The full per-tool inventory (models, gating, workers, skills) is in the "Peer review and advisor" section of your CLAUDE.md project instructions.`);
|
|
34808
35259
|
return lines.join("\n");
|
|
34809
35260
|
}
|
|
35261
|
+
/**
|
|
35262
|
+
* Applies the resolved Gemini review model to a persona requiring the Gemini
|
|
35263
|
+
* catalog: swaps `.model` and rewrites every literal occurrence of the
|
|
35264
|
+
* default id in `.description` so the two never disagree about which model
|
|
35265
|
+
* actually backs the tool. A mismatch here is user-visible — `tools/list`
|
|
35266
|
+
* would advertise "backed by gemini-3.1-pro-preview" while dispatch actually
|
|
35267
|
+
* ran the flash fallback — so every call site that resolves a
|
|
35268
|
+
* `requiresGeminiCatalog` persona MUST go through this helper rather than
|
|
35269
|
+
* setting `.model` directly (a prior draft of this fix did exactly that in
|
|
35270
|
+
* `routes/mcp/handler.ts`'s `activePersonas()` and left the description
|
|
35271
|
+
* stale). Relies on every `requiresGeminiCatalog` persona's description
|
|
35272
|
+
* literally containing `GEMINI_REVIEW_DEFAULT_MODEL`'s exact string — true
|
|
35273
|
+
* for both current entries (gemini-critic, gemini-reviewer); keep it true for
|
|
35274
|
+
* any future one, or `replaceAll` silently no-ops.
|
|
35275
|
+
*/
|
|
35276
|
+
function resolveGeminiPersona(p, geminiModel) {
|
|
35277
|
+
const model = geminiModel ?? "gemini-3.1-pro-preview";
|
|
35278
|
+
return {
|
|
35279
|
+
...p,
|
|
35280
|
+
model,
|
|
35281
|
+
description: p.description.replaceAll(GEMINI_REVIEW_DEFAULT_MODEL, model)
|
|
35282
|
+
};
|
|
35283
|
+
}
|
|
34810
35284
|
/** Convenience: every persona that should be registered for the given mode. */
|
|
34811
35285
|
function personasFor(opts) {
|
|
34812
35286
|
const result = [];
|
|
34813
35287
|
for (const p of PERSONAS_READ) {
|
|
34814
|
-
if (p.requiresGeminiCatalog
|
|
35288
|
+
if (p.requiresGeminiCatalog) {
|
|
35289
|
+
if (!opts.geminiAvailable) continue;
|
|
35290
|
+
result.push(resolveGeminiPersona(p, opts.geminiModel));
|
|
35291
|
+
continue;
|
|
35292
|
+
}
|
|
34815
35293
|
result.push(p);
|
|
34816
35294
|
}
|
|
34817
35295
|
if (opts.codexCli) for (const p of PERSONAS_WRITE) result.push(p);
|
|
@@ -35997,6 +36475,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
|
|
|
35997
36475
|
return [...new Set(names)];
|
|
35998
36476
|
}
|
|
35999
36477
|
//#endregion
|
|
36000
|
-
export { handleMcpDelete as $,
|
|
36478
|
+
export { handleMcpDelete as $, UPSTREAM_INACTIVITY_TIMEOUT_MS as $t, satisfiesMinVersion as A, readResponseBodyCapped as At, rememberThinkingHistoryRepair as B, provisionTreeSitterAssets as Bt, availableToolCommands as C, getTokenCount as Ct, vscodeRipgrepPath as D, createResponses as Dt, toolbeltSkipSet as E, resolveMcpToolTimeoutMs as Et, injectAdvisorTool as F, colbertDegradedWarning as Ft, isControllerClosedError as G, toolbeltPathOverride as Gt, repairRejectedThinkingHistory as H, DEFINITION_OF_GREATNESS as Ht, isAdvisorRequested as I, provisionAndIndexColbert as It, relayAnthropicStream as J, DEFAULT_CLAUDE_MODEL_FALLBACKS as Jt, logStreamError as K, BUDGET_SMALL_FAST_CATALOG_ID as Kt, resolveAdvisorEffort as L, extractTarGzMember as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, normalizeOpenAIUsage as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, provisionBrowserAssets as Nt, TOOLBELT_TOOLS$1 as O, createChatCompletions as Ot, buildAdvisorStream as P, hasSupportedBrowserInstalled as Pt, clampEffort as Q, UPSTREAM_FETCH_TIMEOUT_MS as Qt, resolveAdvisorModel as R, extractZipMember as Rt, buildEnv as S, createMessages as St, toolbeltEnabled as T, warnOnTokenPriceDrift as Tt, buildAnthropicErrorEvent as U, shouldUseInsecureTls as Ut, repairKnownThinkingHistory as V, CONDENSED_OPERATING_SEQUENCE as Vt, buildOpenAIErrorEvent as W, collapsePathKeys as Wt, UNKNOWN_EFFORT_ANCHOR as X, DEFAULT_CODEX_MODEL_FALLBACKS as Xt, EFFORT_ORDER as Y, DEFAULT_CODEX_MODEL as Yt, bucketEffort as Z, DEFAULT_PORT as Zt, appendPlanReminder as _, scribeModel as _t, buildPeerAwarenessSnippet as a, upstreamMaxConnections as an, browseAgentEnabled as at, resolveWorkerRunOpts as b, shimDefaultsToXhigh as bt, personasFor as c, withOneMSuffixForLead as cn, fleetToolsEnabled as ct, EXPLORE_DEFAULT_MODEL as d, implementerFastModel as dt, generateRandomPort as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, nativeSubagentModel as ft, TEST_DEFAULT_MODEL as g, scoutModel as gt, REVIEW_DEFAULT_MODEL as h, reviewerModel as ht, buildAgentPrompt as i, upstreamAllowH2 as in, brainstormModel as it, searchWeb as j, parseJsonOrDiagnose as jt, assetFor as k, MAX_RESPONSE_BODY_BYTES as kt, BROWSE_DEFAULT_MODEL as l, withInstallLock as ln, geminiAvailable as lt, PLAN_DEFAULT_MODEL as m, reviewerFastModel as mt, MCP_GROUPS as n, pickClaudeDefault as nn, agentToolsEnabled as nt, buildPeerAwarenessSummary as o, classifyMessagesRoute as on, browserCompoundToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, resolveGeminiReviewModel as pt, readIteratorWithTimeout as q, BUDGET_SMALL_FAST_SLUG as qt, assertMcpToolSurfaceConsistent as r, resolveLeadSlugArg as rn, artifactToolsEnabled as rt, enumerateInjectedMcpToolNames as s, withOneMSuffix as sn, browserToolsEnabled as st, GROUP_META as t, isBudgetClaudeLead as tn, REVIEW_FAST_DEFAULT_MODEL as tt, DEFAULT_MODEL_CHAIN as u, generalPurposeFastModel as ut, resolveDefaultModel as v, standInToolEnabled as vt, buildToolbeltAwareness as w, assembleResponsesPayload as wt, runWorkerAgent as x, countTokens as xt, resolveModeDefaults as y, workerToolsEnabled as yt, formatThinkingRepairDecline as z, warmTreeSitterPool as zt };
|
|
36001
36479
|
|
|
36002
|
-
//# sourceMappingURL=peer-mcp-personas-
|
|
36480
|
+
//# sourceMappingURL=peer-mcp-personas-Bd56EmiO.js.map
|