github-router 0.3.285 → 0.3.289

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/README.md +8 -5
  2. package/dist/{attribution-settings-B8M2fvhz.js → attribution-settings-Cmz2jt7P.js} +44 -17
  3. package/dist/attribution-settings-Cmz2jt7P.js.map +1 -0
  4. package/dist/{auth-VUL2Zxvw.js → auth-DG4vh8-F.js} +3 -3
  5. package/dist/{auth-VUL2Zxvw.js.map → auth-DG4vh8-F.js.map} +1 -1
  6. package/dist/browser-ext/manifest.json +1 -1
  7. package/dist/{check-usage-CcLdFPGr.js → check-usage-BqN7mBYv.js} +4 -4
  8. package/dist/{check-usage-CcLdFPGr.js.map → check-usage-BqN7mBYv.js.map} +1 -1
  9. package/dist/{claude-e-JIQGhR.js → claude-C-9xFI4b.js} +23 -17
  10. package/dist/claude-C-9xFI4b.js.map +1 -0
  11. package/dist/{codex-DSnVZ9oK.js → codex-DRW0yb8x.js} +5 -5
  12. package/dist/{codex-DSnVZ9oK.js.map → codex-DRW0yb8x.js.map} +1 -1
  13. package/dist/{debug-CtzYWxpJ.js → debug-B5TjPTTH.js} +2 -2
  14. package/dist/{debug-CtzYWxpJ.js.map → debug-B5TjPTTH.js.map} +1 -1
  15. package/dist/engine-iEqGdx6T.js +2 -0
  16. package/dist/{gate-discovery-BCFwLm0q.js → gate-discovery-Cz6kwIVG.js} +5 -5
  17. package/dist/{gate-discovery-BCFwLm0q.js.map → gate-discovery-Cz6kwIVG.js.map} +1 -1
  18. package/dist/{get-copilot-usage-B-EDAqQb.js → get-copilot-usage-BjA0nyGR.js} +2 -2
  19. package/dist/{get-copilot-usage-B-EDAqQb.js.map → get-copilot-usage-BjA0nyGR.js.map} +1 -1
  20. package/dist/{internal-artifact-open-Dj5Nlq0L.js → internal-artifact-open-BskmUpnb.js} +2 -2
  21. package/dist/{internal-artifact-open-Dj5Nlq0L.js.map → internal-artifact-open-BskmUpnb.js.map} +1 -1
  22. package/dist/{internal-first-mate-guard-DOgFVki5.js → internal-first-mate-guard-5XEHMaqy.js} +3 -3
  23. package/dist/{internal-first-mate-guard-DOgFVki5.js.map → internal-first-mate-guard-5XEHMaqy.js.map} +1 -1
  24. package/dist/{internal-first-mate-guard-B7p4NttK.js → internal-first-mate-guard-DHDQ6hFz.js} +1 -1
  25. package/dist/{internal-plan-review-DRFHET_B.js → internal-plan-review-BUuMx4ku.js} +3 -3
  26. package/dist/{internal-plan-review-DRFHET_B.js.map → internal-plan-review-BUuMx4ku.js.map} +1 -1
  27. package/dist/{internal-prompt-submit-f1LR2P2G.js → internal-prompt-submit-CQQ15xdO.js} +4 -4
  28. package/dist/{internal-prompt-submit-f1LR2P2G.js.map → internal-prompt-submit-CQQ15xdO.js.map} +1 -1
  29. package/dist/{internal-session-bind-DhZhxJ3T.js → internal-session-bind-D04W2yWI.js} +2 -2
  30. package/dist/{internal-session-bind-DhZhxJ3T.js.map → internal-session-bind-D04W2yWI.js.map} +1 -1
  31. package/dist/{internal-stop-hook-Cf-7w3RH.js → internal-stop-hook-Dvkppwo7.js} +5 -5
  32. package/dist/{internal-stop-hook-Cf-7w3RH.js.map → internal-stop-hook-Dvkppwo7.js.map} +1 -1
  33. package/dist/{internal-stop-review-BBsLcbPG.js → internal-stop-review-CdByyJLc.js} +2 -2
  34. package/dist/{internal-stop-review-BBsLcbPG.js.map → internal-stop-review-CdByyJLc.js.map} +1 -1
  35. package/dist/{internal-worker-guard-CKgYFiFO.js → internal-worker-guard-BIPN6Rv9.js} +2 -2
  36. package/dist/{internal-worker-guard-CKgYFiFO.js.map → internal-worker-guard-BIPN6Rv9.js.map} +1 -1
  37. package/dist/{internal-workspace-header-OYgHEnFt.js → internal-workspace-header-BKqejstG.js} +2 -2
  38. package/dist/{internal-workspace-header-OYgHEnFt.js.map → internal-workspace-header-BKqejstG.js.map} +1 -1
  39. package/dist/lifecycle-BTodQvn4.js +2 -0
  40. package/dist/lifecycle-C7JYNz-F.js +2 -0
  41. package/dist/{lifecycle-B7CHqKlF.js → lifecycle-DbM29FLK.js} +2 -2
  42. package/dist/{lifecycle-B7CHqKlF.js.map → lifecycle-DbM29FLK.js.map} +1 -1
  43. package/dist/{lifecycle-D-rL81tT.js → lifecycle-SXaWssN9.js} +2 -2
  44. package/dist/{lifecycle-D-rL81tT.js.map → lifecycle-SXaWssN9.js.map} +1 -1
  45. package/dist/main.js +17 -17
  46. package/dist/{mcp-workspace-header-ucs2SDST.js → mcp-workspace-header-DRCCWlOi.js} +2 -2
  47. package/dist/{mcp-workspace-header-ucs2SDST.js.map → mcp-workspace-header-DRCCWlOi.js.map} +1 -1
  48. package/dist/{models-C16mBK2M.js → models-Dz8d_SnI.js} +3 -3
  49. package/dist/{models-C16mBK2M.js.map → models-Dz8d_SnI.js.map} +1 -1
  50. package/dist/{orchestration-CtM6FYNx.js → orchestration-BrJwZxMN.js} +2 -2
  51. package/dist/{orchestration-CtM6FYNx.js.map → orchestration-BrJwZxMN.js.map} +1 -1
  52. package/dist/paths-CTr59UC6.js +2 -0
  53. package/dist/{paths-wLC0InjX.js → paths-D7_SAaIQ.js} +4 -4
  54. package/dist/{paths-wLC0InjX.js.map → paths-D7_SAaIQ.js.map} +1 -1
  55. package/dist/{peer-mcp-personas-V6stFvpq.js → peer-mcp-personas-Bd56EmiO.js} +533 -55
  56. package/dist/peer-mcp-personas-Bd56EmiO.js.map +1 -0
  57. package/dist/{plan-review-hook-Lf9ISdF6.js → plan-review-hook-CVZsG9MZ.js} +3 -3
  58. package/dist/{plan-review-hook-Lf9ISdF6.js.map → plan-review-hook-CVZsG9MZ.js.map} +1 -1
  59. package/dist/{prompt-submit-hook-BlijaOn7.js → prompt-submit-hook-BW92FX2D.js} +3 -3
  60. package/dist/{prompt-submit-hook-BlijaOn7.js.map → prompt-submit-hook-BW92FX2D.js.map} +1 -1
  61. package/dist/{provision-BpL6gZIt.js → provision-B53wbHwa.js} +4 -4
  62. package/dist/{provision-BpL6gZIt.js.map → provision-B53wbHwa.js.map} +1 -1
  63. package/dist/{self-invocation-CKMjcA5F.js → self-invocation-CP_SOkrr.js} +2 -2
  64. package/dist/{self-invocation-CKMjcA5F.js.map → self-invocation-CP_SOkrr.js.map} +1 -1
  65. package/dist/{serve-CXUf7RtJ.js → serve-aZCEYFe5.js} +37 -24
  66. package/dist/serve-aZCEYFe5.js.map +1 -0
  67. package/dist/{server-setup-Bppjt9xO.js → server-setup-DlztZAGT.js} +258 -99
  68. package/dist/server-setup-DlztZAGT.js.map +1 -0
  69. package/dist/{start-C1-jrHfU.js → start-DwNiXv5N.js} +3 -3
  70. package/dist/{start-C1-jrHfU.js.map → start-DwNiXv5N.js.map} +1 -1
  71. package/dist/{stop-gate-hook-DgJ6sW8N.js → stop-gate-hook-DriRc9xN.js} +3 -3
  72. package/dist/{stop-gate-hook-DgJ6sW8N.js.map → stop-gate-hook-DriRc9xN.js.map} +1 -1
  73. package/dist/{stop-gate-policy-DG5yYWGn.js → stop-gate-policy-DMPanpoR.js} +2 -2
  74. package/dist/{stop-gate-policy-DG5yYWGn.js.map → stop-gate-policy-DMPanpoR.js.map} +1 -1
  75. package/dist/{token-CnlB0884.js → token-BGCjZwtj.js} +2 -2
  76. package/dist/token-BGCjZwtj.js.map +1 -0
  77. package/dist/{worker-dispatch-zW8Zi69V.js → worker-dispatch-BCTMyNE-.js} +2 -2
  78. package/dist/{worker-dispatch-zW8Zi69V.js.map → worker-dispatch-BCTMyNE-.js.map} +1 -1
  79. package/package.json +2 -1
  80. package/dist/attribution-settings-B8M2fvhz.js.map +0 -1
  81. package/dist/claude-e-JIQGhR.js.map +0 -1
  82. package/dist/engine-DMveEa9x.js +0 -2
  83. package/dist/lifecycle-CbmHMMSD.js +0 -2
  84. package/dist/lifecycle-DTcZwa7U.js +0 -2
  85. package/dist/paths-CV9K7Xqm.js +0 -2
  86. package/dist/peer-mcp-personas-V6stFvpq.js.map +0 -1
  87. package/dist/serve-CXUf7RtJ.js.map +0 -1
  88. package/dist/server-setup-Bppjt9xO.js.map +0 -1
  89. package/dist/token-CnlB0884.js.map +0 -1
@@ -1,14 +1,14 @@
1
1
  import { n as explicitPackageRoot } from "./package-root-B-osctCk.js";
2
2
  import { t as getPackageVersion } from "./version-C8x2hQrZ.js";
3
- import { t as PATHS } from "./paths-wLC0InjX.js";
4
- import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-CnlB0884.js";
3
+ import { t as PATHS } from "./paths-D7_SAaIQ.js";
4
+ import { A as state, C as GITHUB_GRAPHQL_URL, D as githubAgentGraphQLHeaders, E as copilotVersion, O as githubAgentHeaders, S as GITHUB_API_BASE_URL, T as copilotHeaders, b as sanitizeTransportText, d as resolveModel, f as sleep$2, g as HTTPError, h as isAllowedCopilotHost, i as tryRefreshAndRetry, v as classifyTransportError, w as copilotBaseUrl, x as withTransientRetry, y as fetchWithTransientRetry } from "./token-BGCjZwtj.js";
5
5
  import { a as resolveExecutable, c as runManagedExeCapture, i as parseIntEnv, l as spawnTaskkillBestEffort, r as parseBoolEnv } from "./exec-y8C_MU8A.js";
6
- import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-B7CHqKlF.js";
7
- import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-ucs2SDST.js";
6
+ import { i as registerExitHandlers$1, n as getInstanceUuid, r as recordWorkerRepo, t as WorktreeRegistry } from "./lifecycle-DbM29FLK.js";
7
+ import { t as MCP_WORKSPACE_HEADER } from "./mcp-workspace-header-DRCCWlOi.js";
8
8
  import { n as ArtifactError, r as applyInsecureTls, t as ArtifactClient } from "./client-CQMfroGV.js";
9
- import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-D-rL81tT.js";
10
- import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DgJ6sW8N.js";
11
- import { t as liveExec } from "./orchestration-CtM6FYNx.js";
9
+ import { n as isPidAlive$1, r as registerColbertExitHandlers, s as trackChild, t as getColbertInstanceUuid } from "./lifecycle-SXaWssN9.js";
10
+ import { f as resolveSealedGate, m as runGateChecks, p as sealedGateIds } from "./stop-gate-hook-DriRc9xN.js";
11
+ import { t as liveExec } from "./orchestration-BrJwZxMN.js";
12
12
  import { createRequire } from "node:module";
13
13
  import consola from "consola";
14
14
  import { chmodSync, closeSync, constants, cpSync, existsSync, mkdirSync, openSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, statSync, unlinkSync, writeFileSync, writeSync } from "node:fs";
@@ -995,6 +995,9 @@ function enumProp$1(values, description) {
995
995
  };
996
996
  }
997
997
  //#endregion
998
+ //#region src/lib/gemini-review-model.ts
999
+ const GEMINI_REVIEW_DEFAULT_MODEL = "gemini-3.1-pro-preview";
1000
+ //#endregion
998
1001
  //#region src/lib/fleet/mesh-egress-agent.ts
999
1002
  /**
1000
1003
  * Runtime-aware "route this fetch through the mesh loopback egress proxy" for a
@@ -18914,7 +18917,7 @@ function logAudit$1(record) {
18914
18917
  try {
18915
18918
  const fs = await import("node:fs/promises");
18916
18919
  const path = await import("node:path");
18917
- const { PATHS } = await import("./paths-CV9K7Xqm.js");
18920
+ const { PATHS } = await import("./paths-CTr59UC6.js");
18918
18921
  const dir = path.join(PATHS.APP_DIR, "browser-mcp");
18919
18922
  await fs.mkdir(dir, { recursive: true });
18920
18923
  const line = JSON.stringify({
@@ -19494,6 +19497,354 @@ function currentInFlight() {
19494
19497
  return inFlight$2;
19495
19498
  }
19496
19499
  //#endregion
19500
+ //#region src/lib/prompt-cache.ts
19501
+ /**
19502
+ * Conservative eligibility floor, in UTF-8 BYTES — never `.length`, which
19503
+ * counts UTF-16 code units and undercounts anything outside the BMP (an
19504
+ * emoji is 2 code units but 4 bytes). This is a proxy for "the prefix is
19505
+ * obviously large enough that marking it as a cache breakpoint is worth one
19506
+ * of the scarce marker slots," and it deliberately does NOT claim to be a
19507
+ * token count: byte-to-token density varies by tokenizer and content — CJK
19508
+ * text carries MORE tokens per byte than ASCII prose (undercounting risk is
19509
+ * the SAFE direction: we'd skip a marker that might have qualified), while a
19510
+ * long run of a repeated character or repeated whitespace carries FEWER
19511
+ * tokens per byte than either, since BPE merges long runs into very few
19512
+ * tokens (overcounting risk: a byte count clearing the floor doesn't
19513
+ * guarantee the real token count clears Anthropic's or Copilot's per-model
19514
+ * minimum). No fixed byte threshold can bound that adversarial case; this
19515
+ * value is chosen so ordinary Claude Code system prompts and tool schemas
19516
+ * (natural-language / JSON, not deliberately repetitive) reliably qualify,
19517
+ * while genuinely small prefixes never burn a marker for no benefit.
19518
+ */
19519
+ const MIN_CACHEABLE_PREFIX_BYTES = 4096;
19520
+ const CACHE_KEY_NAMESPACE = "ghr-cache-v1";
19521
+ const CACHE_DIAGNOSTIC_LIMIT = 128;
19522
+ const GPT56_EXPLICIT_CACHE_MODELS = /* @__PURE__ */ new Set([
19523
+ "gpt-5.6-sol",
19524
+ "gpt-5.6-terra",
19525
+ "gpt-5.6-luna"
19526
+ ]);
19527
+ const priorSignatures = /* @__PURE__ */ new Map();
19528
+ function nonNegativeInt(value) {
19529
+ if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return 0;
19530
+ return Math.floor(value);
19531
+ }
19532
+ /**
19533
+ * Pick the first genuinely POSITIVE numeric candidate from an ordered,
19534
+ * priority-ranked list of usage-shape fields. `??`-chaining these fields is
19535
+ * wrong: it stops at the first field that is merely PRESENT, and a provider
19536
+ * surface that always populates a nested detail object with `0` as a
19537
+ * placeholder (while the real, positive count is reported only in a
19538
+ * lower-priority field, e.g. the top-level one) would have its explicit zero
19539
+ * silently shadow that populated count. Falling through zeros to find a real
19540
+ * positive value fixes that; when every candidate is zero, absent, or
19541
+ * non-numeric, this returns `0` — a genuine all-zero reading, never
19542
+ * `undefined` — so downstream `nonNegativeInt` always has a countable value.
19543
+ */
19544
+ function firstPositive(...candidates) {
19545
+ for (const c of candidates) if (typeof c === "number" && Number.isFinite(c) && c > 0) return c;
19546
+ return 0;
19547
+ }
19548
+ function usageDetails(usage) {
19549
+ const input = usage.input_tokens_details ?? {};
19550
+ const prompt = usage.prompt_tokens_details ?? {};
19551
+ return {
19552
+ cached_tokens: firstPositive(input.cached_tokens, prompt.cached_tokens, usage.cache_read_input_tokens),
19553
+ cache_write_tokens: firstPositive(input.cache_write_tokens, input.cache_creation_tokens, prompt.cache_write_tokens, prompt.cache_creation_tokens, usage.cache_write_tokens, usage.cache_creation_input_tokens),
19554
+ cache_ttl_seconds: firstPositive(input.cache_ttl_seconds, prompt.cache_ttl_seconds, usage.cache_ttl_seconds)
19555
+ };
19556
+ }
19557
+ /**
19558
+ * OpenAI totals INCLUDE cached and cache-write tokens. Normalize them into
19559
+ * mutually exclusive buckets so Anthropic and Pi consumers do not count the
19560
+ * same input twice.
19561
+ */
19562
+ function normalizeOpenAIUsage(usage) {
19563
+ if (!usage) return {
19564
+ totalInput: 0,
19565
+ uncachedInput: 0,
19566
+ output: 0,
19567
+ cacheRead: 0,
19568
+ cacheWrite: 0,
19569
+ totalTokens: 0
19570
+ };
19571
+ const totalInput = nonNegativeInt(usage.input_tokens ?? usage.prompt_tokens);
19572
+ const output = nonNegativeInt(usage.output_tokens ?? usage.completion_tokens);
19573
+ const details = usageDetails(usage);
19574
+ const cacheRead = Math.min(totalInput, nonNegativeInt(details.cached_tokens));
19575
+ const remaining = Math.max(0, totalInput - cacheRead);
19576
+ const cacheWrite = Math.min(remaining, nonNegativeInt(details.cache_write_tokens ?? details.cache_creation_tokens));
19577
+ const uncachedInput = Math.max(0, totalInput - cacheRead - cacheWrite);
19578
+ const reportedTotal = nonNegativeInt(usage.total_tokens);
19579
+ const cacheTtlSeconds = nonNegativeInt(details.cache_ttl_seconds);
19580
+ return {
19581
+ totalInput,
19582
+ uncachedInput,
19583
+ output,
19584
+ cacheRead,
19585
+ cacheWrite,
19586
+ totalTokens: Math.max(reportedTotal, totalInput + output),
19587
+ ...cacheTtlSeconds > 0 ? { cacheTtlSeconds } : {}
19588
+ };
19589
+ }
19590
+ function hash(value) {
19591
+ return createHash("sha256").update(value).digest("hex");
19592
+ }
19593
+ function signatureFor(value) {
19594
+ return hash(typeof value === "string" ? value : JSON.stringify(value ?? null));
19595
+ }
19596
+ function serializedBytes(value) {
19597
+ return Buffer.byteLength(typeof value === "string" ? value : JSON.stringify(value ?? null));
19598
+ }
19599
+ function logCacheSignature(args) {
19600
+ if (parseBoolEnv(process.env.GH_ROUTER_LOG_CACHE) !== true) return;
19601
+ const key = `${args.endpoint}:${args.model}:${args.workload}`;
19602
+ const current = {
19603
+ system: signatureFor(args.system),
19604
+ tools: signatureFor(args.tools),
19605
+ messages: signatureFor(args.messages)
19606
+ };
19607
+ const previous = priorSignatures.get(key);
19608
+ let changed = "cold";
19609
+ if (previous) changed = previous.system !== current.system ? "system" : previous.tools !== current.tools ? "tools" : previous.messages !== current.messages ? "messages" : "none";
19610
+ priorSignatures.set(key, current);
19611
+ if (priorSignatures.size > CACHE_DIAGNOSTIC_LIMIT) {
19612
+ const oldest = priorSignatures.keys().next().value;
19613
+ if (oldest !== void 0) priorSignatures.delete(oldest);
19614
+ }
19615
+ consola.info(`cache-signature endpoint=${args.endpoint} model=${args.model} workload=${args.workload} changed=${changed} system_bytes=${serializedBytes(args.system ?? "")} tools_bytes=${serializedBytes(args.tools ?? [])} messages_bytes=${serializedBytes(args.messages ?? [])}`);
19616
+ }
19617
+ function hasResponsesBreakpoint(input) {
19618
+ const visit = (value) => {
19619
+ if (!value || typeof value !== "object") return false;
19620
+ if (Array.isArray(value)) return value.some(visit);
19621
+ const record = value;
19622
+ return record.prompt_cache_breakpoint !== void 0 || Object.values(record).some(visit);
19623
+ };
19624
+ return visit(input);
19625
+ }
19626
+ function gpt56ExplicitCacheEnabled(model) {
19627
+ if (parseBoolEnv(process.env.GH_ROUTER_DISABLE_GPT56_EXPLICIT_CACHE) === true) return false;
19628
+ return GPT56_EXPLICIT_CACHE_MODELS.has(model);
19629
+ }
19630
+ function responsesCacheKey(payload, opts, stablePrefix) {
19631
+ const digest = hash(JSON.stringify({
19632
+ namespace: CACHE_KEY_NAMESPACE,
19633
+ model: payload.model,
19634
+ workload: opts.workload,
19635
+ scope: opts.scope ?? "",
19636
+ stablePrefix,
19637
+ tools: payload.tools ?? []
19638
+ }));
19639
+ return `${CACHE_KEY_NAMESPACE}-${digest.slice(0, 48)}`;
19640
+ }
19641
+ /**
19642
+ * Add GPT-5.6 explicit caching only to router-owned REUSABLE-PREFIX payloads.
19643
+ * Public passthrough routes never call this helper, and existing caller
19644
+ * fields always win. Live shape acceptance is pinned by compatibility probe
19645
+ * `gpt56_explicit_cache_breakpoint`.
19646
+ *
19647
+ * **`"conversation"` is deliberately EXCLUDED and left untouched (a no-op),
19648
+ * same as `"passthrough"`/`"one-shot"`.** A live-verified regression: on a
19649
+ * growing multi-turn conversation (Claude Code's translated main loop and
19650
+ * the worker-agent loop, both of which pass `workload: "conversation"`),
19651
+ * marking only the stable SYSTEM block with an explicit breakpoint measured
19652
+ * substantially worse than leaving caching provider-managed and implicit.
19653
+ * Explicit mode is a distinct
19654
+ * caching strategy from Copilot's provider-managed automatic caching, not an
19655
+ * addition to it — turning it on for a request marks only the bytes an
19656
+ * explicit breakpoint names, and the REST of that request's prefix (here,
19657
+ * the entire un-marked growing message history) stops receiving automatic
19658
+ * prefix-growth caching too. Measured on `gpt-5.6-sol` with explicit mode
19659
+ * force-enabled for conversation workloads: turn 1 (cold)
19660
+ * `input_tokens=27038, cache_write=2031, cache_read=0`; turn 2
19661
+ * `input_tokens=27054, cache_read=2031`; turn 3 `input_tokens=27071,
19662
+ * cache_read=2031` — the ~2k-token system block cached once and never grew,
19663
+ * while the other ~25k tokens of accumulating history were recomputed from
19664
+ * scratch on every single turn. `"reusable-prefix"` calls (peer/advisor/
19665
+ * worker-tool/browser-compressor prefixes reused verbatim across many
19666
+ * DISCRETE calls, never a single request whose own history keeps growing)
19667
+ * do not have this failure mode and keep the explicit treatment below.
19668
+ */
19669
+ function applyResponsesCachePolicy(payload, opts) {
19670
+ logCacheSignature({
19671
+ endpoint: "/responses",
19672
+ model: payload.model,
19673
+ workload: opts.workload,
19674
+ system: opts.stablePrefix ?? payload.instructions,
19675
+ tools: payload.tools,
19676
+ messages: payload.input
19677
+ });
19678
+ if (opts.workload !== "reusable-prefix" || !gpt56ExplicitCacheEnabled(payload.model) || payload.prompt_cache_key !== void 0 || payload.prompt_cache_options !== void 0 || hasResponsesBreakpoint(payload.input)) return payload;
19679
+ const stablePrefix = opts.stablePrefix ?? payload.instructions;
19680
+ const stableBytes = serializedBytes(stablePrefix ?? "") + serializedBytes(payload.tools ?? []);
19681
+ if (!stablePrefix || stableBytes < MIN_CACHEABLE_PREFIX_BYTES) return payload;
19682
+ const input = typeof payload.input === "string" ? [{
19683
+ role: "user",
19684
+ content: payload.input
19685
+ }] : [...payload.input];
19686
+ const stableContent = [{
19687
+ type: "input_text",
19688
+ text: stablePrefix,
19689
+ prompt_cache_breakpoint: { mode: "explicit" }
19690
+ }];
19691
+ let nextInput;
19692
+ let removeInstructions = false;
19693
+ if (payload.instructions === stablePrefix) {
19694
+ nextInput = [{
19695
+ role: "system",
19696
+ content: stableContent
19697
+ }, ...input];
19698
+ removeInstructions = true;
19699
+ } else {
19700
+ const stableSystemIndex = input.findIndex((item) => item.role === "system" && item.content === stablePrefix);
19701
+ if (stableSystemIndex < 0) return payload;
19702
+ nextInput = [...input];
19703
+ nextInput[stableSystemIndex] = {
19704
+ ...nextInput[stableSystemIndex],
19705
+ content: stableContent
19706
+ };
19707
+ }
19708
+ const next = {
19709
+ ...payload,
19710
+ input: nextInput,
19711
+ prompt_cache_key: responsesCacheKey(payload, opts, stablePrefix),
19712
+ prompt_cache_options: {
19713
+ mode: "explicit",
19714
+ ttl: "30m"
19715
+ }
19716
+ };
19717
+ if (removeInstructions) delete next.instructions;
19718
+ return next;
19719
+ }
19720
+ function itemHasCacheControl(value) {
19721
+ return !!value && typeof value === "object" && value.cache_control !== void 0;
19722
+ }
19723
+ function hasClaudeCacheControl(body) {
19724
+ if (itemHasCacheControl(body.system)) return true;
19725
+ if (Array.isArray(body.system) && body.system.some(itemHasCacheControl)) return true;
19726
+ if (Array.isArray(body.tools) && body.tools.some(itemHasCacheControl)) return true;
19727
+ if (!Array.isArray(body.messages)) return false;
19728
+ return body.messages.some((message) => {
19729
+ if (!message || typeof message !== "object") return false;
19730
+ const content = message.content;
19731
+ return itemHasCacheControl(content) || Array.isArray(content) && content.some(itemHasCacheControl);
19732
+ });
19733
+ }
19734
+ /**
19735
+ * UTF-8 byte length of the tools array alone — the eligibility floor for the
19736
+ * TOOL breakpoint. Marked on the last non-deferred tool, it caches only the
19737
+ * tools prefix (Claude's wire order is tools, then system, then messages), so
19738
+ * its own size — not the combined system+tools size — is what determines
19739
+ * whether that marker is worth spending. See `MIN_CACHEABLE_PREFIX_BYTES`.
19740
+ */
19741
+ function claudeToolsPrefixBytes(body) {
19742
+ return serializedBytes(body.tools ?? []);
19743
+ }
19744
+ /**
19745
+ * UTF-8 byte length of tools + system combined — the eligibility floor for
19746
+ * the SYSTEM breakpoint. Marked on the last system text block, it caches
19747
+ * everything up to and including system (tools THEN system in wire order),
19748
+ * so the combined size is the right measure — checked SEPARATELY from the
19749
+ * tools-only floor above so a large system prompt behind tiny tools doesn't
19750
+ * smuggle a useless tools-only marker in under the combined total, and a
19751
+ * large tools array behind an empty system doesn't get double-counted as
19752
+ * "small" just because system alone is tiny.
19753
+ */
19754
+ function claudeSystemPrefixBytes(body) {
19755
+ return claudeToolsPrefixBytes(body) + serializedBytes(body.system ?? "");
19756
+ }
19757
+ function markClaudeSystem(body) {
19758
+ if (typeof body.system === "string" && body.system.length > 0) {
19759
+ body.system = [{
19760
+ type: "text",
19761
+ text: body.system,
19762
+ cache_control: { type: "ephemeral" }
19763
+ }];
19764
+ return true;
19765
+ }
19766
+ if (!Array.isArray(body.system)) return false;
19767
+ for (let index = body.system.length - 1; index >= 0; index--) {
19768
+ const block = body.system[index];
19769
+ if (block && typeof block === "object" && block.type === "text" && typeof block.text === "string") {
19770
+ body.system[index] = {
19771
+ ...block,
19772
+ cache_control: { type: "ephemeral" }
19773
+ };
19774
+ return true;
19775
+ }
19776
+ }
19777
+ return false;
19778
+ }
19779
+ function markClaudeTool(body) {
19780
+ if (!Array.isArray(body.tools)) return false;
19781
+ for (let index = body.tools.length - 1; index >= 0; index--) {
19782
+ const tool = body.tools[index];
19783
+ if (tool && typeof tool === "object" && tool.defer_loading !== true) {
19784
+ body.tools[index] = {
19785
+ ...tool,
19786
+ cache_control: { type: "ephemeral" }
19787
+ };
19788
+ return true;
19789
+ }
19790
+ }
19791
+ return false;
19792
+ }
19793
+ /**
19794
+ * Apply the bounded Claude anchor policy to router-generated Messages bodies.
19795
+ * Caller-owned marker layouts are returned byte-for-byte unchanged.
19796
+ *
19797
+ * Marks at most TWO breakpoints — the last non-deferred tool and the stable
19798
+ * system boundary — each gated on its OWN eligibility check
19799
+ * (`claudeToolsPrefixBytes` / `claudeSystemPrefixBytes`) rather than one
19800
+ * combined check, so a large system prompt behind tiny tools doesn't also
19801
+ * mark a tools breakpoint too small to be worth a marker slot, and vice
19802
+ * versa. Anthropic's own hard ceiling is FOUR `cache_control` blocks per
19803
+ * request (probe `cache_control_marker_limit_5`); this policy only ever
19804
+ * spends up to two of them (`hasClaudeCacheControl` already refuses to run
19805
+ * at all once the caller has marked anything itself, so the two never
19806
+ * combine with a caller-owned marker to approach that ceiling).
19807
+ *
19808
+ * There used to be a third, message-level marking path gated on
19809
+ * `opts.workload === "conversation"`. It was removed as dead code: every
19810
+ * production call site of this function (`src/routes/mcp/handler.ts`,
19811
+ * `src/services/advisor/advisor.ts`) passes `workload: "reusable-prefix"`,
19812
+ * so the per-message branch never ran outside its own unit test, which gave
19813
+ * false confidence that production traffic exercised it. `CacheWorkload`
19814
+ * keeps `"conversation"` as a shared enum value — the Responses-side policy,
19815
+ * which has no per-message logic of its own, still uses it — so passing it
19816
+ * here remains type-valid; it now behaves identically to
19817
+ * `"reusable-prefix"`.
19818
+ */
19819
+ function applyClaudeCachePolicy(rawBody, opts) {
19820
+ if (opts.workload === "passthrough" || opts.workload === "one-shot" || parseBoolEnv(process.env.GH_ROUTER_DISABLE_CLAUDE_CACHE_POLICY) === true) return rawBody;
19821
+ let parsed;
19822
+ try {
19823
+ parsed = JSON.parse(rawBody);
19824
+ } catch {
19825
+ return rawBody;
19826
+ }
19827
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return rawBody;
19828
+ const body = parsed;
19829
+ if (typeof body.model !== "string" || !body.model.startsWith("claude-") || hasClaudeCacheControl(body)) return rawBody;
19830
+ const toolsEligible = claudeToolsPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
19831
+ const systemEligible = claudeSystemPrefixBytes(body) >= MIN_CACHEABLE_PREFIX_BYTES;
19832
+ if (!toolsEligible && !systemEligible) return rawBody;
19833
+ let markers = 0;
19834
+ if (toolsEligible && markClaudeTool(body)) markers++;
19835
+ if (systemEligible && markClaudeSystem(body)) markers++;
19836
+ if (markers === 0) return rawBody;
19837
+ logCacheSignature({
19838
+ endpoint: "/messages",
19839
+ model: body.model,
19840
+ workload: opts.workload,
19841
+ system: body.system,
19842
+ tools: body.tools,
19843
+ messages: body.messages
19844
+ });
19845
+ return JSON.stringify(body);
19846
+ }
19847
+ //#endregion
19497
19848
  //#region src/lib/vision-preflight.ts
19498
19849
  /**
19499
19850
  * Outbound vision handling.
@@ -20418,7 +20769,7 @@ async function callViaChat(model, systemPrompt, userMessage, tool, signal) {
20418
20769
  * Image parts use `input_image` (vs chat's `image_url`) — see
20419
20770
  * `toResponsesContent`. */
20420
20771
  async function callViaResponses(model, systemPrompt, userMessage, tool, signal) {
20421
- const payload = {
20772
+ const payload = applyResponsesCachePolicy({
20422
20773
  model,
20423
20774
  stream: false,
20424
20775
  input: [{
@@ -20438,7 +20789,10 @@ async function callViaResponses(model, systemPrompt, userMessage, tool, signal)
20438
20789
  type: "function",
20439
20790
  name: tool.name
20440
20791
  }
20441
- };
20792
+ }, {
20793
+ workload: "reusable-prefix",
20794
+ stablePrefix: systemPrompt
20795
+ });
20442
20796
  const resp = await createResponses(payload, void 0, signal, true);
20443
20797
  const output = Array.isArray(resp.output) ? resp.output : [];
20444
20798
  for (const item of output) {
@@ -23916,6 +24270,10 @@ const FALLBACK_TOKEN_PRICES = Object.freeze({
23916
24270
  "gemini-3.1-pro-preview": {
23917
24271
  in: 200,
23918
24272
  out: 1200
24273
+ },
24274
+ "gemini-3.7-flash": {
24275
+ in: 75,
24276
+ out: 375
23919
24277
  }
23920
24278
  });
23921
24279
  /**
@@ -23976,6 +24334,22 @@ function catalogTokenPrices(modelId) {
23976
24334
  * takes ~0.9s. That figure is deliberately NOT surfaced to the model, because a
23977
24335
  * second speed axis invites optimising a routing choice that policy already
23978
24336
  * settles (see the decorrelation note below).
24337
+ *
24338
+ * `gemini-3.7-flash` and the re-measured `gemini-3.1-pro-preview` (2026-08-13,
24339
+ * n=5, one untimed warmup rep) used the script's newer `GH_ROUTER_BENCH_STREAM=1`
24340
+ * mode, which splits wall-clock into TTFT and a decode-phase rate excluding the
24341
+ * first delta's own time+tokens. This row still records the SAME metric family
24342
+ * as every other row here (total tokens / total wall-clock, i.e. the script's
24343
+ * "total tok/s" column) — NOT the new decode-phase figure — so the table stays
24344
+ * internally comparable across rows measured at different times. The decode
24345
+ * figure is materially different and worth knowing for routing decisions that
24346
+ * care about steady-state throughput specifically: at matched effort,
24347
+ * `gemini-3.7-flash` decodes at ~275-315 tok/s (Google's own ~340 tok/s
24348
+ * Artificial Analysis figure, roughly confirmed once TTFT is excluded) against
24349
+ * `gpt-5.6-terra`'s ~135-190 tok/s decode — gemini-3.7-flash is the faster
24350
+ * decoder, but its ~1.1-1.7s TTFT (vs terra's ~0.9-1.0s) drags its TOTAL rate
24351
+ * below terra's on short responses, which is exactly why this row uses total,
24352
+ * not decode, to stay consistent with its neighbors.
23979
24353
  */
23980
24354
  const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
23981
24355
  "gpt-5.6-luna": 120,
@@ -23988,9 +24362,10 @@ const INDICATIVE_TOKENS_PER_SECOND = Object.freeze({
23988
24362
  "gpt-5.6-sol": 75,
23989
24363
  "grok-4.5": 70,
23990
24364
  "gpt-5.5": 65,
24365
+ "gemini-3.7-flash": 91,
23991
24366
  "gemini-3.6-flash": 45,
23992
24367
  "gemini-3.5-flash": 40,
23993
- "gemini-3.1-pro-preview": 25
24368
+ "gemini-3.1-pro-preview": 33
23994
24369
  });
23995
24370
  /** Returns the approximate, indicative output speed when it was measured. */
23996
24371
  function indicativeTokensPerSecond(modelId) {
@@ -24467,6 +24842,10 @@ function neutralToolsToResponses(tools) {
24467
24842
  /** Assemble the full Responses payload from the neutral request shape. */
24468
24843
  function assembleResponsesPayload(opts) {
24469
24844
  const input = [];
24845
+ if (opts.dynamicInstructions) input.push({
24846
+ role: "system",
24847
+ content: opts.dynamicInstructions
24848
+ });
24470
24849
  for (const m of opts.messages) for (const item of neutralMessageToResponsesInput(m)) input.push(item);
24471
24850
  const payload = {
24472
24851
  model: opts.model,
@@ -24483,7 +24862,10 @@ function assembleResponsesPayload(opts) {
24483
24862
  if (typeof opts.maxOutputTokens === "number" && opts.maxOutputTokens > 0) payload.max_output_tokens = Math.max(opts.maxOutputTokens, RESPONSES_MIN_MAX_OUTPUT_TOKENS);
24484
24863
  if (opts.stopSequences && opts.stopSequences.length > 0) payload.stop = [...opts.stopSequences];
24485
24864
  if (opts.parallelToolCalls === false) payload.parallel_tool_calls = false;
24486
- return payload;
24865
+ return opts.cachePolicy ? applyResponsesCachePolicy(payload, {
24866
+ ...opts.cachePolicy,
24867
+ stablePrefix: opts.cachePolicy.stablePrefix ?? opts.instructions
24868
+ }) : payload;
24487
24869
  }
24488
24870
  //#endregion
24489
24871
  //#region src/lib/worker-agent/context-budget.ts
@@ -25018,7 +25400,10 @@ function mapResponsesUsage(u) {
25018
25400
  prompt_tokens: u.input_tokens ?? 0,
25019
25401
  completion_tokens: u.output_tokens ?? 0,
25020
25402
  total_tokens: u.total_tokens ?? 0,
25021
- prompt_tokens_details: u.input_tokens_details?.cached_tokens != null ? { cached_tokens: u.input_tokens_details.cached_tokens } : void 0
25403
+ prompt_tokens_details: u.input_tokens_details != null ? {
25404
+ cached_tokens: u.input_tokens_details.cached_tokens ?? 0,
25405
+ cache_write_tokens: u.input_tokens_details.cache_write_tokens ?? u.input_tokens_details.cache_creation_tokens ?? 0
25406
+ } : void 0
25022
25407
  };
25023
25408
  }
25024
25409
  /**
@@ -25267,6 +25652,7 @@ function buildResponsesPayload(context, resolved) {
25267
25652
  messages,
25268
25653
  tools: piToolsToNeutral(context.tools),
25269
25654
  reasoningEffort: resolved.thinking,
25655
+ cachePolicy: { workload: "conversation" },
25270
25656
  stream: true
25271
25657
  });
25272
25658
  }
@@ -25489,12 +25875,13 @@ function emptyUsage() {
25489
25875
  }
25490
25876
  function deriveUsage(u) {
25491
25877
  if (!u) return emptyUsage();
25878
+ const normalized = normalizeOpenAIUsage(u);
25492
25879
  return {
25493
- input: u.prompt_tokens ?? 0,
25494
- output: u.completion_tokens ?? 0,
25495
- cacheRead: u.prompt_tokens_details?.cached_tokens ?? 0,
25496
- cacheWrite: 0,
25497
- totalTokens: u.total_tokens ?? 0,
25880
+ input: normalized.uncachedInput,
25881
+ output: normalized.output,
25882
+ cacheRead: normalized.cacheRead,
25883
+ cacheWrite: normalized.cacheWrite,
25884
+ totalTokens: normalized.totalTokens,
25498
25885
  cost: {
25499
25886
  input: 0,
25500
25887
  output: 0,
@@ -27013,6 +27400,7 @@ function shimDefaultsToXhigh(id) {
27013
27400
  * live tool list would be a silent regression (the snippet would name
27014
27401
  * a tool the live catalog doesn't expose).
27015
27402
  */
27403
+ const REVIEW_FAST_DEFAULT_MODEL = "gemini-3.7-flash";
27016
27404
  /**
27017
27405
  * Gate for the `stand_in` tool.
27018
27406
  *
@@ -27021,9 +27409,8 @@ function shimDefaultsToXhigh(id) {
27021
27409
  * - an OpenAI frontier model (`gpt-5.6-sol`, else `gpt-5.5` — see
27022
27410
  * `resolveOpenAiFrontier`)
27023
27411
  * - `claude-opus-5` (stand_in's Anthropic slot)
27024
- * - any `gemini-3.X.*pro` (gemini_critic's model family — matches the
27025
- * same regex `geminiAvailable()` uses, so the gate stays in sync if
27026
- * the GA slug renames `gemini-3.1-pro-preview` → `gemini-3.1-pro`)
27412
+ * - the preferred Gemini reviewer model (`gemini-3.1-pro-preview`, falling
27413
+ * back to `gemini-3.7-flash` after the preview's 2026-09-01 removal)
27027
27414
  *
27028
27415
  * If any one is missing, `stand_in` is dropped from `tools/list` AND
27029
27416
  * fails `tools/call` with -32601 (mirroring the `worker` capability's
@@ -27032,10 +27419,47 @@ function shimDefaultsToXhigh(id) {
27032
27419
  * `claude-opus-5` is a single-segment slug (dotted == dashed), so the
27033
27420
  * catalog probe matches Copilot's actual id shape directly.
27034
27421
  */
27035
- function geminiAvailable(source = state) {
27422
+ /**
27423
+ * Any live-catalog model matching Google's `gemini-3.x-pro` family, excluding
27424
+ * the two known literals — catches a GA rename of the preview slug (e.g.
27425
+ * `gemini-3.1-pro-preview` -> `gemini-3.1-pro`) so a vendor rename doesn't
27426
+ * silently downgrade every Gemini-gated resolver to the flash fallback while a
27427
+ * real pro-tier successor is actually present in the catalog. This is the
27428
+ * same regex the removed `geminiAvailable()` used, for the same reason —
27429
+ * losing it here was a real regression caught in review, not a deliberate
27430
+ * simplification.
27431
+ */
27432
+ function findGeminiProGaRename(models) {
27433
+ return models.find((m) => /^gemini-3\..*pro/i.test(m.id) && m.id !== "gemini-3.1-pro-preview" && m.id !== "gemini-3.7-flash")?.id;
27434
+ }
27435
+ function resolveGeminiReviewModel(source = state) {
27036
27436
  const models = source.models?.data;
27037
- if (!models) return false;
27038
- return models.some((m) => /^gemini-3\..*pro/i.test(m.id));
27437
+ if (!models) return void 0;
27438
+ if (models.some((m) => m.id === "gemini-3.1-pro-preview")) return GEMINI_REVIEW_DEFAULT_MODEL;
27439
+ const gaRename = findGeminiProGaRename(models);
27440
+ if (gaRename) return gaRename;
27441
+ if (models.some((m) => m.id === "gemini-3.7-flash")) return REVIEW_FAST_DEFAULT_MODEL;
27442
+ }
27443
+ /**
27444
+ * Gemini review candidates in preference order, for resolvers that ALSO need
27445
+ * `firstPresentInCatalog`'s `requireToolCalls`/`minContextTokens` enforcement
27446
+ * (`resolveGeminiReviewModel()` only checks id presence, not those capability
27447
+ * flags). Mirrors `resolveGeminiReviewModel()`'s own preference order: the
27448
+ * known preview id, then a GA rename of it, then the flash fallback — kept as
27449
+ * a shared helper so `reviewerModel()`/`brainstormModel()` can't drift from
27450
+ * `resolveGeminiReviewModel()`'s GA-rename handling the way the hardcoded
27451
+ * per-resolver chains did before this was extracted.
27452
+ */
27453
+ function geminiReviewChainCandidates() {
27454
+ const gaRename = findGeminiProGaRename(state.models?.data ?? []);
27455
+ return [
27456
+ GEMINI_REVIEW_DEFAULT_MODEL,
27457
+ ...gaRename ? [gaRename] : [],
27458
+ REVIEW_FAST_DEFAULT_MODEL
27459
+ ];
27460
+ }
27461
+ function geminiAvailable(source = state) {
27462
+ return resolveGeminiReviewModel(source) != null;
27039
27463
  }
27040
27464
  /**
27041
27465
  * First id in `chain` that is present in the live catalog. With
@@ -27107,7 +27531,7 @@ function nativeSubagentModel() {
27107
27531
  * was one model checking its own output. Not merely the same lab: the same
27108
27532
  * model. Two independent blind audits flagged it, and the repo already applies
27109
27533
  * the opposite rule one layer down, where `worker-review` runs
27110
- * `REVIEW_DEFAULT_MODEL` precisely so the reviewer's lab is decorrelated from the
27534
+ * `GEMINI_REVIEW_DEFAULT_MODEL` precisely so the reviewer's lab is decorrelated from the
27111
27535
  * producer's.
27112
27536
  *
27113
27537
  * The Anthropic lead and the OpenAI-frontier `implementer` are the two producers
@@ -27115,7 +27539,7 @@ function nativeSubagentModel() {
27115
27539
  * frontier remains the fallback: a same-lab reviewer still beats no reviewer.
27116
27540
  */
27117
27541
  function reviewerModel() {
27118
- return firstPresentInCatalog([REVIEW_DEFAULT_MODEL, ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
27542
+ return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
27119
27543
  }
27120
27544
  /** Model for `brainstorm`. Absent → inherits the lead's model.
27121
27545
  *
@@ -27124,7 +27548,7 @@ function reviewerModel() {
27124
27548
  * `implementer`/`reviewer`, so a same-lab brainstormer would mostly restate
27125
27549
  * what the lead already thought of. */
27126
27550
  function brainstormModel() {
27127
- return firstPresentInCatalog([REVIEW_DEFAULT_MODEL, ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
27551
+ return firstPresentInCatalog([...geminiReviewChainCandidates(), ...OPENAI_FRONTIER_MODELS], { requireToolCalls: true });
27128
27552
  }
27129
27553
  /** Model for `scribe`. Absent → inherits the lead's model.
27130
27554
  *
@@ -27146,18 +27570,24 @@ function scribeModel() {
27146
27570
  * impostor wearing the cheap agent's name.
27147
27571
  *
27148
27572
  * `gpt-5.6-luna` leads because it is the cheapest 1M-context model in the
27149
- * catalog; `gemini-3.6-flash` remains the cross-vendor fallback so an OpenAI-side
27573
+ * catalog; `gemini-3.7-flash` remains the cross-vendor fallback so an OpenAI-side
27150
27574
  * outage does not remove the scout. Both entries must continue advertising at
27151
27575
  * least 1M context so Claude Code's `[1m]` accounting remains honest if an
27152
27576
  * upstream catalog entry shrinks.
27153
27577
  *
27578
+ * The fallback moved off `gemini-3.6-flash` on 2026-08-13: `gemini-3.7-flash`
27579
+ * is strictly better on every axis this chain cares about — half the price
27580
+ * (75/375 vs 150/750 per 1M), materially faster (measured tool-call p50 ~1.2s
27581
+ * against 3.6's ~2.6s), same 1M window, same vendor, so the cross-vendor
27582
+ * property the fallback exists for is preserved.
27583
+ *
27154
27584
  * This chain deliberately uses literal ids rather than `EXPLORE_DEFAULT_MODEL`:
27155
27585
  * the explore worker default and scout's cross-vendor fallback are independent
27156
27586
  * policies, so retuning one must not silently collapse the other. There is no
27157
27587
  * 400K last resort. On a catalog carrying neither chain member, `scout` is
27158
27588
  * dropped rather than inheriting the lead or presenting a narrower-context agent.
27159
27589
  */
27160
- const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.6-flash"]);
27590
+ const SCOUT_MODEL_CHAIN = Object.freeze(["gpt-5.6-luna", "gemini-3.7-flash"]);
27161
27591
  function scoutModel() {
27162
27592
  return firstPresentInCatalog(SCOUT_MODEL_CHAIN, {
27163
27593
  requireToolCalls: true,
@@ -27172,7 +27602,17 @@ function scoutModel() {
27172
27602
  * well-specified, mechanical changes at a lower tier. Both entries are 1M+;
27173
27603
  * their different speed and effort properties stay out of shared claims. */
27174
27604
  function implementerFastModel() {
27175
- return firstPresentInCatalog(["gpt-5.6-terra", REVIEW_DEFAULT_MODEL], {
27605
+ return firstPresentInCatalog(["gpt-5.6-terra", GEMINI_REVIEW_DEFAULT_MODEL], {
27606
+ requireToolCalls: true,
27607
+ minContextTokens: ONE_M_TOKENS
27608
+ });
27609
+ }
27610
+ /** Model for `reviewer-fast` — the cheaper Google review tier. Absent → the
27611
+ * agent is dropped. Single-entry by design: inheriting the lead or falling
27612
+ * across labs would defeat both its cost purpose and its decorrelation from
27613
+ * the OpenAI-backed implementer. */
27614
+ function reviewerFastModel() {
27615
+ return firstPresentInCatalog([REVIEW_FAST_DEFAULT_MODEL], {
27176
27616
  requireToolCalls: true,
27177
27617
  minContextTokens: ONE_M_TOKENS
27178
27618
  });
@@ -27473,6 +27913,7 @@ function resolveOpusCriticModel() {
27473
27913
  }
27474
27914
  function activePersonas() {
27475
27915
  return PERSONAS_READ.filter((p) => !p.requiresGeminiCatalog || geminiAvailable()).map((p) => {
27916
+ if (p.requiresGeminiCatalog) return resolveGeminiPersona(p, resolveGeminiReviewModel());
27476
27917
  if (p.toolNameHttp !== "opus_critic") return p;
27477
27918
  const model = resolveOpusCriticModel();
27478
27919
  const allowedEfforts = model === "claude-opus-5" ? [
@@ -27761,7 +28202,7 @@ function jsonPathPreflightCap(body, scope) {
27761
28202
  async function dispatchModelCall(args) {
27762
28203
  const resolvedModel = resolveModel(args.model);
27763
28204
  if (args.endpoint === "/v1/responses") {
27764
- const payload = {
28205
+ const payload = applyResponsesCachePolicy({
27765
28206
  model: resolvedModel,
27766
28207
  instructions: args.instructions,
27767
28208
  input: [{
@@ -27776,7 +28217,7 @@ async function dispatchModelCall(args) {
27776
28217
  }],
27777
28218
  stream: false,
27778
28219
  reasoning: { effort: args.effort }
27779
- };
28220
+ }, { workload: "reusable-prefix" });
27780
28221
  return extractResponsesText(await withTransientRetry(() => createResponses(payload, void 0, args.signal), {
27781
28222
  signal: args.signal,
27782
28223
  label: resolvedModel
@@ -27784,7 +28225,7 @@ async function dispatchModelCall(args) {
27784
28225
  }
27785
28226
  if (args.endpoint === "/v1/messages") {
27786
28227
  const maxTokens = args.effort === "low" ? 4096 : args.effort === "medium" ? 8192 : args.effort === "high" ? 16384 : 32768;
27787
- const body = JSON.stringify({
28228
+ const body = applyClaudeCachePolicy(JSON.stringify({
27788
28229
  model: resolvedModel,
27789
28230
  max_tokens: maxTokens,
27790
28231
  system: args.instructions,
@@ -27807,7 +28248,7 @@ async function dispatchModelCall(args) {
27807
28248
  role: "user",
27808
28249
  content: args.userText
27809
28250
  }]
27810
- });
28251
+ }), { workload: "reusable-prefix" });
27811
28252
  return extractMessagesText(await (await withTransientRetry(() => createMessages(body, void 0, args.signal), {
27812
28253
  signal: args.signal,
27813
28254
  label: resolvedModel
@@ -29309,7 +29750,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
29309
29750
  }
29310
29751
  const conversationText = renderConversationAsText(conversation, maxUnits, measure);
29311
29752
  if (advisorUsesResponses(resolvedAdvisorModel)) {
29312
- const payload = {
29753
+ const payload = applyResponsesCachePolicy({
29313
29754
  model: resolvedAdvisorModel,
29314
29755
  instructions: advisorSystem,
29315
29756
  input: [{
@@ -29321,7 +29762,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
29321
29762
  }],
29322
29763
  stream: false,
29323
29764
  reasoning: { effort: advisorEffort }
29324
- };
29765
+ }, { workload: "reusable-prefix" });
29325
29766
  const response = await withTransientRetry(() => createResponses(payload, void 0, signal), {
29326
29767
  signal,
29327
29768
  label: resolvedAdvisorModel
@@ -29346,7 +29787,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
29346
29787
  const advisorEntry = state.models?.data?.find((m) => m.id === resolvedAdvisorModel);
29347
29788
  const limits = advisorEntry?.capabilities?.limits;
29348
29789
  const maxTokens = limits?.max_non_streaming_output_tokens ?? limits?.max_output_tokens ?? ADVISOR_FALLBACK_MAX_OUTPUT_TOKENS;
29349
- const advisorBody = JSON.stringify({
29790
+ const advisorBody = applyClaudeCachePolicy(JSON.stringify({
29350
29791
  model: resolvedAdvisorModel,
29351
29792
  max_tokens: maxTokens,
29352
29793
  system: advisorSystem,
@@ -29359,7 +29800,7 @@ async function runAdvisor(conversation, advisorModel, advisorEffort, signal, adv
29359
29800
  thinking: { type: "adaptive" },
29360
29801
  ...advertisedEffortLadder(resolvedAdvisorModel) ? { output_config: { effort: advisorEffort } } : {}
29361
29802
  } : {}
29362
- });
29803
+ }), { workload: "reusable-prefix" });
29363
29804
  const json = await (await withTransientRetry(() => createMessages(advisorBody, {}, signal), {
29364
29805
  signal,
29365
29806
  label: resolvedAdvisorModel
@@ -31637,7 +32078,7 @@ function advisorTool(getMessages) {
31637
32078
  const release = acquireInFlightSlot();
31638
32079
  if (!release) throw new Error(`advisor: MCP in-flight cap (${MAX_INFLIGHT_TOOLS_CALL}) saturated; retry shortly`);
31639
32080
  try {
31640
- const text = extractResponsesText(await createResponses({
32081
+ const payload = applyResponsesCachePolicy({
31641
32082
  model: resolvedModel,
31642
32083
  instructions: advisorSystem,
31643
32084
  input: [{
@@ -31649,7 +32090,8 @@ function advisorTool(getMessages) {
31649
32090
  }],
31650
32091
  stream: false,
31651
32092
  reasoning: { effort: ADVISOR_DEFAULT_EFFORT }
31652
- }, void 0, signal, true));
32093
+ }, { workload: "reusable-prefix" });
32094
+ const text = extractResponsesText(await createResponses(payload, void 0, signal, true));
31653
32095
  if (!text) throw new Error("advisor returned empty output");
31654
32096
  return textResult(text);
31655
32097
  } finally {
@@ -32927,7 +33369,7 @@ function appendPlanReminder(messages, planState) {
32927
33369
  * gemini-3.1-pro-preview is pinned to `high` because the model rejects
32928
33370
  * `xhigh` at the wire with a Copilot 400. `high` is the realistic ceiling.
32929
33371
  */
32930
- const STAND_IN_MODELS = Object.freeze([
33372
+ const STAND_IN_MODELS_BASE = Object.freeze([
32931
33373
  {
32932
33374
  key: "gpt-5.6-sol",
32933
33375
  model: "gpt-5.6-sol",
@@ -32947,6 +33389,13 @@ const STAND_IN_MODELS = Object.freeze([
32947
33389
  effort: "high"
32948
33390
  }
32949
33391
  ]);
33392
+ function standInModels() {
33393
+ const geminiModel = resolveGeminiReviewModel();
33394
+ return STAND_IN_MODELS_BASE.map((config) => config.key === "gemini-3.1-pro-preview" ? {
33395
+ ...config,
33396
+ model: geminiModel ?? "gemini-3.7-flash"
33397
+ } : config);
33398
+ }
32950
33399
  const SYSTEM_PROMPT_R1 = `You are one of three frontier reasoning models the user has authorized to stand in for them on a bounded decision while they are unavailable. Your task: pick the best option from those provided.
32951
33400
 
32952
33401
  Respond with ONLY a single JSON object — no prose, no markdown fences, no preamble. Schema:
@@ -32998,7 +33447,7 @@ const RETRY_PROMPT_SUFFIX = `\n\nYour previous response was not valid JSON match
32998
33447
  async function runStandIn(input, signal) {
32999
33448
  const validIds = new Set(input.options.map((o) => o.id));
33000
33449
  const r1UserText = buildRound1UserText(input);
33001
- const r1 = await Promise.all(STAND_IN_MODELS.map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R1, r1UserText, validIds, signal)));
33450
+ const r1 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R1, r1UserText, validIds, signal)));
33002
33451
  const successfulR1 = r1.filter((r) => isVote(r.vote));
33003
33452
  const nmiR1 = gapAbstainVerdict(successfulR1, r1, null);
33004
33453
  if (nmiR1) return nmiR1;
@@ -33018,7 +33467,7 @@ async function runStandIn(input, signal) {
33018
33467
  notes: `Only ${successfulR1.length} of 3 models returned a parseable round-1 vote; insufficient signal to run round 2.`
33019
33468
  }, r1, null);
33020
33469
  const r2UserTextBase = buildRound2UserTextBase(input, r1);
33021
- const r2 = await Promise.all(STAND_IN_MODELS.map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R2, r2UserTextBase + `\n\nYou are ${cfg.key}. Reconsider and vote.`, validIds, signal)));
33470
+ const r2 = await Promise.all(standInModels().map((cfg) => callAndParse(cfg, SYSTEM_PROMPT_R2, r2UserTextBase + `\n\nYou are ${cfg.key}. Reconsider and vote.`, validIds, signal)));
33022
33471
  const successfulR2 = r2.filter((r) => isVote(r.vote));
33023
33472
  if (successfulR2.length < 2) return withDerivedNotes({
33024
33473
  verdict: "no_consensus",
@@ -33195,7 +33644,7 @@ function aggregateVotes(results) {
33195
33644
  topCount = count;
33196
33645
  topSumConfidence = sumConfidence;
33197
33646
  }
33198
- const total = STAND_IN_MODELS.length;
33647
+ const total = standInModels().length;
33199
33648
  if (topChoice && topCount === total) return {
33200
33649
  verdict: "consensus",
33201
33650
  winner: topChoice,
@@ -33246,7 +33695,7 @@ function gapAbstainVerdict(successful, r1, r2) {
33246
33695
  const gapVotes = successful.filter((r) => r.vote.choice === null && r.vote.needMoreInfo);
33247
33696
  if (gapVotes.length < 2) return null;
33248
33697
  const gaps = gapVotes.map((r) => `- ${r.key}: ${r.vote.needMoreInfo}`).join("\n");
33249
- const header = gapVotes.length === STAND_IN_MODELS.length ? "All three models reported they need more context to decide:" : `${gapVotes.length} of 3 models reported they need more context to decide:`;
33698
+ const header = gapVotes.length === standInModels().length ? "All three models reported they need more context to decide:" : `${gapVotes.length} of 3 models reported they need more context to decide:`;
33250
33699
  return withDerivedNotes({
33251
33700
  verdict: "need_more_info",
33252
33701
  recommendation: null,
@@ -33262,7 +33711,7 @@ function gapAbstainVerdict(successful, r1, r2) {
33262
33711
  */
33263
33712
  function freshestVotes(r1, r2) {
33264
33713
  const out = [];
33265
- for (const cfg of STAND_IN_MODELS) {
33714
+ for (const cfg of standInModels()) {
33266
33715
  const r2Entry = r2?.find((r) => r.key === cfg.key);
33267
33716
  const r1Entry = r1.find((r) => r.key === cfg.key);
33268
33717
  const vote = r2Entry && isVote(r2Entry.vote) ? r2Entry.vote : r1Entry && isVote(r1Entry.vote) ? r1Entry.vote : null;
@@ -33300,7 +33749,7 @@ function withDerivedNotes(result, r1, r2) {
33300
33749
  }
33301
33750
  function voteRecord(r1, r2) {
33302
33751
  const record = {};
33303
- for (const cfg of STAND_IN_MODELS) {
33752
+ for (const cfg of standInModels()) {
33304
33753
  const r1Entry = r1.find((r) => r.key === cfg.key);
33305
33754
  const r2Entry = r2?.find((r) => r.key === cfg.key) ?? null;
33306
33755
  record[cfg.key] = {
@@ -34529,7 +34978,7 @@ const PERSONAS_READ = Object.freeze([
34529
34978
  {
34530
34979
  agentName: "gemini-critic",
34531
34980
  toolNameHttp: "gemini_critic",
34532
- model: "gemini-3.1-pro-preview",
34981
+ model: GEMINI_REVIEW_DEFAULT_MODEL,
34533
34982
  endpoint: "/v1/chat/completions",
34534
34983
  description: "Adversarial third-lab critic backed by gemini-3.1-pro-preview (Google), strong on formal reasoning, invariants, proofs, and cross-checking another critic's conclusion. It reviews plans, designs, mathematical arguments, and large artifacts for assumption gaps or invariant failures, then returns a focused critique or no-material-objection style verdict. Use when codex_critic's result needs an independent lab check or when the artifact hinges on formal correctness. Not for line-level diff review, use gemini_reviewer or codex_reviewer; pass the artifact and constraints verbatim.",
34535
34984
  baseInstructions: GEMINI_CRITIC_BASE,
@@ -34565,7 +35014,7 @@ const PERSONAS_READ = Object.freeze([
34565
35014
  {
34566
35015
  agentName: "gemini-reviewer",
34567
35016
  toolNameHttp: "gemini_reviewer",
34568
- model: "gemini-3.1-pro-preview",
35017
+ model: GEMINI_REVIEW_DEFAULT_MODEL,
34569
35018
  endpoint: "/v1/chat/completions",
34570
35019
  description: "Line-level code reviewer backed by gemini-3.1-pro-preview (Google), providing second-lab coverage that catches a different slice of concrete-code defects than codex_reviewer. It reviews diffs, files, or function bodies and returns severity-ranked findings with file:line citations and suggested fixes. Use alongside codex_reviewer when a non-trivial diff benefits from cross-lab code-review coverage, especially around invariants or edge cases. Not for architecture or product-design review, use codex_critic or gemini_critic; pass the code artifact verbatim.",
34571
35020
  baseInstructions: GEMINI_REVIEWER_BASE,
@@ -34734,15 +35183,16 @@ function buildPeerAwarenessSnippet(opts) {
34734
35183
  const powerBrowseAvailable = opts.browseAvailable && opts.powerBrowseAvailable === true;
34735
35184
  const criticList = ["`codex_critic` (gpt-5.6-sol)", "`codex_reviewer` (gpt-5.3-codex)"];
34736
35185
  if (opts.geminiAvailable) {
34737
- criticList.push("`gemini_reviewer` (gemini-3.1-pro, line-level code review)");
34738
- criticList.push("`gemini_critic` (gemini-3.1-pro)");
35186
+ const geminiModel = opts.geminiModel ?? "gemini-3.1-pro-preview";
35187
+ criticList.push(`\`gemini_reviewer\` (${geminiModel}, line-level code review)`);
35188
+ criticList.push(`\`gemini_critic\` (${geminiModel})`);
34739
35189
  }
34740
35190
  criticList.push("`opus_critic` (Opus 5)");
34741
35191
  const codexCliClause = opts.codexCli ? " `mcp__codex-cli__codex` dispatches to `codex-implementer` (gpt-5.3-codex with workspace-write) for end-to-end coding tasks." : "";
34742
35192
  const para2Parts = [`\`mcp__${searchKey}__code\` is the one-stop code search (no extra model call). Its DEFAULT mode (or \`mode:"semantic"\`) ranks by MEANING via ColBERT over a per-workspace index, the first thing to reach for on intent/concept questions ("where is retry/backoff handled", "how does auth work"); when that index isn't ready it transparently falls back to lexical (the response \`source\` says which engine ran). Forced modes cover the rest: \`lexical\` (BM25F-ranked + tree-sitter, best for exact symbols), \`exact\`, \`regex\`, \`complete\` (exhaustive set), \`ast_pattern\`+\`ast_lang\` for multi-line AST shapes, \`scan\` for a whole-workspace symbol outline, \`multiline\` for cross-line regex. Multiple queries can run in a single turn. The index covers code-shaped files; for unstructured files (logs, \`.csv\`, \`.env*\`, config-only wiring), \`grep\`/\`glob\` still apply.`];
34743
35193
  if (opts.workerToolsAvailable) para2Parts.push(`\`worker-*\` are background Agent subagents (subagent_type) that run the matching worker in its own context and deliver the result as a completion notification, so a long run never blocks the turn: \`worker-explore\` (read-only research), \`worker-review\` (reads the code to verify a change or claim), \`worker-plan\` (ordered implementation plan), \`worker-implement\` (edit/write/bash; ALWAYS runs in an isolated git worktree and returns the diff via a saved patch file; for in-place edits use the \`implementer\` subagent), \`worker-test\` (independent test author; also always worktree-isolated)${opts.browseAvailable ? ", `worker-browse` (autonomous browser agent driving a real browser)" : ""}. The raw \`mcp__${workersKey}__*\` tools they call are guarded (a direct main-thread call is redirected to the matching agent); Workers themselves have \`code_search\`.`);
34744
35194
  const catchAllClause = opts.generalPurposeFastAvailable === false ? "" : " Catch-all on a fast, economical non-lead model for work no specialist fits: `general-purpose-fast`.";
34745
- para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure), \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
35195
+ para2Parts.push(`Native subagents (Task), each in its own context so heavy work never fills yours: \`implementer\` (coding changes needing judgment or with ambiguous scope)${opts.implementerFastAvailable === false ? "" : ", `implementer-fast` (well-specified, mechanical coding changes)"}, \`reviewer\` (something exists and you want it assessed, including reproducing and root-causing a failure)${opts.reviewerFastAvailable === false ? "" : ", `reviewer-fast` (lower-stakes assessment on a cheaper cross-lab model)"}, \`brainstorm\` (you do not yet know which approach to take)${opts.scoutAvailable === false ? "" : ", `scout` (find or understand something in the repo, cheap)"}, \`scribe\` (docs and ADRs that trail the code).${catchAllClause}`);
34746
35196
  if (opts.workerToolsAvailable) para2Parts.push(`For a bounded, well-scoped implementation, prefer the \`implementer\` subagent over \`worker-implement\`; reach for \`worker-implement\` only when you specifically need git-worktree isolation, parallel variants, or a throwaway experiment.`);
34747
35197
  if (opts.workerToolsAvailable) para2Parts.push(`\`mcp__${orchestrateKey}__decompose\` composes an open-ended ask into a typed, VERIFIED workflow IR (a strong driver decorrelated by a cross-lab critic, so the decompose step isn't a single point of failure), and \`mcp__${orchestrateKey}__run_workflow\` executes that IR through a frozen kernel delivering max(orchestrated, baseline) over a sealed executable gate, so it never ships worse than a plain single-model run. \`mcp__${orchestrateKey}__verify_workflow\` checks an IR's floor invariants before you run it, and \`mcp__${orchestrateKey}__attest_step\` audits that a finished run's producers were each checked by a different lab. They suit non-trivial, role-separated asks; a trivial ask does not need them.`);
34748
35198
  else para2Parts.push(`\`mcp__${orchestrateKey}__verify_workflow\` statically checks a workflow IR's floor invariants and \`mcp__${orchestrateKey}__attest_step\` audits a run's cross-lab lineage (the \`decompose\`/\`run_workflow\` composer + kernel need the worker backend, unavailable here).`);
@@ -34787,6 +35237,7 @@ function buildPeerAwarenessSummary(opts) {
34787
35237
  renderNative("implementer"),
34788
35238
  opts.implementerFastAvailable === false ? void 0 : renderNative("implementer-fast"),
34789
35239
  renderNative("reviewer"),
35240
+ opts.reviewerFastAvailable === false ? void 0 : renderNative("reviewer-fast"),
34790
35241
  renderNative("brainstorm"),
34791
35242
  opts.scoutAvailable === false ? void 0 : renderNative("scout"),
34792
35243
  renderNative("scribe"),
@@ -34807,11 +35258,38 @@ function buildPeerAwarenessSummary(opts) {
34807
35258
  lines.push(`Each tool's own description carries when to use it and when not. The full per-tool inventory (models, gating, workers, skills) is in the "Peer review and advisor" section of your CLAUDE.md project instructions.`);
34808
35259
  return lines.join("\n");
34809
35260
  }
35261
+ /**
35262
+ * Applies the resolved Gemini review model to a persona requiring the Gemini
35263
+ * catalog: swaps `.model` and rewrites every literal occurrence of the
35264
+ * default id in `.description` so the two never disagree about which model
35265
+ * actually backs the tool. A mismatch here is user-visible — `tools/list`
35266
+ * would advertise "backed by gemini-3.1-pro-preview" while dispatch actually
35267
+ * ran the flash fallback — so every call site that resolves a
35268
+ * `requiresGeminiCatalog` persona MUST go through this helper rather than
35269
+ * setting `.model` directly (a prior draft of this fix did exactly that in
35270
+ * `routes/mcp/handler.ts`'s `activePersonas()` and left the description
35271
+ * stale). Relies on every `requiresGeminiCatalog` persona's description
35272
+ * literally containing `GEMINI_REVIEW_DEFAULT_MODEL`'s exact string — true
35273
+ * for both current entries (gemini-critic, gemini-reviewer); keep it true for
35274
+ * any future one, or `replaceAll` silently no-ops.
35275
+ */
35276
+ function resolveGeminiPersona(p, geminiModel) {
35277
+ const model = geminiModel ?? "gemini-3.1-pro-preview";
35278
+ return {
35279
+ ...p,
35280
+ model,
35281
+ description: p.description.replaceAll(GEMINI_REVIEW_DEFAULT_MODEL, model)
35282
+ };
35283
+ }
34810
35284
  /** Convenience: every persona that should be registered for the given mode. */
34811
35285
  function personasFor(opts) {
34812
35286
  const result = [];
34813
35287
  for (const p of PERSONAS_READ) {
34814
- if (p.requiresGeminiCatalog && !opts.geminiAvailable) continue;
35288
+ if (p.requiresGeminiCatalog) {
35289
+ if (!opts.geminiAvailable) continue;
35290
+ result.push(resolveGeminiPersona(p, opts.geminiModel));
35291
+ continue;
35292
+ }
34815
35293
  result.push(p);
34816
35294
  }
34817
35295
  if (opts.codexCli) for (const p of PERSONAS_WRITE) result.push(p);
@@ -35997,6 +36475,6 @@ function enumerateInjectedMcpToolNames(groupKeys, opts = {}) {
35997
36475
  return [...new Set(names)];
35998
36476
  }
35999
36477
  //#endregion
36000
- export { handleMcpDelete as $, resolveLeadSlugArg as $t, satisfiesMinVersion as A, hasSupportedBrowserInstalled as At, rememberThinkingHistoryRepair as B, collapsePathKeys as Bt, availableToolCommands as C, resolveMcpToolTimeoutMs as Ct, vscodeRipgrepPath as D, readResponseBodyCapped as Dt, toolbeltSkipSet as E, MAX_RESPONSE_BODY_BYTES as Et, injectAdvisorTool as F, warmTreeSitterPool as Ft, isControllerClosedError as G, DEFAULT_CODEX_MODEL as Gt, repairRejectedThinkingHistory as H, BUDGET_SMALL_FAST_CATALOG_ID as Ht, isAdvisorRequested as I, provisionTreeSitterAssets as It, relayAnthropicStream as J, UPSTREAM_FETCH_TIMEOUT_MS as Jt, logStreamError as K, DEFAULT_CODEX_MODEL_FALLBACKS as Kt, resolveAdvisorEffort as L, CONDENSED_OPERATING_SEQUENCE as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, provisionAndIndexColbert as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, extractTarGzMember as Nt, TOOLBELT_TOOLS$1 as O, parseJsonOrDiagnose as Ot, buildAdvisorStream as P, extractZipMember as Pt, clampEffort as Q, pickClaudeDefault as Qt, resolveAdvisorModel as R, DEFINITION_OF_GREATNESS as Rt, buildEnv as S, warnOnTokenPriceDrift as St, toolbeltEnabled as T, createChatCompletions as Tt, buildAnthropicErrorEvent as U, BUDGET_SMALL_FAST_SLUG as Ut, repairKnownThinkingHistory as V, toolbeltPathOverride as Vt, buildOpenAIErrorEvent as W, DEFAULT_CLAUDE_MODEL_FALLBACKS as Wt, UNKNOWN_EFFORT_ANCHOR as X, generateRandomPort as Xt, EFFORT_ORDER as Y, UPSTREAM_INACTIVITY_TIMEOUT_MS as Yt, bucketEffort as Z, isBudgetClaudeLead as Zt, appendPlanReminder as _, shimDefaultsToXhigh as _t, buildPeerAwarenessSnippet as a, withInstallLock as an, browserCompoundToolsEnabled as at, resolveWorkerRunOpts as b, getTokenCount as bt, personasFor as c, geminiAvailable as ct, EXPLORE_DEFAULT_MODEL as d, nativeSubagentModel as dt, upstreamAllowH2 as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, reviewerModel as ft, TEST_DEFAULT_MODEL as g, workerToolsEnabled as gt, REVIEW_DEFAULT_MODEL as h, standInToolEnabled as ht, buildAgentPrompt as i, withOneMSuffixForLead as in, browseAgentEnabled as it, searchWeb as j, colbertDegradedWarning as jt, assetFor as k, provisionBrowserAssets as kt, BROWSE_DEFAULT_MODEL as l, generalPurposeFastModel as lt, PLAN_DEFAULT_MODEL as m, scribeModel as mt, MCP_GROUPS as n, classifyMessagesRoute as nn, artifactToolsEnabled as nt, buildPeerAwarenessSummary as o, browserToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, scoutModel as pt, readIteratorWithTimeout as q, DEFAULT_PORT as qt, assertMcpToolSurfaceConsistent as r, withOneMSuffix as rn, brainstormModel as rt, enumerateInjectedMcpToolNames as s, fleetToolsEnabled as st, GROUP_META as t, upstreamMaxConnections as tn, agentToolsEnabled as tt, DEFAULT_MODEL_CHAIN as u, implementerFastModel as ut, resolveDefaultModel as v, countTokens as vt, buildToolbeltAwareness as w, createResponses as wt, runWorkerAgent as x, assembleResponsesPayload as xt, resolveModeDefaults as y, createMessages as yt, formatThinkingRepairDecline as z, shouldUseInsecureTls as zt };
36478
+ export { handleMcpDelete as $, UPSTREAM_INACTIVITY_TIMEOUT_MS as $t, satisfiesMinVersion as A, readResponseBodyCapped as At, rememberThinkingHistoryRepair as B, provisionTreeSitterAssets as Bt, availableToolCommands as C, getTokenCount as Ct, vscodeRipgrepPath as D, createResponses as Dt, toolbeltSkipSet as E, resolveMcpToolTimeoutMs as Et, injectAdvisorTool as F, colbertDegradedWarning as Ft, isControllerClosedError as G, toolbeltPathOverride as Gt, repairRejectedThinkingHistory as H, DEFINITION_OF_GREATNESS as Ht, isAdvisorRequested as I, provisionAndIndexColbert as It, relayAnthropicStream as J, DEFAULT_CLAUDE_MODEL_FALLBACKS as Jt, logStreamError as K, BUDGET_SMALL_FAST_CATALOG_ID as Kt, resolveAdvisorEffort as L, extractTarGzMember as Lt, ADVISOR_INTERNAL_TOOL_NAME as M, normalizeOpenAIUsage as Mt, ADVISOR_TOOL_INSTRUCTIONS as N, provisionBrowserAssets as Nt, TOOLBELT_TOOLS$1 as O, createChatCompletions as Ot, buildAdvisorStream as P, hasSupportedBrowserInstalled as Pt, clampEffort as Q, UPSTREAM_FETCH_TIMEOUT_MS as Qt, resolveAdvisorModel as R, extractZipMember as Rt, buildEnv as S, createMessages as St, toolbeltEnabled as T, warnOnTokenPriceDrift as Tt, buildAnthropicErrorEvent as U, shouldUseInsecureTls as Ut, repairKnownThinkingHistory as V, CONDENSED_OPERATING_SEQUENCE as Vt, buildOpenAIErrorEvent as W, collapsePathKeys as Wt, UNKNOWN_EFFORT_ANCHOR as X, DEFAULT_CODEX_MODEL_FALLBACKS as Xt, EFFORT_ORDER as Y, DEFAULT_CODEX_MODEL as Yt, bucketEffort as Z, DEFAULT_PORT as Zt, appendPlanReminder as _, scribeModel as _t, buildPeerAwarenessSnippet as a, upstreamMaxConnections as an, browseAgentEnabled as at, resolveWorkerRunOpts as b, shimDefaultsToXhigh as bt, personasFor as c, withOneMSuffixForLead as cn, fleetToolsEnabled as ct, EXPLORE_DEFAULT_MODEL as d, implementerFastModel as dt, generateRandomPort as en, handleMcpPost as et, EXPLORE_DEFAULT_THINKING as f, nativeSubagentModel as ft, TEST_DEFAULT_MODEL as g, scoutModel as gt, REVIEW_DEFAULT_MODEL as h, reviewerModel as ht, buildAgentPrompt as i, upstreamAllowH2 as in, brainstormModel as it, searchWeb as j, parseJsonOrDiagnose as jt, assetFor as k, MAX_RESPONSE_BODY_BYTES as kt, BROWSE_DEFAULT_MODEL as l, withInstallLock as ln, geminiAvailable as lt, PLAN_DEFAULT_MODEL as m, reviewerFastModel as mt, MCP_GROUPS as n, pickClaudeDefault as nn, agentToolsEnabled as nt, buildPeerAwarenessSummary as o, classifyMessagesRoute as on, browserCompoundToolsEnabled as ot, IMPLEMENT_DEFAULT_MODEL as p, resolveGeminiReviewModel as pt, readIteratorWithTimeout as q, BUDGET_SMALL_FAST_SLUG as qt, assertMcpToolSurfaceConsistent as r, resolveLeadSlugArg as rn, artifactToolsEnabled as rt, enumerateInjectedMcpToolNames as s, withOneMSuffix as sn, browserToolsEnabled as st, GROUP_META as t, isBudgetClaudeLead as tn, REVIEW_FAST_DEFAULT_MODEL as tt, DEFAULT_MODEL_CHAIN as u, generalPurposeFastModel as ut, resolveDefaultModel as v, standInToolEnabled as vt, buildToolbeltAwareness as w, assembleResponsesPayload as wt, runWorkerAgent as x, countTokens as xt, resolveModeDefaults as y, workerToolsEnabled as yt, formatThinkingRepairDecline as z, warmTreeSitterPool as zt };
36001
36479
 
36002
- //# sourceMappingURL=peer-mcp-personas-V6stFvpq.js.map
36480
+ //# sourceMappingURL=peer-mcp-personas-Bd56EmiO.js.map