github-router 0.3.322 → 0.3.324

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/dist/{aic-ledger-BW3zMiCx.js → aic-ledger-DRU89gJg.js} +2 -2
  2. package/dist/{aic-ledger-BW3zMiCx.js.map → aic-ledger-DRU89gJg.js.map} +1 -1
  3. package/dist/{attribution-settings-z-U2L7ab.js → attribution-settings-sTUXqFzx.js} +649 -101
  4. package/dist/attribution-settings-sTUXqFzx.js.map +1 -0
  5. package/dist/{auth-Iun7ftd8.js → auth-DSLN1arH.js} +3 -3
  6. package/dist/{auth-Iun7ftd8.js.map → auth-DSLN1arH.js.map} +1 -1
  7. package/dist/browser-ext/manifest.json +1 -1
  8. package/dist/{check-usage-1c-0z3iy.js → check-usage-B_12ALvf.js} +4 -4
  9. package/dist/{check-usage-1c-0z3iy.js.map → check-usage-B_12ALvf.js.map} +1 -1
  10. package/dist/{claude-DtWf7DEw.js → claude-Bni6m56W.js} +61 -49
  11. package/dist/claude-Bni6m56W.js.map +1 -0
  12. package/dist/{codex-CDtFT-Qp.js → codex-DAnu246m.js} +5 -5
  13. package/dist/{codex-CDtFT-Qp.js.map → codex-DAnu246m.js.map} +1 -1
  14. package/dist/{copilot-discount-I6BrT7QE.js → copilot-discount-D8t7Q3Tp.js} +2 -2
  15. package/dist/{copilot-discount-I6BrT7QE.js.map → copilot-discount-D8t7Q3Tp.js.map} +1 -1
  16. package/dist/{debug-CDyXUPcC.js → debug-D8iD6G4O.js} +2 -2
  17. package/dist/{debug-CDyXUPcC.js.map → debug-D8iD6G4O.js.map} +1 -1
  18. package/dist/engine-4uuaURb0.js +2 -0
  19. package/dist/{fast-profile-contract-_DhIWank.js → fast-profile-contract-DmsyQQrw.js} +6 -11
  20. package/dist/fast-profile-contract-DmsyQQrw.js.map +1 -0
  21. package/dist/{gate-discovery-j_hQ1IPc.js → gate-discovery-Z0QyjqM7.js} +5 -5
  22. package/dist/{gate-discovery-j_hQ1IPc.js.map → gate-discovery-Z0QyjqM7.js.map} +1 -1
  23. package/dist/{get-copilot-usage-BWCAzmSc.js → get-copilot-usage-Bs-FEPMf.js} +2 -2
  24. package/dist/{get-copilot-usage-BWCAzmSc.js.map → get-copilot-usage-Bs-FEPMf.js.map} +1 -1
  25. package/dist/hooks.mjs +32 -12
  26. package/dist/hooks.sha256 +1 -1
  27. package/dist/{internal-aic-status-CKXxxb30.js → internal-aic-status-BaUKlisM.js} +3 -3
  28. package/dist/{internal-aic-status-CKXxxb30.js.map → internal-aic-status-BaUKlisM.js.map} +1 -1
  29. package/dist/{internal-artifact-open-CA1trJ15.js → internal-artifact-open-U-Ht7R-O.js} +2 -2
  30. package/dist/{internal-artifact-open-CA1trJ15.js.map → internal-artifact-open-U-Ht7R-O.js.map} +1 -1
  31. package/dist/{internal-fast-dispatch-guard-CZ4BZuAG.js → internal-fast-dispatch-guard-Dnjoj0pV.js} +4 -5
  32. package/dist/internal-fast-dispatch-guard-Dnjoj0pV.js.map +1 -0
  33. package/dist/{internal-fast-dispatch-guard-D5kozMpo.js → internal-fast-dispatch-guard-uztSu0c8.js} +1 -1
  34. package/dist/{internal-first-mate-guard-CiFH6X1H.js → internal-first-mate-guard-Cj_g7RFr.js} +3 -3
  35. package/dist/{internal-first-mate-guard-CiFH6X1H.js.map → internal-first-mate-guard-Cj_g7RFr.js.map} +1 -1
  36. package/dist/{internal-first-mate-guard-B_or0eN3.js → internal-first-mate-guard-ZqPHrYlh.js} +1 -1
  37. package/dist/{internal-max-dispatch-guard-BGOHRYRt.js → internal-max-dispatch-guard-CxG_c3cc.js} +1 -1
  38. package/dist/{internal-max-dispatch-guard-CnBZxo5l.js → internal-max-dispatch-guard-Sp_rCYk_.js} +2 -2
  39. package/dist/{internal-max-dispatch-guard-CnBZxo5l.js.map → internal-max-dispatch-guard-Sp_rCYk_.js.map} +1 -1
  40. package/dist/{internal-plan-review-C73snnD8.js → internal-plan-review-Co01iHOW.js} +3 -3
  41. package/dist/{internal-plan-review-C73snnD8.js.map → internal-plan-review-Co01iHOW.js.map} +1 -1
  42. package/dist/{internal-prompt-submit-dov4UvBo.js → internal-prompt-submit-dflz0bqH.js} +4 -4
  43. package/dist/{internal-prompt-submit-dov4UvBo.js.map → internal-prompt-submit-dflz0bqH.js.map} +1 -1
  44. package/dist/{internal-session-bind-C3hrb9rP.js → internal-session-bind-BVQy7Vd4.js} +2 -2
  45. package/dist/{internal-session-bind-C3hrb9rP.js.map → internal-session-bind-BVQy7Vd4.js.map} +1 -1
  46. package/dist/{internal-stop-hook-BGTPRCtO.js → internal-stop-hook-DihBVzn2.js} +5 -5
  47. package/dist/{internal-stop-hook-BGTPRCtO.js.map → internal-stop-hook-DihBVzn2.js.map} +1 -1
  48. package/dist/{internal-stop-review-MBVYtrYS.js → internal-stop-review-CrnG4IJo.js} +2 -2
  49. package/dist/{internal-stop-review-MBVYtrYS.js.map → internal-stop-review-CrnG4IJo.js.map} +1 -1
  50. package/dist/{internal-worker-guard-dxGawAY3.js → internal-worker-guard-Dd14aDXQ.js} +2 -2
  51. package/dist/{internal-worker-guard-dxGawAY3.js.map → internal-worker-guard-Dd14aDXQ.js.map} +1 -1
  52. package/dist/{internal-workspace-header-_aEXOz5q.js → internal-workspace-header-booLOzwD.js} +2 -2
  53. package/dist/{internal-workspace-header-_aEXOz5q.js.map → internal-workspace-header-booLOzwD.js.map} +1 -1
  54. package/dist/{lifecycle-CJB2BwSJ.js → lifecycle-BYPEqFdV.js} +2 -2
  55. package/dist/{lifecycle-CJB2BwSJ.js.map → lifecycle-BYPEqFdV.js.map} +1 -1
  56. package/dist/{lifecycle-_BE0SnNs.js → lifecycle-C1LDvDD6.js} +2 -2
  57. package/dist/{lifecycle-_BE0SnNs.js.map → lifecycle-C1LDvDD6.js.map} +1 -1
  58. package/dist/lifecycle-Cz99P30h.js +2 -0
  59. package/dist/lifecycle-DqTFEJpS.js +2 -0
  60. package/dist/main.js +20 -20
  61. package/dist/{mcp-workspace-header-mXRLE8kg.js → mcp-workspace-header-BIW0SRgs.js} +2 -2
  62. package/dist/{mcp-workspace-header-mXRLE8kg.js.map → mcp-workspace-header-BIW0SRgs.js.map} +1 -1
  63. package/dist/{models-CSFIVxAH.js → models-B9tdaM6B.js} +3 -3
  64. package/dist/{models-CSFIVxAH.js.map → models-B9tdaM6B.js.map} +1 -1
  65. package/dist/{orchestration-B8kypXcZ.js → orchestration-DvsdorFq.js} +2 -2
  66. package/dist/{orchestration-B8kypXcZ.js.map → orchestration-DvsdorFq.js.map} +1 -1
  67. package/dist/{paths-CHBAj_t9.js → paths-BnZwolac.js} +4 -4
  68. package/dist/{paths-CHBAj_t9.js.map → paths-BnZwolac.js.map} +1 -1
  69. package/dist/paths-BzM6uxmd.js +2 -0
  70. package/dist/{peer-mcp-personas-Gb0LfII0.js → peer-mcp-personas-DcAErb_d.js} +201 -66
  71. package/dist/peer-mcp-personas-DcAErb_d.js.map +1 -0
  72. package/dist/{plan-review-hook-BMmw7wxR.js → plan-review-hook-BlsGnG02.js} +3 -3
  73. package/dist/{plan-review-hook-BMmw7wxR.js.map → plan-review-hook-BlsGnG02.js.map} +1 -1
  74. package/dist/{prompt-submit-hook-CGgT5m4Q.js → prompt-submit-hook-BJFLzCaT.js} +3 -3
  75. package/dist/{prompt-submit-hook-CGgT5m4Q.js.map → prompt-submit-hook-BJFLzCaT.js.map} +1 -1
  76. package/dist/{provision-DagZEkdZ.js → provision-B-a5KUCL.js} +4 -4
  77. package/dist/{provision-DagZEkdZ.js.map → provision-B-a5KUCL.js.map} +1 -1
  78. package/dist/{self-invocation-DahN1gw9.js → self-invocation-opSXHEs0.js} +2 -2
  79. package/dist/{self-invocation-DahN1gw9.js.map → self-invocation-opSXHEs0.js.map} +1 -1
  80. package/dist/{serve-DOMsmIk-.js → serve-BGGcy8Nq.js} +14 -13
  81. package/dist/serve-BGGcy8Nq.js.map +1 -0
  82. package/dist/{server-setup-DQVRJluz.js → server-setup-CZ3tEsgq.js} +237 -49
  83. package/dist/server-setup-CZ3tEsgq.js.map +1 -0
  84. package/dist/{start-CqSFR_s8.js → start-XFIxMRkE.js} +3 -3
  85. package/dist/{start-CqSFR_s8.js.map → start-XFIxMRkE.js.map} +1 -1
  86. package/dist/{stop-gate-hook-D7N759TG.js → stop-gate-hook-DvNM8zFf.js} +3 -3
  87. package/dist/{stop-gate-hook-D7N759TG.js.map → stop-gate-hook-DvNM8zFf.js.map} +1 -1
  88. package/dist/{stop-gate-policy-C0gt04R0.js → stop-gate-policy-C5pXbY59.js} +2 -2
  89. package/dist/{stop-gate-policy-C0gt04R0.js.map → stop-gate-policy-C5pXbY59.js.map} +1 -1
  90. package/dist/{token-86uk6y4P.js → token-Css-ARGL.js} +2 -2
  91. package/dist/{token-86uk6y4P.js.map → token-Css-ARGL.js.map} +1 -1
  92. package/dist/{worker-dispatch-B-OA7tvU.js → worker-dispatch-CcQygiAA.js} +28 -2
  93. package/dist/{worker-dispatch-B-OA7tvU.js.map → worker-dispatch-CcQygiAA.js.map} +1 -1
  94. package/package.json +1 -1
  95. package/dist/attribution-settings-z-U2L7ab.js.map +0 -1
  96. package/dist/claude-DtWf7DEw.js.map +0 -1
  97. package/dist/engine-DA5laGIh.js +0 -2
  98. package/dist/fast-profile-contract-_DhIWank.js.map +0 -1
  99. package/dist/internal-fast-dispatch-guard-CZ4BZuAG.js.map +0 -1
  100. package/dist/lifecycle-D0KsoSqz.js +0 -2
  101. package/dist/lifecycle-DimEguJO.js +0 -2
  102. package/dist/paths-eHoOwFzZ.js +0 -2
  103. package/dist/peer-mcp-personas-Gb0LfII0.js.map +0 -1
  104. package/dist/serve-DOMsmIk-.js.map +0 -1
  105. package/dist/server-setup-DQVRJluz.js.map +0 -1
@@ -1,12 +1,12 @@
1
- import { An as oneMContextDisabled, Fn as CHEAPEST_PROFILE_NATIVE_AGENT_NAMES, Hn as CHEAP_PROFILE_NATIVE_EFFORTS, In as CHEAPEST_PROFILE_NATIVE_EFFORTS, Vn as CHEAP_PROFILE_NATIVE_AGENT_NAMES, a as buildAgentPrompt, jn as withOneMSuffix, l as maxPersonasFor, ln as CONDENSED_OPERATING_SEQUENCE, n as MCP_GROUPS, t as GROUP_META, u as personasFor, un as DEFINITION_OF_GREATNESS } from "./peer-mcp-personas-Gb0LfII0.js";
2
- import { a as isUnderClaudeConfigMirror, f as writeRuntimeFileSecure, t as PATHS } from "./paths-CHBAj_t9.js";
3
- import { S as LUNA_SCOUT_ALIAS_ID, _ as CHEAP_EXPLORE_ALIAS_ID, b as CHEAP_PLAN_ALIAS_ID, f as CHEAPEST_EXPLORE_ALIAS_ID, g as CHEAPEST_REVIEWER_ALIAS_ID, h as CHEAPEST_PLAN_ALIAS_ID, m as CHEAPEST_IMPLEMENTER_ALIAS_ID, p as CHEAPEST_GENERAL_PURPOSE_ALIAS_ID, v as CHEAP_GENERAL_PURPOSE_ALIAS_ID, x as CHEAP_REVIEWER_ALIAS_ID, y as CHEAP_IMPLEMENTER_ALIAS_ID } from "./server-setup-DQVRJluz.js";
4
- import { c as FAST_PROFILE_MODELS, d as FAST_PROFILE_NATIVE_MODELS, l as FAST_PROFILE_NATIVE_AGENT_NAMES, u as FAST_PROFILE_NATIVE_EFFORTS } from "./fast-profile-contract-_DhIWank.js";
1
+ import { Ln as BALANCED_PROFILE_NATIVE_AGENT_NAMES, Nn as oneMContextDisabled, Pn as withOneMSuffix, Rn as BALANCED_PROFILE_NATIVE_EFFORTS, Un as CHEAPEST_PROFILE_NATIVE_AGENT_NAMES, Wn as CHEAPEST_PROFILE_NATIVE_EFFORTS, Xn as CHEAP_PROFILE_NATIVE_EFFORTS, Yn as CHEAP_PROFILE_NATIVE_AGENT_NAMES, a as buildAgentPrompt, fn as CONDENSED_OPERATING_SEQUENCE, l as maxPersonasFor, n as MCP_GROUPS, pn as DEFINITION_OF_GREATNESS, t as GROUP_META, u as personasFor } from "./peer-mcp-personas-DcAErb_d.js";
2
+ import { a as isUnderClaudeConfigMirror, f as writeRuntimeFileSecure, t as PATHS } from "./paths-BnZwolac.js";
3
+ import { C as CHEAP_REVIEWER_ALIAS_ID, S as CHEAP_PLAN_ALIAS_ID, _ as CHEAPEST_GENERAL_PURPOSE_ALIAS_ID, b as CHEAP_EXPLORE_ALIAS_ID, f as BALANCED_EXPLORE_ALIAS_ID, g as CHEAPEST_EXPLORE_ALIAS_ID, h as BALANCED_REVIEWER_ALIAS_ID, m as BALANCED_PLAN_ALIAS_ID, p as BALANCED_GENERAL_PURPOSE_ALIAS_ID, v as CHEAPEST_PLAN_ALIAS_ID, w as LUNA_SCOUT_ALIAS_ID, x as CHEAP_GENERAL_PURPOSE_ALIAS_ID, y as CHEAPEST_REVIEWER_ALIAS_ID } from "./server-setup-CZ3tEsgq.js";
4
+ import { c as FAST_PROFILE_MODELS, d as FAST_PROFILE_NATIVE_MODELS, l as FAST_PROFILE_NATIVE_AGENT_NAMES, u as FAST_PROFILE_NATIVE_EFFORTS } from "./fast-profile-contract-DmsyQQrw.js";
5
5
  import { C as MAX_PARALLELISM_RULE, S as MAX_COORDINATOR_PROMPT, T as maxNativePrompt, a as MAX_PROFILE_NATIVE_AGENT_NAMES, i as MAX_PROFILE_MODELS, o as MAX_PROFILE_NATIVE_EFFORTS, s as MAX_PROFILE_NATIVE_MODELS, w as maxNativeDescription, x as MAX_COORDINATOR_DESCRIPTION } from "./max-profile-contract-Bl7I_EOC.js";
6
- import "./self-invocation-DahN1gw9.js";
7
- import { a as STRIPPED_AUTH_ROUTING_ENV_KEYS, n as buildCodexProviderConfigFlags } from "./provision-DagZEkdZ.js";
8
- import { n as buildWorkspaceHeaderHelperCommand } from "./mcp-workspace-header-mXRLE8kg.js";
9
- import { a as dispatcherAgentName, c as dispatcherTools, n as activeDispatchModes, o as dispatcherDescription, s as dispatcherPrompt } from "./worker-dispatch-B-OA7tvU.js";
6
+ import "./self-invocation-opSXHEs0.js";
7
+ import { a as STRIPPED_AUTH_ROUTING_ENV_KEYS, n as buildCodexProviderConfigFlags } from "./provision-B-a5KUCL.js";
8
+ import { n as buildWorkspaceHeaderHelperCommand } from "./mcp-workspace-header-BIW0SRgs.js";
9
+ import { a as dispatcherAgentName, c as dispatcherTools, n as activeDispatchModes, o as dispatcherDescription, s as dispatcherPrompt } from "./worker-dispatch-CcQygiAA.js";
10
10
  import consola from "consola";
11
11
  import path from "node:path";
12
12
  import { randomBytes } from "node:crypto";
@@ -222,12 +222,15 @@ function nonEmptyModel(id) {
222
222
  function fileToolSteer(bashUses) {
223
223
  return `Use the dedicated Edit/Write/Read tools for file changes and Grep/Glob for search; reserve Bash for running ${bashUses}, tests, and git. Do not shell out (sed/awk/python/here-docs) to read or edit files.`;
224
224
  }
225
- /** The read-only half of `fileToolSteer`, for agents that never write. */
226
- function readOnlyToolSteer() {
227
- return "Use Read to read files and Grep/Glob plus the semantic code search tool to find them; Bash is for read-only inspection such as git log, git blame, and git show. Do not modify any file, and do not run mutating commands.";
225
+ /** The read-only half of `fileToolSteer`, for agents that never write.
226
+ * Names the semantic code-search tool only when the launch enabled it;
227
+ * otherwise the `code` tool is lexical-only and naming semantic search
228
+ * would send the agent at a mode that just degrades to lexical. */
229
+ function readOnlyToolSteer(semanticAvailable = true) {
230
+ return `Use Read to read files and ${semanticAvailable ? "Grep/Glob plus the semantic code search tool" : "Grep/Glob"} to find them; Bash is for read-only inspection such as git log, git blame, and git show. Do not modify any file, and do not run mutating commands.`;
228
231
  }
229
- function reviewerToolSteer() {
230
- return "Use Read to read files and Grep/Glob plus the semantic code search tool to find them. Use Bash for builds, tests, reproductions, and read-only git inspection; do not modify source-controlled files or use the shell to edit them.";
232
+ function reviewerToolSteer(semanticAvailable = true) {
233
+ return `Use Read to read files and ${semanticAvailable ? "Grep/Glob plus the semantic code search tool" : "Grep/Glob"} to find them. Use Bash for builds, tests, reproductions, and read-only git inspection; do not modify source-controlled files or use the shell to edit them.`;
231
234
  }
232
235
  /**
233
236
  * `tools:` allowlist for read-only natives (`scout`, fast `Explore`, `brainstorm`), modelled
@@ -375,15 +378,103 @@ function buildMaxProfileAgentDefinitions(opts) {
375
378
  for (const name of Object.keys(out)) if (name !== "worker-browse" && !roster.has(name)) delete out[name];
376
379
  return out;
377
380
  }
381
+ /**
382
+ * Shared prompt bodies for the pinned four-agent profiles
383
+ * (`fast`/`cheap`/`cheap1m`/`cheapest`/`balanced`).
384
+ *
385
+ * Cost doctrine, cheapest to most expensive:
386
+ * 1. LEXICAL code search (`mode:"lexical"`/`"exact"`) — zero model cost,
387
+ * exact symbols, filenames, errors, routes, config keys. Always first.
388
+ * 2. SEMANTIC code search (`mode:"semantic"`) — meaning-ranked via ColBERT,
389
+ * for concepts and intent questions. Mentioned ONLY when the launch
390
+ * enabled it (`semanticSearchAvailable`); otherwise the `code` tool is
391
+ * lexical-only and agents must not be told to reach for semantic.
392
+ * 3. `Explore` (budget model) — reads the narrowed files and synthesizes a
393
+ * file:line evidence report. Only the conclusion flows upward.
394
+ * 4. `Plan` (Sol) / `reviewer` (Sonnet/Luna/Gemini) / `oracle` — expensive
395
+ * models see ONLY the synthesized subset, never raw search output.
396
+ *
397
+ * Per-mode tuning:
398
+ * - `cheapest` (straightforward tasks): implicit delegation. The lead
399
+ * handles simple work inline; role descriptions carry no "use
400
+ * proactively" push and no fan-out instruction, so the lead does not pay
401
+ * handoff overhead for work it already holds context for.
402
+ * - `fast`/`cheap`/`cheap1m`/`balanced` (complex tasks): explicit
403
+ * delegation. Descriptions push proactive parallel `Explore` fan-out,
404
+ * `Plan`-first architecture, `General-Purpose` mixed execution, and
405
+ * post-integration `reviewer` verification.
406
+ */
407
+ function pinnedSearchGuidance(semanticAvailable, capitalize = true) {
408
+ const lexical = `${capitalize ? "Start with" : "start with"} exact lexical and symbol search (\`code\` with mode:"lexical" or "exact", plus Grep/Glob) for symbols, filenames, errors, routes, flags, and config keys — it costs no model call`;
409
+ if (!semanticAvailable) return `${lexical}. Pair it with surrounding context lines so callers, guards, and types are visible without a second round trip, and inspect the file whenever the surrounding logic determines the answer.`;
410
+ return `${lexical}; pair semantic search (meaning-ranked, best for intent/concept questions where literal keywords may not appear) with exact lexical and symbol search so that neither naming drift nor synonym mismatch hides a result. Include surrounding context lines in your search results so that callers, guards, and types are visible without a second round trip, and inspect the file whenever the surrounding logic determines the answer.`;
411
+ }
412
+ /**
413
+ * Implicit (cheapest) tuning is composed from the SAME shared remainder as
414
+ * the explicit base — only the proactive-delegation head sentences differ.
415
+ * The tails below are each written once and copied by both variants, so a
416
+ * fix to shared wording lands everywhere and the implicit/explicit diff
417
+ * stays exactly the few sentences that carry the tuning.
418
+ */
419
+ const EXPLORE_DESC_TAIL = "Use when the question spans more than a couple of files. Do not use for planning, edits, or single-file reads. Returns a structured evidence report with file:line citations. Never edits files.";
420
+ function pinnedExploreDescription(explicit) {
421
+ return (explicit ? "Read-only codebase exploration specialist. Use proactively, and launch several in parallel via `Task(subagent_type:\"Explore\")`, to map architecture, trace call chains, or locate the files and symbols a task will touch. " : "Read-only codebase exploration specialist for mapping architecture, tracing call chains, or locating the files and symbols a task touches. ") + EXPLORE_DESC_TAIL;
422
+ }
423
+ function pinnedExplorePrompt(opts) {
424
+ const searchGuidance = pinnedSearchGuidance(opts.semanticAvailable, !opts.explicit);
425
+ return `You are a codebase exploration specialist. Your mission is to map repository structure, discover implementation patterns, trace call chains, and locate the exact files, symbols, and declarations that are relevant to the request. This is read-only work. Do not modify files, do not propose diffs, and do not delegate to other agents. You cannot ask clarifying questions mid-run: ground every answer in repository evidence. If the request is ambiguous, explore the most probable interpretations and record the ambiguity in your report. Start broad, then converge. ${opts.explicit ? "Issue independent searches in parallel in one turn rather than one at a time, and " : ""}${searchGuidance} Confirm every claim at the source before you report it. Stop when further searching stops changing your answer. When you can name the exact files and lines a change would touch, you are done. Report what the repository contains, not what it ought to contain. Do not design a solution or recommend an approach. Return a self-contained result the lead can act on immediately without needing to re-run your discovery.
426
+
427
+ Return format:
428
+ Answer: a direct response to what was asked, in a few sentences.
429
+ Inventory: each relevant file and symbol as file:line, with a one-line description of its role.
430
+ Entry points: where control enters this area, as file:line.
431
+ Conventions in use: the patterns, idioms, error handling, and test style that any change here would be expected to follow, each with a file:line example.
432
+ Gaps and unknowns: what you could not confirm, and where you would look next.
433
+
434
+ ` + readOnlyToolSteer(opts.semanticAvailable);
435
+ }
436
+ const PLAN_DESC_HEAD = "Architecture and implementation planning specialist";
437
+ const PLAN_DESC_TAIL = " Returns a decision-complete, ordered implementation plan with runnable acceptance criteria. Never edits files.";
438
+ function pinnedPlanDescription(explicit) {
439
+ return PLAN_DESC_HEAD + (explicit ? ". Use proactively in plan mode, and whenever sequencing, cross-boundary interfaces, invariants, migration risk, or acceptance criteria deserve a dedicated pass before any code is written. Delegates repository discovery to `Explore` rather than reading broadly itself." : " for sequencing, cross-boundary interfaces, invariants, migration risk, and acceptance criteria before any code is written.") + PLAN_DESC_TAIL;
440
+ }
441
+ function pinnedPlanPrompt(opts) {
442
+ return `You are a software architect and planning specialist. Your mission is to turn a request into a decision-complete implementation plan: an ordered sequence of changes, the invariants that must hold throughout, and acceptance criteria a reviewer can actually run. This is read-only work. Do not modify repository files. Produce the architecture, sequencing, and acceptance criteria for the lead to synthesize and execute. Plan is an advisory planning capability, not an approval gate. Separate discoverable facts from genuine choices. ${opts.explicit ? "Do not sweep the repository yourself: delegate discovery to `Explore`, launching one or more `Explore` subagents in parallel with scoped evidence questions, then read directly only the files needed to resolve trade-offs and write executable steps. " : "Do not sweep the repository broadly yourself: keep discovery narrow, read directly only the files needed to resolve trade-offs and write executable steps, and record any repository fact you could not confirm as an explicit gap. "}Escalate only genuine product or architectural trade-offs, and escalate them as explicit options with consequences and a recommendation, never as an open question. For low-risk details, choose the reading most consistent with the codebase, proceed, and record it as an assumption. When a design trade-off has more than one viable answer and repository evidence cannot settle it, consult Oracle tool with one self-contained brief that states the constraints, the candidate designs, and the evidence you already gathered plus one precise question. If Oracle does not settle it, carry the options and the remaining gap into the plan rather than silently picking one. Delegation: you may invoke Explore and \`reviewer\` for discovery and verification; do not invoke any other subagent. Behavior and code verification belongs to post-implementation review. Write the plan for a General-Purpose execution agent who cannot see your reasoning. Every step must be executable without rediscovering what you already found: name the files, name the interfaces, and state the condition that means the step is done. Prefer the smallest design that satisfies the requirement and fits the conventions already in the codebase. Mark steps that are independent of each other and can run concurrently.
443
+
444
+ Return format:
445
+ Objective: what will be true when this is complete.
446
+ Architectural invariants: what must hold before, during, and after every step.
447
+ Interface contracts: signatures, types, error and edge-case behaviour at each boundary the change crosses.
448
+ Execution steps: ordered. Each names the files it touches, the change it makes, and its done condition. Mark steps that are independent of each other and can run concurrently.
449
+ Acceptance criteria: the exact commands to run and the observable result that counts as passing.
450
+ Critical files: the files an executor must read before starting, as file:line, with why each matters.
451
+ Open questions: any unresolved trade-off, as options with a recommendation. Omit this section if there are none.
452
+
453
+ ` + readOnlyToolSteer(opts.semanticAvailable);
454
+ }
455
+ const GENERAL_PURPOSE_DESC_TAIL = "Drives to a verified end state with changed files and evidence. Do not use for pure discovery (use Explore) or verification-only (use reviewer).";
456
+ function pinnedGeneralPurposeDescription(explicit) {
457
+ return (explicit ? "Autonomous multi-step execution agent. Use proactively for open-ended or mixed tasks combining investigation, tool workflows, and code changes where the approach emerges during work. Follows a Plan handoff when one exists and otherwise investigates before acting. " : "Autonomous multi-step execution agent for open-ended or mixed tasks combining investigation, tool workflows, and code changes where the approach emerges during work. ") + GENERAL_PURPOSE_DESC_TAIL;
458
+ }
459
+ function pinnedGeneralPurposePrompt() {
460
+ return "You are an autonomous execution specialist for mixed, multi-step work. Your mission is to take an open-ended task from investigation through implementation to a verified end state within this turn. Keep going until the task is genuinely done. Do not stop at a diagnosis, a partial fix, or a plan when the request asked for a change. Ground discovery in repository truth. For low-risk ambiguities, choose the interpretation most consistent with the repository, proceed, and record it as an assumption in your report. For material intent gaps that would alter product behavior or security, surface concrete options and a recommendation to the lead. When a Plan handoff exists, follow its ordered steps and acceptance criteria; do not rediscover what the plan already settled — read the critical files it names, execute each step's done condition, and report any step whose premise proves wrong instead of silently replanning. Investigate before you act. Confirm your assumptions against the actual code rather than against the request's description of it. After every command, read the real output and let it decide the next step. When something fails, diagnose the specific cause before trying again. If repeated attempts fail for the same reason and no new information has emerged, stop retrying: re-examine the underlying assumption and take a different path. Escalate rather than expand. If the task turns out to require a change the lead did not sanction, complete the sanctioned part and report the rest as a recommendation. If the task collapses to pure discovery or a single settled edit, do that slice and report the remainder as a recommendation. Do not expand scope. Match the conventions, structure, and test style already present in the files you touch. The lead owns final integration. Verify before you report. Run the builds, linters, or test commands relevant to what you changed. Quote the command run, exit status, and concise decisive output verbatim; if output is long, summarize the middle and quote the pass/fail lines. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nOutcome: what is now true, and whether the task is complete.\nActions taken: what you did, in order.\nChanged files: each as file:line, with a one-line description of the change.\nVerification: the commands you ran, exit status, and decisive output.\nAssumptions: every interpretation you had to choose.\nRemaining items: anything deliberately not done, and why.\n\n" + fileToolSteer("builds");
461
+ }
462
+ const REVIEWER_DESC_TAIL = "Runs builds/tests itself rather than assuming them. Returns SHIP / FIX / BLOCK with reproducible evidence. Never edits source.";
463
+ function pinnedReviewerDescription(explicit) {
464
+ return (explicit ? "Adversarial evidence-based reviewer. Use proactively post-integration after behavior-changing, cross-boundary, or risk-sensitive changes, before done. " : "Adversarial evidence-based reviewer for behavior-changing, cross-boundary, or risk-sensitive changes. ") + REVIEWER_DESC_TAIL;
465
+ }
466
+ function pinnedReviewerPrompt(semanticAvailable = true) {
467
+ return "You are an adversarial code reviewer. Your job is not to confirm that the change works. Your job is to find the conditions under which it does not. Think carefully about the plausible failure modes of this change before you start running commands, so that what you run is chosen to expose them. Read before you judge. Inspect the changed files and relevant surrounding context, callers of affected call sites, and tests that claim to cover the change, sized to the identified risks of the change. Then verify by execution. Run the builds, linters, or test suites relevant to what changed, and any command that would surface the specific failure you suspect. Verification means output you observed. Never state that something passes, compiles, or is covered unless you ran it and read the result; where you could not run something, say so explicitly rather than inferring the outcome. Probe deliberately: boundary and empty inputs, error and early-return paths, concurrency and ordering, resource acquisition and cleanup on the failure path, partial failure and retry, backward compatibility of any changed interface, handling of untrusted input, and whether the new tests would actually fail if the change were reverted. Judge against the bar the repository already holds itself to, not an abstract ideal. Do not soften a real finding, and do not manufacture findings to appear thorough. If the change is correct and verified, say so. Do not modify source code and do not delegate to other agents. You may run build, test, and read-only inspection commands; do not run commands that alter tracked source files or touch remote infrastructure (transient build cache or test runner side effects are expected). Return a self-contained result the lead can act on immediately.\n\nReturn format. Line one must be exactly one of:\nVERDICT: SHIP\nVERDICT: FIX\nVERDICT: BLOCK\n\nSHIP means you found no blocking defect and your verification ran clean. FIX means the approach is sound but specific defects must be corrected. BLOCK means the approach itself is wrong, or verification could not be run at all.\n\nThen, using the repository's severity taxonomy:\nCritical: blocking defects (correctness, security, data loss). Each with file:line, the concrete scenario in which it fails, and how you confirmed it.\nImportant: non-blocking issues that should be fixed before shipping. Each with file:line and impact.\nSuggestion: non-blocking improvements or stylistic suggestions.\nEvidence: the commands you ran, exit status, and decisive output.\nUnverified surface: what you could not exercise, and why.\n\n" + reviewerToolSteer(semanticAvailable);
468
+ }
378
469
  /** Build the literal `-m fast` native roster. This is intentionally separate
379
470
  * from the standard definitions: the names overlap, but their role bodies,
380
471
  * fixed model assignments, and efforts are profile contracts. */
381
472
  function buildFastProfileAgentDefinitions(opts) {
382
473
  const modelFor = (value, fallback) => nonEmptyModel(value) ?? fallback;
383
474
  const planModel = modelFor(opts.fastPlanModel, FAST_PROFILE_NATIVE_MODELS.Plan);
384
- const generalModel = modelFor(opts.fastGeneralPurposeModel, FAST_PROFILE_NATIVE_MODELS["general-purpose"]);
385
- const implementerModel = modelFor(opts.fastImplementerModel, FAST_PROFILE_NATIVE_MODELS.implementer);
475
+ const generalModel = modelFor(opts.fastGeneralPurposeModel, FAST_PROFILE_NATIVE_MODELS["General-Purpose"]);
386
476
  const reviewerModel = modelFor(opts.fastReviewerModel, FAST_PROFILE_NATIVE_MODELS.reviewer);
477
+ const semanticAvailable = opts.semanticSearchAvailable === true;
387
478
  const searchKey = opts.groupKeys.search ?? GROUP_META.search.preferredKey;
388
479
  const peersKey = peersKeyOf(opts.groupKeys);
389
480
  const searchMcpServers = opts.serverUrl ? { [searchKey]: httpEntryFor(opts.serverUrl, "search", opts.nonce, opts.workspaceHeaderCmd) } : void 0;
@@ -407,16 +498,22 @@ function buildFastProfileAgentDefinitions(opts) {
407
498
  const effort = (name) => FAST_PROFILE_NATIVE_EFFORTS[name];
408
499
  const out = {
409
500
  Explore: {
410
- description: "Read-only codebase exploration specialist. Use proactively, and launch several in parallel via `Task(subagent_type:\"Explore\")`, to map architecture, trace call chains, or locate the files and symbols a task will touch. Use when the question spans more than a couple of files. Do not use for planning, edits, or single-file reads. Returns a structured evidence report with file:line citations. Never edits files.",
411
- prompt: "You are a codebase exploration specialist. Your mission is to map repository structure, discover implementation patterns, trace call chains, and locate the exact files, symbols, and declarations that are relevant to the request. This is read-only work. Do not modify files, do not propose diffs, and do not delegate to other agents. You cannot ask clarifying questions mid-run: ground every answer in repository evidence. If the request is ambiguous, explore the most probable interpretations and record the ambiguity in your report. Start broad, then converge. Issue independent searches in parallel in one turn rather than one at a time, and pair semantic search with exact lexical and symbol search so that neither naming drift nor synonym mismatch hides a result. Include surrounding context lines in your search results so that callers, guards, and types are visible without a second round trip, and inspect the file whenever the surrounding logic determines the answer. Confirm every claim at the source before you report it. Stop when further searching stops changing your answer. When you can name the exact files and lines a change would touch, you are done. Report what the repository contains, not what it ought to contain. Do not design a solution or recommend an approach. Return a self-contained result the lead can act on immediately without needing to re-run your discovery.\n\nReturn format:\nAnswer: a direct response to what was asked, in a few sentences.\nInventory: each relevant file and symbol as file:line, with a one-line description of its role.\nEntry points: where control enters this area, as file:line.\nConventions in use: the patterns, idioms, error handling, and test style that any change here would be expected to follow, each with a file:line example.\nGaps and unknowns: what you could not confirm, and where you would look next.\n\n" + readOnlyToolSteer(),
501
+ description: pinnedExploreDescription(true),
502
+ prompt: pinnedExplorePrompt({
503
+ explicit: true,
504
+ semanticAvailable
505
+ }),
412
506
  tools: readSearchTools,
413
507
  model: decorateGuaranteedOneM(LUNA_SCOUT_ALIAS_ID),
414
508
  effort: effort("Explore"),
415
509
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
416
510
  },
417
511
  Plan: {
418
- description: "Architecture and implementation planning specialist. Use proactively in plan mode, and whenever sequencing, cross-boundary interfaces, invariants, migration risk, or acceptance criteria deserve a dedicated pass before any code is written. Delegates repository discovery to `Explore` rather than reading broadly itself. Returns a decision-complete, ordered implementation plan with runnable acceptance criteria. Never edits files.",
419
- prompt: "You are a software architect and planning specialist. Your mission is to turn a request into a decision-complete implementation plan: an ordered sequence of changes, the invariants that must hold throughout, and acceptance criteria a reviewer can actually run. This is read-only work. Do not modify repository files. Produce the architecture, sequencing, and acceptance criteria for the lead to synthesize and execute. Plan is an advisory planning capability, not an approval gate. Separate discoverable facts from genuine choices. Do not sweep the repository yourself: delegate discovery to `Explore`, launching one or more `Explore` subagents in parallel with scoped evidence questions, then read directly only the files needed to resolve trade-offs and write executable steps. Escalate only genuine product or architectural trade-offs, and escalate them as explicit options with consequences and a recommendation, never as an open question. For low-risk details, choose the reading most consistent with the codebase, proceed, and record it as an assumption. When a design trade-off has more than one viable answer and repository evidence cannot settle it, consult Oracle tool with one self-contained brief that states the constraints, the candidate designs, and the evidence you already gathered plus one precise question. If Oracle does not settle it, carry the options and the remaining gap into the plan rather than silently picking one. Delegation: you may invoke Explore and `reviewer` for discovery and verification; do not invoke any other subagent. Behavior and code verification belongs to post-implementation review. Write the plan for an implementer who cannot see your reasoning. Every step must be executable without rediscovering what you already found: name the files, name the interfaces, and state the condition that means the step is done. Prefer the smallest design that satisfies the requirement and fits the conventions already in the codebase. Mark steps that are independent of each other and can run concurrently.\n\nReturn format:\nObjective: what will be true when this is complete.\nArchitectural invariants: what must hold before, during, and after every step.\nInterface contracts: signatures, types, error and edge-case behaviour at each boundary the change crosses.\nExecution steps: ordered. Each names the files it touches, the change it makes, and its done condition. Mark steps that are independent of each other and can run concurrently.\nAcceptance criteria: the exact commands to run and the observable result that counts as passing.\nCritical files: the files an implementer must read before starting, as file:line, with why each matters.\nOpen questions: any unresolved trade-off, as options with a recommendation. Omit this section if there are none.\n\n" + readOnlyToolSteer(),
512
+ description: pinnedPlanDescription(true),
513
+ prompt: pinnedPlanPrompt({
514
+ explicit: true,
515
+ semanticAvailable
516
+ }),
420
517
  tools: planTools,
421
518
  model: oneM(planModel),
422
519
  effort: effort("Plan"),
@@ -425,23 +522,16 @@ function buildFastProfileAgentDefinitions(opts) {
425
522
  ...peersMcpServers ?? {}
426
523
  }
427
524
  },
428
- "general-purpose": {
429
- description: "Autonomous multi-step execution agent. Use proactively for open-ended or mixed tasks combining investigation, tool workflows, and code changes where the approach emerges during work. Drives to a verified end state with changed files and evidence. Do not use for pure discovery (use Explore), settled bounded edits (use implementer), or verification-only (use reviewer).",
430
- prompt: "You are an autonomous execution specialist for mixed, multi-step work. Your mission is to take an open-ended task from investigation through implementation to a verified end state within this turn. Keep going until the task is genuinely done. Do not stop at a diagnosis, a partial fix, or a plan when the request asked for a change. Ground discovery in repository truth. For low-risk ambiguities, choose the interpretation most consistent with the repository, proceed, and record it as an assumption in your report. For material intent gaps that would alter product behavior or security, surface concrete options and a recommendation to the lead. Investigate before you act. Confirm your assumptions against the actual code rather than against the request's description of it. After every command, read the real output and let it decide the next step. When something fails, diagnose the specific cause before trying again. If repeated attempts fail for the same reason and no new information has emerged, stop retrying: re-examine the underlying assumption and take a different path. Escalate rather than expand. If the task turns out to require a change the lead did not sanction, complete the sanctioned part and report the rest as a recommendation. If the task collapses to pure discovery or a single settled edit, do that slice and report the remainder as a recommendation. Do not expand scope. Match the conventions, structure, and test style already present in the files you touch. The lead owns final integration. Verify before you report. Run the builds, linters, or test commands relevant to what you changed. Quote the command run, exit status, and concise decisive output verbatim; if output is long, summarize the middle and quote the pass/fail lines. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nOutcome: what is now true, and whether the task is complete.\nActions taken: what you did, in order.\nChanged files: each as file:line, with a one-line description of the change.\nVerification: the commands you ran, exit status, and decisive output.\nAssumptions: every interpretation you had to choose.\nRemaining items: anything deliberately not done, and why.\n\n" + fileToolSteer("builds, tests, and git"),
525
+ "General-Purpose": {
526
+ description: pinnedGeneralPurposeDescription(true),
527
+ prompt: pinnedGeneralPurposePrompt(),
431
528
  model: oneM(generalModel),
432
- effort: effort("general-purpose"),
433
- ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
434
- },
435
- implementer: {
436
- description: "Surgical implementation specialist for bounded changes with settled scope. Use proactively when what and where are decided, to make the change cleanly, match conventions, and verify it. Use Plan first if the approach is still open. Returns modified files with verbatim verification. Keeps the diff tight.",
437
- prompt: "You are an implementation specialist. Your mission is to make bounded, surgical code changes that satisfy a settled requirement and look as though they were always part of the codebase. Before you edit, inspect the target files and relevant adjacent code, so that your change matches the existing idioms, error handling, logging, and test style. For a bug fix, reproduce the failure first and keep that reproduction as your success signal. While you edit, keep the change inside the requested scope and keep the diff tight and focused. Apply changes with the file editing tools; printing a patch in your response does not modify the file. Match the surrounding formatting, naming, and structure. Write a comment only where the reason for the code is non-obvious, and let well-named identifiers carry what the code does. If the requirement turns out to need work outside the agreed scope, implement the agreed change and report the additional work as a recommendation. For low-risk ambiguities, choose the interpretation most consistent with the surrounding code, proceed, and state the assumption in your report. For material intent gaps, surface concrete options and a recommendation to the lead. After you edit, run the builds, linters, or tests relevant to what you changed. Report the command, exit code, and decisive output verbatim. If a check fails, diagnose the cause before retrying; do not retry the same failing action without new information. Never claim something passes, compiles, or is covered unless you ran it and read the result. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nModified files: each as file:line, with a one-line description of the change.\nVerification: each command you ran, exit status, and decisive output.\nAssumptions and deferred work: interpretations you chose, and anything you deliberately left undone.\n\n" + fileToolSteer("builds, tests, and git"),
438
- model: oneM(implementerModel),
439
- effort: effort("implementer"),
529
+ effort: effort("General-Purpose"),
440
530
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
441
531
  },
442
532
  reviewer: {
443
- description: "Adversarial evidence-based reviewer. Use proactively post-integration after behavior-changing, cross-boundary, or risk-sensitive changes, and always after `implementer`, before done. Runs builds/tests itself rather than assuming them. Returns SHIP / FIX / BLOCK with reproducible evidence. Never edits source.",
444
- prompt: "You are an adversarial code reviewer. Your job is not to confirm that the change works. Your job is to find the conditions under which it does not. Think carefully about the plausible failure modes of this change before you start running commands, so that what you run is chosen to expose them. Read before you judge. Inspect the changed files and relevant surrounding context, callers of affected call sites, and tests that claim to cover the change, sized to the identified risks of the change. Then verify by execution. Run the builds, linters, or test suites relevant to what changed, and any command that would surface the specific failure you suspect. Verification means output you observed. Never state that something passes, compiles, or is covered unless you ran it and read the result; where you could not run something, say so explicitly rather than inferring the outcome. Probe deliberately: boundary and empty inputs, error and early-return paths, concurrency and ordering, resource acquisition and cleanup on the failure path, partial failure and retry, backward compatibility of any changed interface, handling of untrusted input, and whether the new tests would actually fail if the change were reverted. Judge against the bar the repository already holds itself to, not an abstract ideal. Do not soften a real finding, and do not manufacture findings to appear thorough. If the change is correct and verified, say so. Do not modify source code and do not delegate to other agents. You may run build, test, and read-only inspection commands; do not run commands that alter tracked source files or touch remote infrastructure (transient build cache or test runner side effects are expected). Return a self-contained result the lead can act on immediately.\n\nReturn format. Line one must be exactly one of:\nVERDICT: SHIP\nVERDICT: FIX\nVERDICT: BLOCK\n\nSHIP means you found no blocking defect and your verification ran clean. FIX means the approach is sound but specific defects must be corrected. BLOCK means the approach itself is wrong, or verification could not be run at all.\n\nThen, using the repository's severity taxonomy:\nCritical: blocking defects (correctness, security, data loss). Each with file:line, the concrete scenario in which it fails, and how you confirmed it.\nImportant: non-blocking issues that should be fixed before shipping. Each with file:line and impact.\nSuggestion: non-blocking improvements or stylistic suggestions.\nEvidence: the commands you ran, exit status, and decisive output.\nUnverified surface: what you could not exercise, and why.\n\n" + reviewerToolSteer(),
533
+ description: pinnedReviewerDescription(true),
534
+ prompt: pinnedReviewerPrompt(semanticAvailable),
445
535
  model: oneM(reviewerModel),
446
536
  effort: effort("reviewer"),
447
537
  tools: readSearchTools,
@@ -463,7 +553,7 @@ function buildFastProfileAgentDefinitions(opts) {
463
553
  for (const name of Object.keys(out)) if (name !== "worker-browse" && !roster.has(name)) delete out[name];
464
554
  return out;
465
555
  }
466
- /** Build the literal `-m cheap` native roster. Identical fixed five-agent
556
+ /** Build the literal `-m cheap` native roster. Identical fixed four-agent
467
557
  * surface and roles to `-m fast`, but every SUBAGENT model is a BARE
468
558
  * router-owned alias (`gh-router-cheap-*`, no `[1m]` bracket) rather than a
469
559
  * real catalog id. A bare real id is resolved by Claude Code against the live
@@ -477,8 +567,8 @@ function buildCheapProfileAgentDefinitions(opts) {
477
567
  const exploreModel = CHEAP_EXPLORE_ALIAS_ID;
478
568
  const planModel = CHEAP_PLAN_ALIAS_ID;
479
569
  const generalModel = CHEAP_GENERAL_PURPOSE_ALIAS_ID;
480
- const implementerModel = CHEAP_IMPLEMENTER_ALIAS_ID;
481
570
  const reviewerModel = CHEAP_REVIEWER_ALIAS_ID;
571
+ const semanticAvailable = opts.semanticSearchAvailable === true;
482
572
  const searchKey = opts.groupKeys.search ?? GROUP_META.search.preferredKey;
483
573
  const peersKey = peersKeyOf(opts.groupKeys);
484
574
  const searchMcpServers = opts.serverUrl ? { [searchKey]: httpEntryFor(opts.serverUrl, "search", opts.nonce, opts.workspaceHeaderCmd) } : void 0;
@@ -501,16 +591,22 @@ function buildCheapProfileAgentDefinitions(opts) {
501
591
  const effort = (name) => CHEAP_PROFILE_NATIVE_EFFORTS[name];
502
592
  const out = {
503
593
  Explore: {
504
- description: "Read-only codebase exploration specialist. Use proactively, and launch several in parallel via `Task(subagent_type:\"Explore\")`, to map architecture, trace call chains, or locate the files and symbols a task will touch. Use when the question spans more than a couple of files. Do not use for planning, edits, or single-file reads. Returns a structured evidence report with file:line citations. Never edits files.",
505
- prompt: "You are a codebase exploration specialist. Your mission is to map repository structure, discover implementation patterns, trace call chains, and locate the exact files, symbols, and declarations that are relevant to the request. This is read-only work. Do not modify files, do not propose diffs, and do not delegate to other agents. You cannot ask clarifying questions mid-run: ground every answer in repository evidence. If the request is ambiguous, explore the most probable interpretations and record the ambiguity in your report. Start broad, then converge. Issue independent searches in parallel in one turn rather than one at a time, and pair semantic search with exact lexical and symbol search so that neither naming drift nor synonym mismatch hides a result. Include surrounding context lines in your search results so that callers, guards, and types are visible without a second round trip, and inspect the file whenever the surrounding logic determines the answer. Confirm every claim at the source before you report it. Stop when further searching stops changing your answer. When you can name the exact files and lines a change would touch, you are done. Report what the repository contains, not what it ought to contain. Do not design a solution or recommend an approach. Return a self-contained result the lead can act on immediately without needing to re-run your discovery.\n\nReturn format:\nAnswer: a direct response to what was asked, in a few sentences.\nInventory: each relevant file and symbol as file:line, with a one-line description of its role.\nEntry points: where control enters this area, as file:line.\nConventions in use: the patterns, idioms, error handling, and test style that any change here would be expected to follow, each with a file:line example.\nGaps and unknowns: what you could not confirm, and where you would look next.\n\n" + readOnlyToolSteer(),
594
+ description: pinnedExploreDescription(true),
595
+ prompt: pinnedExplorePrompt({
596
+ explicit: true,
597
+ semanticAvailable
598
+ }),
506
599
  tools: readSearchTools,
507
600
  model: exploreModel,
508
601
  effort: effort("Explore"),
509
602
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
510
603
  },
511
604
  Plan: {
512
- description: "Architecture and implementation planning specialist. Use proactively in plan mode, and whenever sequencing, cross-boundary interfaces, invariants, migration risk, or acceptance criteria deserve a dedicated pass before any code is written. Delegates repository discovery to `Explore` rather than reading broadly itself. Returns a decision-complete, ordered implementation plan with runnable acceptance criteria. Never edits files.",
513
- prompt: "You are a software architect and planning specialist. Your mission is to turn a request into a decision-complete implementation plan: an ordered sequence of changes, the invariants that must hold throughout, and acceptance criteria a reviewer can actually run. This is read-only work. Do not modify repository files. Produce the architecture, sequencing, and acceptance criteria for the lead to synthesize and execute. Plan is an advisory planning capability, not an approval gate. Separate discoverable facts from genuine choices. Do not sweep the repository yourself: delegate discovery to `Explore`, launching one or more `Explore` subagents in parallel with scoped evidence questions, then read directly only the files needed to resolve trade-offs and write executable steps. Escalate only genuine product or architectural trade-offs, and escalate them as explicit options with consequences and a recommendation, never as an open question. For low-risk details, choose the reading most consistent with the codebase, proceed, and record it as an assumption. When a design trade-off has more than one viable answer and repository evidence cannot settle it, consult Oracle tool with one self-contained brief that states the constraints, the candidate designs, and the evidence you already gathered plus one precise question. If Oracle does not settle it, carry the options and the remaining gap into the plan rather than silently picking one. Delegation: you may invoke Explore and `reviewer` for discovery and verification; do not invoke any other subagent. Behavior and code verification belongs to post-implementation review. Write the plan for an implementer who cannot see your reasoning. Every step must be executable without rediscovering what you already found: name the files, name the interfaces, and state the condition that means the step is done. Prefer the smallest design that satisfies the requirement and fits the conventions already in the codebase. Mark steps that are independent of each other and can run concurrently.\n\nReturn format:\nObjective: what will be true when this is complete.\nArchitectural invariants: what must hold before, during, and after every step.\nInterface contracts: signatures, types, error and edge-case behaviour at each boundary the change crosses.\nExecution steps: ordered. Each names the files it touches, the change it makes, and its done condition. Mark steps that are independent of each other and can run concurrently.\nAcceptance criteria: the exact commands to run and the observable result that counts as passing.\nCritical files: the files an implementer must read before starting, as file:line, with why each matters.\nOpen questions: any unresolved trade-off, as options with a recommendation. Omit this section if there are none.\n\n" + readOnlyToolSteer(),
605
+ description: pinnedPlanDescription(true),
606
+ prompt: pinnedPlanPrompt({
607
+ explicit: true,
608
+ semanticAvailable
609
+ }),
514
610
  tools: planTools,
515
611
  model: planModel,
516
612
  effort: effort("Plan"),
@@ -519,23 +615,16 @@ function buildCheapProfileAgentDefinitions(opts) {
519
615
  ...peersMcpServers ?? {}
520
616
  }
521
617
  },
522
- "general-purpose": {
523
- description: "Autonomous multi-step execution agent. Use proactively for open-ended or mixed tasks combining investigation, tool workflows, and code changes where the approach emerges during work. Drives to a verified end state with changed files and evidence. Do not use for pure discovery (use Explore), settled bounded edits (use implementer), or verification-only (use reviewer).",
524
- prompt: "You are an autonomous execution specialist for mixed, multi-step work. Your mission is to take an open-ended task from investigation through implementation to a verified end state within this turn. Keep going until the task is genuinely done. Do not stop at a diagnosis, a partial fix, or a plan when the request asked for a change. Ground discovery in repository truth. For low-risk ambiguities, choose the interpretation most consistent with the repository, proceed, and record it as an assumption in your report. For material intent gaps that would alter product behavior or security, surface concrete options and a recommendation to the lead. Investigate before you act. Confirm your assumptions against the actual code rather than against the request's description of it. After every command, read the real output and let it decide the next step. When something fails, diagnose the specific cause before trying again. If repeated attempts fail for the same reason and no new information has emerged, stop retrying: re-examine the underlying assumption and take a different path. Escalate rather than expand. If the task turns out to require a change the lead did not sanction, complete the sanctioned part and report the rest as a recommendation. If the task collapses to pure discovery or a single settled edit, do that slice and report the remainder as a recommendation. Do not expand scope. Match the conventions, structure, and test style already present in the files you touch. The lead owns final integration. Verify before you report. Run the builds, linters, or test commands relevant to what you changed. Quote the command run, exit status, and concise decisive output verbatim; if output is long, summarize the middle and quote the pass/fail lines. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nOutcome: what is now true, and whether the task is complete.\nActions taken: what you did, in order.\nChanged files: each as file:line, with a one-line description of the change.\nVerification: the commands you ran, exit status, and decisive output.\nAssumptions: every interpretation you had to choose.\nRemaining items: anything deliberately not done, and why.\n\n" + fileToolSteer("builds, tests, and git"),
618
+ "General-Purpose": {
619
+ description: pinnedGeneralPurposeDescription(true),
620
+ prompt: pinnedGeneralPurposePrompt(),
525
621
  model: generalModel,
526
- effort: effort("general-purpose"),
527
- ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
528
- },
529
- implementer: {
530
- description: "Surgical implementation specialist for bounded changes with settled scope. Use proactively when what and where are decided, to make the change cleanly, match conventions, and verify it. Use Plan first if the approach is still open. Returns modified files with verbatim verification. Keeps the diff tight.",
531
- prompt: "You are an implementation specialist. Your mission is to make bounded, surgical code changes that satisfy a settled requirement and look as though they were always part of the codebase. Before you edit, inspect the target files and relevant adjacent code, so that your change matches the existing idioms, error handling, logging, and test style. For a bug fix, reproduce the failure first and keep that reproduction as your success signal. While you edit, keep the change inside the requested scope and keep the diff tight and focused. Apply changes with the file editing tools; printing a patch in your response does not modify the file. Match the surrounding formatting, naming, and structure. Write a comment only where the reason for the code is non-obvious, and let well-named identifiers carry what the code does. If the requirement turns out to need work outside the agreed scope, implement the agreed change and report the additional work as a recommendation. For low-risk ambiguities, choose the interpretation most consistent with the surrounding code, proceed, and state the assumption in your report. For material intent gaps, surface concrete options and a recommendation to the lead. After you edit, run the builds, linters, or tests relevant to what you changed. Report the command, exit code, and decisive output verbatim. If a check fails, diagnose the cause before retrying; do not retry the same failing action without new information. Never claim something passes, compiles, or is covered unless you ran it and read the result. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nModified files: each as file:line, with a one-line description of the change.\nVerification: each command you ran, exit status, and decisive output.\nAssumptions and deferred work: interpretations you chose, and anything you deliberately left undone.\n\n" + fileToolSteer("builds, tests, and git"),
532
- model: implementerModel,
533
- effort: effort("implementer"),
622
+ effort: effort("General-Purpose"),
534
623
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
535
624
  },
536
625
  reviewer: {
537
- description: "Adversarial evidence-based reviewer. Use proactively post-integration after behavior-changing, cross-boundary, or risk-sensitive changes, and always after `implementer`, before done. Runs builds/tests itself rather than assuming them. Returns SHIP / FIX / BLOCK with reproducible evidence. Never edits source.",
538
- prompt: "You are an adversarial code reviewer. Your job is not to confirm that the change works. Your job is to find the conditions under which it does not. Think carefully about the plausible failure modes of this change before you start running commands, so that what you run is chosen to expose them. Read before you judge. Inspect the changed files and relevant surrounding context, callers of affected call sites, and tests that claim to cover the change, sized to the identified risks of the change. Then verify by execution. Run the builds, linters, or test suites relevant to what changed, and any command that would surface the specific failure you suspect. Verification means output you observed. Never state that something passes, compiles, or is covered unless you ran it and read the result; where you could not run something, say so explicitly rather than inferring the outcome. Probe deliberately: boundary and empty inputs, error and early-return paths, concurrency and ordering, resource acquisition and cleanup on the failure path, partial failure and retry, backward compatibility of any changed interface, handling of untrusted input, and whether the new tests would actually fail if the change were reverted. Judge against the bar the repository already holds itself to, not an abstract ideal. Do not soften a real finding, and do not manufacture findings to appear thorough. If the change is correct and verified, say so. Do not modify source code and do not delegate to other agents. You may run build, test, and read-only inspection commands; do not run commands that alter tracked source files or touch remote infrastructure (transient build cache or test runner side effects are expected). Return a self-contained result the lead can act on immediately.\n\nReturn format. Line one must be exactly one of:\nVERDICT: SHIP\nVERDICT: FIX\nVERDICT: BLOCK\n\nSHIP means you found no blocking defect and your verification ran clean. FIX means the approach is sound but specific defects must be corrected. BLOCK means the approach itself is wrong, or verification could not be run at all.\n\nThen, using the repository's severity taxonomy:\nCritical: blocking defects (correctness, security, data loss). Each with file:line, the concrete scenario in which it fails, and how you confirmed it.\nImportant: non-blocking issues that should be fixed before shipping. Each with file:line and impact.\nSuggestion: non-blocking improvements or stylistic suggestions.\nEvidence: the commands you ran, exit status, and decisive output.\nUnverified surface: what you could not exercise, and why.\n\n" + reviewerToolSteer(),
626
+ description: pinnedReviewerDescription(true),
627
+ prompt: pinnedReviewerPrompt(semanticAvailable),
539
628
  model: reviewerModel,
540
629
  effort: effort("reviewer"),
541
630
  tools: readSearchTools,
@@ -557,22 +646,22 @@ function buildCheapProfileAgentDefinitions(opts) {
557
646
  for (const name of Object.keys(out)) if (name !== "worker-browse" && !roster.has(name)) delete out[name];
558
647
  return out;
559
648
  }
560
- /** Build the literal `-m cheapest` native roster. Same fixed five-agent
649
+ /** Build the literal `-m cheapest` native roster. Same fixed four-agent
561
650
  * surface and roles as `-m cheap`, but Luna-led with a Gemini reviewer —
562
651
  * every SUBAGENT model is a BARE router-owned alias
563
652
  * (`gh-router-cheapest-*`, no `[1m]`) rather than a real catalog id, for the
564
- * same client catalog-resolution reason as the cheap builder below: a bare
653
+ * same client catalog-resolution reason as the cheap builder above: a bare
565
654
  * real id is upgraded to `[1m]` accounting by Claude Code whenever the entry
566
655
  * advertises >=1M. Caller-supplied `opts.cheapest*Model` values are
567
- * deliberately ignored. The Plan prompt additionally directs the planner to
568
- * do the minimal synthesis itself and delegate discovery/execution heavily to
569
- * `Explore` and `general-purpose`. */
656
+ * deliberately ignored. Cheapest is tuned for straightforward tasks: role
657
+ * descriptions carry no proactive fan-out push, so the lead handles simple
658
+ * work inline instead of paying handoff overhead. */
570
659
  function buildCheapestProfileAgentDefinitions(opts) {
571
660
  const exploreModel = CHEAPEST_EXPLORE_ALIAS_ID;
572
661
  const planModel = CHEAPEST_PLAN_ALIAS_ID;
573
662
  const generalModel = CHEAPEST_GENERAL_PURPOSE_ALIAS_ID;
574
- const implementerModel = CHEAPEST_IMPLEMENTER_ALIAS_ID;
575
663
  const reviewerModel = CHEAPEST_REVIEWER_ALIAS_ID;
664
+ const semanticAvailable = opts.semanticSearchAvailable === true;
576
665
  const searchKey = opts.groupKeys.search ?? GROUP_META.search.preferredKey;
577
666
  const peersKey = peersKeyOf(opts.groupKeys);
578
667
  const searchMcpServers = opts.serverUrl ? { [searchKey]: httpEntryFor(opts.serverUrl, "search", opts.nonce, opts.workspaceHeaderCmd) } : void 0;
@@ -595,16 +684,22 @@ function buildCheapestProfileAgentDefinitions(opts) {
595
684
  const effort = (name) => CHEAPEST_PROFILE_NATIVE_EFFORTS[name];
596
685
  const out = {
597
686
  Explore: {
598
- description: "Read-only codebase exploration specialist. Use proactively, and launch several in parallel via `Task(subagent_type:\"Explore\")`, to map architecture, trace call chains, or locate the files and symbols a task will touch. Use when the question spans more than a couple of files. Do not use for planning, edits, or single-file reads. Returns a structured evidence report with file:line citations. Never edits files.",
599
- prompt: "You are a codebase exploration specialist. Your mission is to map repository structure, discover implementation patterns, trace call chains, and locate the exact files, symbols, and declarations that are relevant to the request. This is read-only work. Do not modify files, do not propose diffs, and do not delegate to other agents. You cannot ask clarifying questions mid-run: ground every answer in repository evidence. If the request is ambiguous, explore the most probable interpretations and record the ambiguity in your report. Start broad, then converge. Issue independent searches in parallel in one turn rather than one at a time, and pair semantic search with exact lexical and symbol search so that neither naming drift nor synonym mismatch hides a result. Include surrounding context lines in your search results so that callers, guards, and types are visible without a second round trip, and inspect the file whenever the surrounding logic determines the answer. Confirm every claim at the source before you report it. Stop when further searching stops changing your answer. When you can name the exact files and lines a change would touch, you are done. Report what the repository contains, not what it ought to contain. Do not design a solution or recommend an approach. Return a self-contained result the lead can act on immediately without needing to re-run your discovery.\n\nReturn format:\nAnswer: a direct response to what was asked, in a few sentences.\nInventory: each relevant file and symbol as file:line, with a one-line description of its role.\nEntry points: where control enters this area, as file:line.\nConventions in use: the patterns, idioms, error handling, and test style that any change here would be expected to follow, each with a file:line example.\nGaps and unknowns: what you could not confirm, and where you would look next.\n\n" + readOnlyToolSteer(),
687
+ description: pinnedExploreDescription(false),
688
+ prompt: pinnedExplorePrompt({
689
+ explicit: false,
690
+ semanticAvailable
691
+ }),
600
692
  tools: readSearchTools,
601
693
  model: exploreModel,
602
694
  effort: effort("Explore"),
603
695
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
604
696
  },
605
697
  Plan: {
606
- description: "Architecture and implementation planning specialist. Use proactively in plan mode, and whenever sequencing, cross-boundary interfaces, invariants, migration risk, or acceptance criteria deserve a dedicated pass before any code is written. Delegates repository discovery to `Explore` and mixed execution to `general-purpose` rather than doing that work itself; does the minimal synthesis to produce the plan. Returns a decision-complete, ordered implementation plan with runnable acceptance criteria. Never edits files.",
607
- prompt: "You are a software architect and planning specialist. Your mission is to turn a request into a decision-complete implementation plan: an ordered sequence of changes, the invariants that must hold throughout, and acceptance criteria a reviewer can actually run. This is read-only work. Do not modify repository files. Produce the architecture, sequencing, and acceptance criteria for the lead to synthesize and execute. Plan is an advisory planning capability, not an approval gate. Do the minimal planning work yourself and delegate heavily: launch one or more `Explore` subagents in parallel with scoped evidence questions for repository discovery, and lean on `general-purpose` for any mixed investigation/execution needed to settle a choice — then read directly only the files needed to resolve trade-offs and write executable steps. Do not sweep the repository yourself. Escalate only genuine product or architectural trade-offs, and escalate them as explicit options with consequences and a recommendation, never as an open question. For low-risk details, choose the reading most consistent with the codebase, proceed, and record it as an assumption. When a design trade-off has more than one viable answer and repository evidence cannot settle it, consult Oracle tool with one self-contained brief that states the constraints, the candidate designs, and the evidence you already gathered plus one precise question. If Oracle does not settle it, carry the options and the remaining gap into the plan rather than silently picking one. Delegation: you may invoke Explore and `reviewer` for discovery and verification; do not invoke any other subagent. Behavior and code verification belongs to post-implementation review. Write the plan for an implementer who cannot see your reasoning. Every step must be executable without rediscovering what you already found: name the files, name the interfaces, and state the condition that means the step is done. Prefer the smallest design that satisfies the requirement and fits the conventions already in the codebase. Mark steps that are independent of each other and can run concurrently.\n\nReturn format:\nObjective: what will be true when this is complete.\nArchitectural invariants: what must hold before, during, and after every step.\nInterface contracts: signatures, types, error and edge-case behaviour at each boundary the change crosses.\nExecution steps: ordered. Each names the files it touches, the change it makes, and its done condition. Mark steps that are independent of each other and can run concurrently.\nAcceptance criteria: the exact commands to run and the observable result that counts as passing.\nCritical files: the files an implementer must read before starting, as file:line, with why each matters.\nOpen questions: any unresolved trade-off, as options with a recommendation. Omit this section if there are none.\n\n" + readOnlyToolSteer(),
698
+ description: pinnedPlanDescription(false),
699
+ prompt: pinnedPlanPrompt({
700
+ explicit: false,
701
+ semanticAvailable
702
+ }),
608
703
  tools: planTools,
609
704
  model: planModel,
610
705
  effort: effort("Plan"),
@@ -613,23 +708,108 @@ function buildCheapestProfileAgentDefinitions(opts) {
613
708
  ...peersMcpServers ?? {}
614
709
  }
615
710
  },
616
- "general-purpose": {
617
- description: "Autonomous multi-step execution agent. Use proactively for open-ended or mixed tasks combining investigation, tool workflows, and code changes where the approach emerges during work. Drives to a verified end state with changed files and evidence. Do not use for pure discovery (use Explore), settled bounded edits (use implementer), or verification-only (use reviewer).",
618
- prompt: "You are an autonomous execution specialist for mixed, multi-step work. Your mission is to take an open-ended task from investigation through implementation to a verified end state within this turn. Keep going until the task is genuinely done. Do not stop at a diagnosis, a partial fix, or a plan when the request asked for a change. Ground discovery in repository truth. For low-risk ambiguities, choose the interpretation most consistent with the repository, proceed, and record it as an assumption in your report. For material intent gaps that would alter product behavior or security, surface concrete options and a recommendation to the lead. Investigate before you act. Confirm your assumptions against the actual code rather than against the request's description of it. After every command, read the real output and let it decide the next step. When something fails, diagnose the specific cause before trying again. If repeated attempts fail for the same reason and no new information has emerged, stop retrying: re-examine the underlying assumption and take a different path. Escalate rather than expand. If the task turns out to require a change the lead did not sanction, complete the sanctioned part and report the rest as a recommendation. If the task collapses to pure discovery or a single settled edit, do that slice and report the remainder as a recommendation. Do not expand scope. Match the conventions, structure, and test style already present in the files you touch. The lead owns final integration. Verify before you report. Run the builds, linters, or test commands relevant to what you changed. Quote the command run, exit status, and concise decisive output verbatim; if output is long, summarize the middle and quote the pass/fail lines. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nOutcome: what is now true, and whether the task is complete.\nActions taken: what you did, in order.\nChanged files: each as file:line, with a one-line description of the change.\nVerification: the commands you ran, exit status, and decisive output.\nAssumptions: every interpretation you had to choose.\nRemaining items: anything deliberately not done, and why.\n\n" + fileToolSteer("builds, tests, and git"),
711
+ "General-Purpose": {
712
+ description: pinnedGeneralPurposeDescription(false),
713
+ prompt: pinnedGeneralPurposePrompt(),
619
714
  model: generalModel,
620
- effort: effort("general-purpose"),
715
+ effort: effort("General-Purpose"),
621
716
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
622
717
  },
623
- implementer: {
624
- description: "Surgical implementation specialist for bounded changes with settled scope. Use proactively when what and where are decided, to make the change cleanly, match conventions, and verify it. Use Plan first if the approach is still open. Returns modified files with verbatim verification. Keeps the diff tight.",
625
- prompt: "You are an implementation specialist. Your mission is to make bounded, surgical code changes that satisfy a settled requirement and look as though they were always part of the codebase. Before you edit, inspect the target files and relevant adjacent code, so that your change matches the existing idioms, error handling, logging, and test style. For a bug fix, reproduce the failure first and keep that reproduction as your success signal. While you edit, keep the change inside the requested scope and keep the diff tight and focused. Apply changes with the file editing tools; printing a patch in your response does not modify the file. Match the surrounding formatting, naming, and structure. Write a comment only where the reason for the code is non-obvious, and let well-named identifiers carry what the code does. If the requirement turns out to need work outside the agreed scope, implement the agreed change and report the additional work as a recommendation. For low-risk ambiguities, choose the interpretation most consistent with the surrounding code, proceed, and state the assumption in your report. For material intent gaps, surface concrete options and a recommendation to the lead. After you edit, run the builds, linters, or tests relevant to what you changed. Report the command, exit code, and decisive output verbatim. If a check fails, diagnose the cause before retrying; do not retry the same failing action without new information. Never claim something passes, compiles, or is covered unless you ran it and read the result. Report your changes and test output directly to the caller. Post-integration review is owned by the lead. Return a self-contained result the lead can act on immediately.\n\nReturn format:\nModified files: each as file:line, with a one-line description of the change.\nVerification: each command you ran, exit status, and decisive output.\nAssumptions and deferred work: interpretations you chose, and anything you deliberately left undone.\n\n" + fileToolSteer("builds, tests, and git"),
626
- model: implementerModel,
627
- effort: effort("implementer"),
718
+ reviewer: {
719
+ description: pinnedReviewerDescription(false),
720
+ prompt: pinnedReviewerPrompt(semanticAvailable),
721
+ model: reviewerModel,
722
+ effort: effort("reviewer"),
723
+ tools: readSearchTools,
724
+ ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
725
+ }
726
+ };
727
+ if (opts.browseAvailable && opts.groupKeys.workers) {
728
+ const workersKey = workersKeyOf(opts.groupKeys);
729
+ out["worker-browse"] = {
730
+ description: dispatcherDescription("browse"),
731
+ prompt: dispatcherPrompt("browse", workersKey),
732
+ model: exploreModel,
733
+ effort: "high",
734
+ tools: dispatcherTools("browse", workersKey),
735
+ ...opts.serverUrl ? { mcpServers: { [workersKey]: httpEntryFor(opts.serverUrl, "workers", opts.nonce, opts.workspaceHeaderCmd) } } : {}
736
+ };
737
+ }
738
+ const roster = opts.nativeRoster == null ? new Set(CHEAPEST_PROFILE_NATIVE_AGENT_NAMES) : opts.nativeRoster instanceof Set ? opts.nativeRoster : new Set(opts.nativeRoster);
739
+ for (const name of Object.keys(out)) if (name !== "worker-browse" && !roster.has(name)) delete out[name];
740
+ return out;
741
+ }
742
+ /** Build the literal `-m balanced` native roster. Same fixed four-agent
743
+ * surface and roles as `-m cheap`, but Sol-led at the 200K default window
744
+ * for the most complex tasks — every SUBAGENT model is a BARE router-owned
745
+ * alias (`gh-router-balanced-*`, no `[1m]`) rather than a real catalog id,
746
+ * for the same client catalog-resolution reason as the cheap builder above.
747
+ * Caller-supplied `opts.balanced*Model` values are deliberately ignored.
748
+ * Explicit delegation tuning matches cheap: proactive parallel `Explore`
749
+ * fan-out, `Plan`-first architecture, `General-Purpose` mixed execution, and
750
+ * post-integration `reviewer` verification. */
751
+ function buildBalancedProfileAgentDefinitions(opts) {
752
+ const exploreModel = BALANCED_EXPLORE_ALIAS_ID;
753
+ const planModel = BALANCED_PLAN_ALIAS_ID;
754
+ const generalModel = BALANCED_GENERAL_PURPOSE_ALIAS_ID;
755
+ const reviewerModel = BALANCED_REVIEWER_ALIAS_ID;
756
+ const semanticAvailable = opts.semanticSearchAvailable === true;
757
+ const searchKey = opts.groupKeys.search ?? GROUP_META.search.preferredKey;
758
+ const peersKey = peersKeyOf(opts.groupKeys);
759
+ const searchMcpServers = opts.serverUrl ? { [searchKey]: httpEntryFor(opts.serverUrl, "search", opts.nonce, opts.workspaceHeaderCmd) } : void 0;
760
+ const peersMcpServers = opts.serverUrl && opts.groupKeys.peers ? { [peersKey]: httpEntryFor(opts.serverUrl, "peers", opts.nonce, opts.workspaceHeaderCmd) } : void 0;
761
+ const oracleTool = opts.groupKeys.peers ? `mcp__${peersKey}__oracle` : void 0;
762
+ const readSearchTools = [
763
+ "Read",
764
+ "Grep",
765
+ "Glob",
766
+ "Bash",
767
+ "WebFetch",
768
+ "WebSearch",
769
+ `mcp__${searchKey}__*`
770
+ ];
771
+ const planTools = [
772
+ ...readSearchTools,
773
+ ...oracleTool ? [oracleTool] : [],
774
+ "Agent"
775
+ ];
776
+ const effort = (name) => BALANCED_PROFILE_NATIVE_EFFORTS[name];
777
+ const out = {
778
+ Explore: {
779
+ description: pinnedExploreDescription(true),
780
+ prompt: pinnedExplorePrompt({
781
+ explicit: true,
782
+ semanticAvailable
783
+ }),
784
+ tools: readSearchTools,
785
+ model: exploreModel,
786
+ effort: effort("Explore"),
787
+ ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
788
+ },
789
+ Plan: {
790
+ description: pinnedPlanDescription(true),
791
+ prompt: pinnedPlanPrompt({
792
+ explicit: true,
793
+ semanticAvailable
794
+ }),
795
+ tools: planTools,
796
+ model: planModel,
797
+ effort: effort("Plan"),
798
+ mcpServers: {
799
+ ...searchMcpServers ?? {},
800
+ ...peersMcpServers ?? {}
801
+ }
802
+ },
803
+ "General-Purpose": {
804
+ description: pinnedGeneralPurposeDescription(true),
805
+ prompt: pinnedGeneralPurposePrompt(),
806
+ model: generalModel,
807
+ effort: effort("General-Purpose"),
628
808
  ...searchMcpServers ? { mcpServers: searchMcpServers } : {}
629
809
  },
630
810
  reviewer: {
631
- description: "Adversarial evidence-based reviewer. Use proactively post-integration after behavior-changing, cross-boundary, or risk-sensitive changes, and always after `implementer`, before done. Runs builds/tests itself rather than assuming them. Returns SHIP / FIX / BLOCK with reproducible evidence. Never edits source.",
632
- prompt: "You are an adversarial code reviewer. Your job is not to confirm that the change works. Your job is to find the conditions under which it does not. Think carefully about the plausible failure modes of this change before you start running commands, so that what you run is chosen to expose them. Read before you judge. Inspect the changed files and relevant surrounding context, callers of affected call sites, and tests that claim to cover the change, sized to the identified risks of the change. Then verify by execution. Run the builds, linters, or test suites relevant to what changed, and any command that would surface the specific failure you suspect. Verification means output you observed. Never state that something passes, compiles, or is covered unless you ran it and read the result; where you could not run something, say so explicitly rather than inferring the outcome. Probe deliberately: boundary and empty inputs, error and early-return paths, concurrency and ordering, resource acquisition and cleanup on the failure path, partial failure and retry, backward compatibility of any changed interface, handling of untrusted input, and whether the new tests would actually fail if the change were reverted. Judge against the bar the repository already holds itself to, not an abstract ideal. Do not soften a real finding, and do not manufacture findings to appear thorough. If the change is correct and verified, say so. Do not modify source code and do not delegate to other agents. You may run build, test, and read-only inspection commands; do not run commands that alter tracked source files or touch remote infrastructure (transient build cache or test runner side effects are expected). Return a self-contained result the lead can act on immediately.\n\nReturn format. Line one must be exactly one of:\nVERDICT: SHIP\nVERDICT: FIX\nVERDICT: BLOCK\n\nSHIP means you found no blocking defect and your verification ran clean. FIX means the approach is sound but specific defects must be corrected. BLOCK means the approach itself is wrong, or verification could not be run at all.\n\nThen, using the repository's severity taxonomy:\nCritical: blocking defects (correctness, security, data loss). Each with file:line, the concrete scenario in which it fails, and how you confirmed it.\nImportant: non-blocking issues that should be fixed before shipping. Each with file:line and impact.\nSuggestion: non-blocking improvements or stylistic suggestions.\nEvidence: the commands you ran, exit status, and decisive output.\nUnverified surface: what you could not exercise, and why.\n\n" + reviewerToolSteer(),
811
+ description: pinnedReviewerDescription(true),
812
+ prompt: pinnedReviewerPrompt(semanticAvailable),
633
813
  model: reviewerModel,
634
814
  effort: effort("reviewer"),
635
815
  tools: readSearchTools,
@@ -647,7 +827,7 @@ function buildCheapestProfileAgentDefinitions(opts) {
647
827
  ...opts.serverUrl ? { mcpServers: { [workersKey]: httpEntryFor(opts.serverUrl, "workers", opts.nonce, opts.workspaceHeaderCmd) } } : {}
648
828
  };
649
829
  }
650
- const roster = opts.nativeRoster == null ? new Set(CHEAPEST_PROFILE_NATIVE_AGENT_NAMES) : opts.nativeRoster instanceof Set ? opts.nativeRoster : new Set(opts.nativeRoster);
830
+ const roster = opts.nativeRoster == null ? new Set(BALANCED_PROFILE_NATIVE_AGENT_NAMES) : opts.nativeRoster instanceof Set ? opts.nativeRoster : new Set(opts.nativeRoster);
651
831
  for (const name of Object.keys(out)) if (name !== "worker-browse" && !roster.has(name)) delete out[name];
652
832
  return out;
653
833
  }
@@ -665,6 +845,7 @@ function buildPeerAgentDefinitions(opts) {
665
845
  if (opts.fastProfile) return buildFastProfileAgentDefinitions(opts);
666
846
  if (opts.cheapProfile) return buildCheapProfileAgentDefinitions(opts);
667
847
  if (opts.cheapestProfile) return buildCheapestProfileAgentDefinitions(opts);
848
+ if (opts.balancedProfile) return buildBalancedProfileAgentDefinitions(opts);
668
849
  const out = {};
669
850
  const personas = personasFor({
670
851
  codexCli: opts.codexCli,
@@ -1119,6 +1300,11 @@ async function writePeerMcpRuntimeFiles(serverUrl, opts) {
1119
1300
  cheapestGeneralPurposeModel: opts.cheapestGeneralPurposeModel,
1120
1301
  cheapestImplementerModel: opts.cheapestImplementerModel,
1121
1302
  cheapestReviewerModel: opts.cheapestReviewerModel,
1303
+ balancedExploreModel: opts.balancedExploreModel,
1304
+ balancedPlanModel: opts.balancedPlanModel,
1305
+ balancedGeneralPurposeModel: opts.balancedGeneralPurposeModel,
1306
+ balancedReviewerModel: opts.balancedReviewerModel,
1307
+ semanticSearchAvailable: opts.semanticSearchAvailable,
1122
1308
  nativeRoster: opts.nativeRoster,
1123
1309
  personaAllowlist: opts.personaAllowlist,
1124
1310
  includeCoordinator: opts.includeCoordinator,
@@ -1128,6 +1314,7 @@ async function writePeerMcpRuntimeFiles(serverUrl, opts) {
1128
1314
  fastProfile: opts.fastProfile,
1129
1315
  cheapProfile: opts.cheapProfile,
1130
1316
  cheapestProfile: opts.cheapestProfile,
1317
+ balancedProfile: opts.balancedProfile,
1131
1318
  implementerEffort: opts.implementerEffort,
1132
1319
  reviewerEffort: opts.reviewerEffort,
1133
1320
  maxProfile: opts.maxProfile,
@@ -1882,6 +2069,176 @@ Return a compact final checkpoint:
1882
2069
  `
1883
2070
  };
1884
2071
  //#endregion
2072
+ //#region src/lib/injected-skills/gather-context-skill.ts
2073
+ const GATHER_CONTEXT_SKILL = {
2074
+ name: "gh-gather-context",
2075
+ md: `---
2076
+ name: gh-gather-context
2077
+ description: Bounded context gathering for non-trivial asks: decomposes the ask, runs lexical code searches to identify relevant files, dispatches bounded parallel explore workers to gather evidence, stitches results into a freshness-stamped context brief plus a compact version. Use when grounded context is needed before planning or changing code.
2078
+ user-invocable: true
2079
+ ---
2080
+
2081
+ # gh-gather-context: bounded context gathering
2082
+
2083
+ Use this skill when a non-trivial ask needs grounded context before planning.
2084
+ All reasoning runs at the 200K default window: the lead and every explore
2085
+ worker use the Luna model at high effort with bare slugs (no 1M accounting).
2086
+ Output is a durable full brief plus a compact downstream version.
2087
+
2088
+ ## Hard bounds
2089
+
2090
+ - Maximum rounds: 3.
2091
+ - Maximum parallel explore workers per round: 6.
2092
+ - Maximum lexical searches per round: 10.
2093
+ - Maximum follow-up reads per round: 5.
2094
+ - Terminate at the first of saturation or a cap.
2095
+ - On cap-hit, return with open unknowns flagged as residual. Do not loop forever.
2096
+
2097
+ ## Evidence tags
2098
+
2099
+ Use these exact tags on every finding and claim:
2100
+
2101
+ - verified-executable: reproduced the symptom, ran the failing test, or ran a check that directly proves the claim. This is the only deterministic confidence tag.
2102
+ - verified-source: read the actual source, config, logs, docs, or primary artifact and cited the relevant locations. This is model-mediated and can still be wrong.
2103
+ - cross-lab-agreed: a different-lab reviewer independently agreed with the claim. This reduces correlated blind spots but is advisory.
2104
+ - unverified: plausible but not confirmed; treat as residual risk.
2105
+
2106
+ ## Procedure
2107
+
2108
+ 1. Restate the ask and define the research target.
2109
+ - Identify whether this is a bug, feature, refactor, incident, or design question.
2110
+ - Name the expected downstream consumer: planner, implementer, or user.
2111
+
2112
+ 2. Decompose the ask into searchable entities.
2113
+ - Extract symbols, filenames, error strings, routes, flags, config keys, and types.
2114
+ - Define what must be true for a correct implementation.
2115
+
2116
+ 3. Run lexical search first, in parallel, in a single turn.
2117
+ - Use mcp__search__code lexically for exact symbols, filenames, errors, routes, flags, and config keys.
2118
+ - Use mcp__search__code semantically only to find concepts, then refine to lexical.
2119
+ - Use git log and git blame when authorship, regression timing, or intent matters.
2120
+ - Use mcp__search__web for upstream APIs, package behavior, protocol docs, or public issues.
2121
+
2122
+ 4. Decompose into bounded explore workers.
2123
+ - Cluster search results into at most 6 coherent investigation areas.
2124
+ - For each area, write a narrow brief: the specific question, the expected artifact, and the files to focus on.
2125
+ - Dispatch ALL explore workers in a single turn via the Agent tool (subagent_type worker-explore). Each runs read-only at the 200K default window and returns a summary with an evidence table and file:line citations. Pass maxWallClockMs 180000 on every worker call so a hung worker is reaped after 3 minutes instead of blocking its slot.
2126
+ - Keep worker results summarized; do not paste every detail into the main context.
2127
+
2128
+ 5. Stitch and verify.
2129
+ - Collect all explore results and deduplicate file references.
2130
+ - Run at most 5 targeted follow-up reads for gaps, in parallel.
2131
+ - Dispatch the worker-review subagent (via the Agent tool, maxWallClockMs 300000) to confirm source-reading for load-bearing claims.
2132
+ - Form a root-cause hypothesis or integration map, and state what would falsify it.
2133
+
2134
+ 6. Run a completeness pass.
2135
+ - Ask: what do we still not know?
2136
+ - Ask: what claim, if false, would break the conclusion?
2137
+ - Ask: have we checked primary sources for every load-bearing claim?
2138
+ - If no material unknowns remain and the root cause is at least verified-source, stop for saturation.
2139
+
2140
+ 7. Persist two outputs under .github-router/context/<slug>/.
2141
+ - context.md: full brief with the ask decomposition, searches run, worker reports, evidence table, hypothesis, freshness metadata (HEAD commit, working-tree diff hash, timestamp, repo path), residuals, and full citations.
2142
+ - context.compact.md: downstream consumable with a one-paragraph ask summary, key files with one-line purposes, critical constraints (APIs, types, patterns, forbidden changes), integration seams, and residual risks.
2143
+ - Downstream phases read by pointer and check freshness instead of re-injecting the whole brief.
2144
+
2145
+ ## Return format
2146
+
2147
+ Return a compact brief, not the whole dump:
2148
+
2149
+ - Context files: paths to context.md and context.compact.md.
2150
+ - Freshness: HEAD commit, diff hash, timestamp.
2151
+ - Termination: saturated or cap-hit; if cap-hit, name the cap.
2152
+ - Summary: 3-8 bullets with confidence tags.
2153
+ - Evidence table: claim, tag, primary source or command, reviewer status.
2154
+ - Residual unknowns: explicit list, or none.
2155
+ - Downstream guidance: recommended next action and what must be rechecked if the tree changes.
2156
+
2157
+ ## Non-goals
2158
+
2159
+ - Do not present verified-source or cross-lab-agreed as deterministic.
2160
+ - Do not hide open unknowns because the answer looks useful.
2161
+ - Do not keep searching after the cap.
2162
+ - Do not paste the entire persisted brief into later turns unless the user asks.
2163
+ `
2164
+ };
2165
+ //#endregion
2166
+ //#region src/lib/injected-skills/implement-skill.ts
2167
+ const IMPLEMENT_SKILL = {
2168
+ name: "gh-implement",
2169
+ md: `---
2170
+ name: gh-implement
2171
+ description: Parallel implementation of an approved plan using bounded Luna workers with isolated worktrees: each worker implements its task, self-tests, self-reviews, and returns a patch; the lead aggregates into a unified diff, runs staged review, and returns the final diff with a report. Use when a user-approved plan is ready for execution.
2172
+ user-invocable: true
2173
+ ---
2174
+
2175
+ # gh-implement: bounded parallel implementation with staged review
2176
+
2177
+ Use this skill only after /gh-plan produced a user-approved plan.md. All
2178
+ implementation runs at the 200K default window: the lead and every task worker
2179
+ use the Luna model at max effort with bare slugs (no 1M accounting). Review is
2180
+ staged: a Luna max pass first, then a Sol medium pass only for major issues.
2181
+
2182
+ ## Hard bounds
2183
+
2184
+ - Maximum concurrent implement workers: 8.
2185
+ - Maximum retries per task: 2.
2186
+ - Maximum review-fix cycles: 2.
2187
+ - Worktrees are auto-removed on success and retained on failure for debugging.
2188
+
2189
+ ## Procedure
2190
+
2191
+ 1. Parse the approved plan.
2192
+ - Read plan.md fully.
2193
+ - Group tasks by parallelGroup; order groups by dependency.
2194
+ - For each group, prepare an isolated git worktree per task plus a narrow task brief (task spec, relevant context excerpt, acceptance criteria, verification commands).
2195
+
2196
+ 2. Dispatch bounded implement workers, one parallel batch per group.
2197
+ - Dispatch ALL tasks in the group in a single turn via the Agent tool (subagent_type worker-implement, with worktree isolation, maxWallClockMs 600000 per task so a hung worker is reaped after 10 minutes instead of blocking its slot).
2198
+ - Each worker runs at the 200K default window and must self-contain its work:
2199
+ a. Implement the change.
2200
+ b. Run the task verification commands (tests, typecheck, lint).
2201
+ c. Self-review against the acceptance criteria.
2202
+ d. Fix any self-found issues (at most 2 internal fix cycles).
2203
+ e. Return the patch plus test results and self-review notes.
2204
+ - Do NOT dispatch the same task twice (no dedup exists); a retry is a new dispatch only after a recorded failure.
2205
+ - For a big artifact, have the worker write it to a file and return the path.
2206
+
2207
+ 3. Aggregate and validate.
2208
+ - Collect all patches and apply them sequentially to the main worktree (or merge the worktrees).
2209
+ - Run the full relevant validation: test suite, typecheck, and lint.
2210
+ - If any task fails validation, route it back to an implement worker (at most 2 retries per task). If it still fails, checkpoint with the failure as residual risk instead of pretending it is solved.
2211
+
2212
+ 4. Run staged review.
2213
+ - Pass 1 (always): dispatch the worker-review subagent (via the Agent tool, maxWallClockMs 300000) over the unified diff for correctness against acceptance criteria, code quality and consistency, security and performance regressions, and test coverage. Categorize findings as minor (style, nits) or major (logic, architecture).
2214
+ - If pass 1 finds no major issues, finish here.
2215
+ - Pass 2 (major issues only): dispatch a fix worker (via the Agent tool, maxWallClockMs 300000) at the 200K default window using the Sol model at medium effort with the flagged areas, the failing checks, and the pass-1 findings. It returns fixed patches or an explicit escalate-to-user with reasons.
2216
+
2217
+ 5. Finalize.
2218
+ - Apply any review fixes and re-run full validation.
2219
+ - Produce the unified diff for the whole plan.
2220
+ - Write .github-router/plans/<slug>/implementation-report.md with task completion status, test results summary, review findings and resolutions, the final diff path, and residual risks.
2221
+
2222
+ ## Return format
2223
+
2224
+ Return:
2225
+
2226
+ - Unified diff path.
2227
+ - Implementation report path.
2228
+ - Task completion status per task id.
2229
+ - Test, typecheck, and lint results.
2230
+ - Review summary (pass 1 findings; pass 2 findings and fixes if used).
2231
+ - Final residual risks and next action.
2232
+
2233
+ ## Non-goals
2234
+
2235
+ - Do not start without a user-approved plan.md.
2236
+ - Do not serialize work that has no data dependency; independent tasks in a group run concurrently.
2237
+ - Do not nest workflow invocations: workers are internal sessions and must not re-invoke /gh-implement.
2238
+ - Do not claim completeness when retries or review cycles are exhausted with open failures.
2239
+ `
2240
+ };
2241
+ //#endregion
1885
2242
  //#region src/lib/injected-skills/orchestrate-skill.ts
1886
2243
  const ORCHESTRATE_SKILL = {
1887
2244
  name: "gh-orchestrate",
@@ -2020,6 +2377,94 @@ Return:
2020
2377
  `
2021
2378
  };
2022
2379
  //#endregion
2380
+ //#region src/lib/injected-skills/plan-skill.ts
2381
+ const PLAN_SKILL = {
2382
+ name: "gh-plan",
2383
+ md: `---
2384
+ name: gh-plan
2385
+ description: Holistic planning from gathered context: ingests the context brief, creates a scoped modular ordered implementation plan with explicit tasks for cheap implementation workers, surfaces open questions for user approval, and persists plan.md. Use when a non-trivial change needs a reviewed plan before implementation.
2386
+ user-invocable: true
2387
+ ---
2388
+
2389
+ # gh-plan: holistic planning for cheap implementation
2390
+
2391
+ Use this skill after /gh-gather-context (or when equivalent context is already
2392
+ available) and before any implementation. The planner runs at the 200K default
2393
+ window using the Sol model at medium effort. The plan must be scoped, modular,
2394
+ non-overlapping, and ordered, with enough detail for Luna implementation
2395
+ workers to execute each task in isolation. User approval is mandatory before
2396
+ implementation.
2397
+
2398
+ ## Prerequisites
2399
+
2400
+ - A freshness-stamped context brief from /gh-gather-context, or equivalent context.
2401
+ - Read context.compact.md first; read context.md sections on demand (residual unknowns, evidence table).
2402
+
2403
+ ## Hard bounds
2404
+
2405
+ - Maximum tasks: 20.
2406
+ - Maximum parallel groups: 5.
2407
+ - Keep planner input well under the 200K window (target at most around 150K tokens of context) so there is headroom for reasoning and output. This is self-discipline, not an enforced cap: prefer the compact brief and read full sections only on demand.
2408
+
2409
+ ## Procedure
2410
+
2411
+ 1. Ingest context.
2412
+ - Read context.compact.md fully.
2413
+ - Read the evidence table and residual unknowns from context.md.
2414
+ - Identify acceptance criteria, constraints, integration seams, and forbidden changes.
2415
+
2416
+ 2. Build a blind-spot table before decomposing.
2417
+ - Wrong-spec risk: judgment-only, mitigated only by user-blessed acceptance criteria.
2418
+ - Root-cause risk: executable-checkable if reproduced or covered by a failing test; otherwise advisory.
2419
+ - Integration risk: usually source-verified plus tests where possible.
2420
+ - Regression risk: executable-checkable when tests, typecheck, or lint cover it.
2421
+ - Review risk: advisory cross-lab review reduces correlated blind spots.
2422
+ - Concurrency or merge risk: source-verified and sometimes executable-checkable.
2423
+ - Missing-test risk: executable-checkable only after a test exists and runs.
2424
+ - Tag every blind spot as executable-checkable or judgment-only.
2425
+
2426
+ 3. Decompose into minimal safe increments.
2427
+ - Each task touches a single file or a tightly coupled file group.
2428
+ - Each task states input artifacts, output artifact, acceptance criteria, verification commands, and rollback concern.
2429
+ - Order tasks by dependency (topological sort); tasks with no data dependency share a parallel group.
2430
+ - Keep each task small enough for one Luna worker at the 200K window (target at most 50K context tokens of relevant files per task).
2431
+ - If the ask needs discovery follow-ups, delegate them to worker-explore background subagents (via the Agent tool, maxWallClockMs 180000) rather than bloating the plan.
2432
+
2433
+ 4. Surface open questions before finalizing.
2434
+ - Ask about ambiguous acceptance criteria, design decisions with multiple valid approaches, risk tolerance, and test strategy.
2435
+ - Present a short candidate list for confirmation where possible.
2436
+
2437
+ 5. Persist the plan to .github-router/plans/<slug>/plan.md.
2438
+ - Ask summary and user-blessed acceptance criteria.
2439
+ - Blind-spot table with executable-checkable or judgment-only tags.
2440
+ - Ordered task list with ids, files, dependencies, parallel groups, acceptance criteria, verification commands, rollback concerns, and estimated context tokens.
2441
+ - Open questions and user answers.
2442
+ - Cost estimate: task count, parallel groups, and context tokens.
2443
+ - Residual risks.
2444
+
2445
+ 6. Checkpoint with the user and wait for explicit approval.
2446
+ - Present the goal, acceptance criteria, task-to-group map, per-task blind spot killed, residual risks, and cost estimate.
2447
+ - If the user rejects scope or cost, downshift to the smallest plan that kills the important blind spots.
2448
+ - Do not proceed to implementation without approval.
2449
+
2450
+ ## Return format
2451
+
2452
+ Return:
2453
+
2454
+ - Plan file: path to the durable plan.md.
2455
+ - Task count and parallel groups.
2456
+ - Open questions and user answers.
2457
+ - Cost estimate.
2458
+ - Residual risks and next action (implementation only after approval).
2459
+
2460
+ ## Non-goals
2461
+
2462
+ - Do not edit implementation files while planning; in plan mode, produce the plan and acceptance criteria only.
2463
+ - Do not present judgment-only conclusions as executable guarantees.
2464
+ - Do not hide open unknowns because the plan looks complete.
2465
+ `
2466
+ };
2467
+ //#endregion
2023
2468
  //#region src/lib/injected-skills/research-skill.ts
2024
2469
  const RESEARCH_SKILL = {
2025
2470
  name: "gh-research",
@@ -2191,6 +2636,88 @@ The dispatcher calls the worker once and relays its result verbatim.
2191
2636
  `
2192
2637
  };
2193
2638
  //#endregion
2639
+ //#region src/lib/skill-model-contract.ts
2640
+ /**
2641
+ * Universal skill model contract for the `/gh-gather-context`, `/gh-plan`,
2642
+ * and `/gh-implement` pipeline skills.
2643
+ *
2644
+ * ALL roles run at the 200K DEFAULT context window (bare slugs, no `[1m]`
2645
+ * accounting bracket) on every profile. This module is deliberately
2646
+ * dependency-free so profile contracts, launch validation, worker dispatch,
2647
+ * and the injected skill bodies can all import the same literals without
2648
+ * cycles.
2649
+ *
2650
+ * Model choices (per pipeline design):
2651
+ * - gatherContext lead + explore agents: Luna, high effort
2652
+ * - plan lead: Sol, medium effort
2653
+ * - implement lead + task agents: Luna, max effort
2654
+ * - review pass 1: Luna, max effort
2655
+ * - review pass 2 (major issues only): Sol, medium effort
2656
+ */
2657
+ const SKILL_LUNA_MODEL_ID = "gpt-5.6-luna";
2658
+ const SKILL_SOL_MODEL_ID = "gpt-5.6-sol";
2659
+ Object.freeze({
2660
+ gatherContext: Object.freeze({
2661
+ lead: SKILL_LUNA_MODEL_ID,
2662
+ exploreAgent: SKILL_LUNA_MODEL_ID,
2663
+ leadEffort: "high",
2664
+ agentEffort: "high"
2665
+ }),
2666
+ plan: Object.freeze({
2667
+ lead: SKILL_SOL_MODEL_ID,
2668
+ leadEffort: "medium"
2669
+ }),
2670
+ implement: Object.freeze({
2671
+ lead: SKILL_LUNA_MODEL_ID,
2672
+ taskAgent: SKILL_LUNA_MODEL_ID,
2673
+ leadEffort: "max",
2674
+ agentEffort: "max"
2675
+ }),
2676
+ review: Object.freeze({
2677
+ pass1: Object.freeze({
2678
+ model: SKILL_LUNA_MODEL_ID,
2679
+ effort: "max"
2680
+ }),
2681
+ pass2: Object.freeze({
2682
+ model: SKILL_SOL_MODEL_ID,
2683
+ effort: "medium"
2684
+ })
2685
+ })
2686
+ });
2687
+ Object.freeze({
2688
+ gatherContext: Object.freeze({
2689
+ maxRounds: 3,
2690
+ maxExploreAgentsPerRound: 6,
2691
+ maxLexicalSearchesPerRound: 10,
2692
+ maxFollowUpReadsPerRound: 5
2693
+ }),
2694
+ plan: Object.freeze({
2695
+ maxTasks: 20,
2696
+ maxParallelGroups: 5
2697
+ }),
2698
+ implement: Object.freeze({
2699
+ maxConcurrentAgents: 8,
2700
+ maxRetriesPerTask: 2,
2701
+ maxReviewFixCycles: 2
2702
+ })
2703
+ });
2704
+ /**
2705
+ * Profiles that receive the pipeline skills. Every pinned profile gets
2706
+ * them; `standard` is intentionally excluded (it keeps the existing
2707
+ * research/orchestrate/worker surface).
2708
+ */
2709
+ const PIPELINE_SKILL_PROFILES = [
2710
+ "fast",
2711
+ "max",
2712
+ "cheap",
2713
+ "cheap1m",
2714
+ "cheapest",
2715
+ "balanced"
2716
+ ];
2717
+ function isPipelineSkillProfile(profileId) {
2718
+ return PIPELINE_SKILL_PROFILES.includes(profileId);
2719
+ }
2720
+ //#endregion
2194
2721
  //#region src/lib/injected-skills/artifact-review-skill.ts
2195
2722
  function buildArtifactReviewSkill(peersKey = "peers") {
2196
2723
  const toolPrefix = `mcp__${peersKey}__artifact_`;
@@ -2329,13 +2856,20 @@ function joinClauses(parts) {
2329
2856
  * there is no quality-for-cost trade being hidden by leading with the cheap
2330
2857
  * tier; reserve `reviewer` for the higher-stakes assessment it is there for. */
2331
2858
  function buildNativeReachClauses(opts) {
2332
- if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest") return joinClauses([
2333
- "`Explore` for broad repository discovery, dependency mapping, and convention tracking",
2334
- "`Plan` for architectural sequencing, interface contracts, migration risk, and runnable acceptance criteria",
2335
- "`general-purpose` for mixed, iterative, or multi-step execution tasks",
2336
- "`implementer` for surgical coding changes matching existing conventions",
2337
- "`reviewer` for independent adversarial verification, reproduction, and root-causing"
2338
- ]);
2859
+ if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest" || opts.profile === "balanced") {
2860
+ if (opts.profile === "cheapest") return joinClauses([
2861
+ "`Explore` for broad repository discovery, dependency mapping, and convention tracking",
2862
+ "`Plan` for architectural sequencing, interface contracts, migration risk, and runnable acceptance criteria",
2863
+ "`General-Purpose` for mixed, iterative, or multi-step execution tasks",
2864
+ "`reviewer` for independent adversarial verification, reproduction, and root-causing"
2865
+ ]);
2866
+ return joinClauses([
2867
+ "`Explore` for broad repository discovery, dependency mapping, and convention tracking (launch in parallel)",
2868
+ "`Plan` for architectural sequencing, interface contracts, migration risk, and runnable acceptance criteria (delegates discovery to `Explore`)",
2869
+ "`General-Purpose` for mixed, iterative, or multi-step execution tasks (follows a Plan handoff when one exists)",
2870
+ "`reviewer` for independent adversarial verification, reproduction, and root-causing after non-trivial changes"
2871
+ ]);
2872
+ }
2339
2873
  const clauses = [];
2340
2874
  const implementerFast = opts.implementerFastAvailable !== false;
2341
2875
  const reviewerFast = opts.reviewerFastAvailable !== false;
@@ -2367,15 +2901,17 @@ function buildOperatingDefaultsDirective(opts = {}) {
2367
2901
  if (opts.profile === "max") {
2368
2902
  const peersKey = opts.peersKey ?? opts.groupKeys?.peers ?? "peers";
2369
2903
  const artifactClause = opts.artifactAvailable ? ` Live human review is available in the artifact panel via \`mcp__${peersKey}__artifact_*\`.` : "";
2370
- return "## Operating defaults (apply when the user has not specified otherwise; the user's explicit direction and the domain's own standards always override)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Handle narrow, obvious, surgical, and single-command work directly; delegate a bounded workstream when it is broad, slow, context-heavy, or benefits from a genuinely independent perspective. " + MAX_PARALLELISM_RULE + " Brief each role with the desired outcome, relevant context, constraints, expected evidence, and verification. Use the roster as complementary capabilities, never as a required Explore → Plan → implement → review sequence. Avoid overlapping assignments and do not ask several models the same generic question. Synthesize results against the repository and executable checks: model agreement is not verification. Use one fresh-context peer only when a consequential judgment remains after direct checks; use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding counsel for one focused consequential uncertainty that evidence and the appropriate roles cannot settle; it has no approval or workflow authority, and a further consultation requires materially new or conflicting evidence." + artifactClause;
2904
+ return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Handle narrow, obvious, surgical, and single-command work directly; delegate a bounded workstream when it is broad, slow, context-heavy, or benefits from a genuinely independent perspective. " + MAX_PARALLELISM_RULE + " Brief each role with the desired outcome, relevant context, constraints, expected evidence, and verification. Use the roster as complementary capabilities, never as a required Explore → Plan → implement → review sequence. Avoid overlapping assignments and do not ask several models the same generic question. Synthesize results against the repository and executable checks: model agreement is not verification. Use one fresh-context peer only when a consequential judgment remains after direct checks; use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding counsel for one focused consequential uncertainty that evidence and the appropriate roles cannot settle; it has no approval or workflow authority, and a further consultation requires materially new or conflicting evidence." + artifactClause + "\n\nPipeline skills (all 200K default context). For non-trivial changes, run in order: (1) `/gh-gather-context` BEFORE planning when grounded context is needed: it decomposes the ask, runs lexical search, fans out to bounded Luna-high explore workers, and writes context.md plus context.compact.md; (2) `/gh-plan` AFTER context and BEFORE implementation: it ingests the context brief with Sol-medium, produces a scoped modular ordered plan.md with explicit tasks for Luna workers, surfaces open questions, and waits for user approval; (3) `/gh-implement` AFTER plan approval: it runs bounded parallel Luna-max task workers in isolated worktrees (each self-tests and self-reviews), aggregates a unified diff, and runs staged review (Luna max, then Sol medium for major issues only). Skip the pipeline for trivial surgical work. Never implement without an approved plan.";
2371
2905
  }
2372
- if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest") {
2906
+ if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest" || opts.profile === "balanced") {
2373
2907
  const isCheap = opts.profile === "cheap" || opts.profile === "cheap1m";
2374
2908
  const isCheapest = opts.profile === "cheapest";
2375
- const profileLabel = isCheapest ? "Cheapest" : isCheap ? "Cheap" : "Fast";
2376
- const oracleDescriptor = isCheapest ? "GPT-5.6 Sol (200K/high)" : isCheap ? "Grok 4.6 (200K/medium)" : "exact Opus 5 (1M/high)";
2909
+ const isBalanced = opts.profile === "balanced";
2910
+ const profileLabel = isCheapest ? "Cheapest" : isCheap ? "Cheap" : isBalanced ? "Balanced" : "Fast";
2911
+ const oracleDescriptor = isCheapest ? "GPT-5.6 Sol (200K/high)" : isCheap || isBalanced ? "Grok 4.6 (200K/medium)" : "exact Opus 5 (1M/high)";
2377
2912
  const astraDescriptor = isCheap && !isCheapest ? "200K/medium" : "200K/high";
2378
- if (opts.fastRuntimeAvailable === false) return `## Operating defaults (apply when the user has not specified otherwise; the user's explicit direction and the domain's own standards always override)
2913
+ const searchGuidance = opts.semanticSearchAvailable === true ? "Search strategy (cheapest first): (1) LEXICAL `code` search (mode:\"lexical\"/\"exact\", plus Grep/Glob) for symbols, filenames, errors, routes, flags, and config keys — zero model cost; (2) SEMANTIC `code` search for intent/concept questions where literal keywords may not appear; (3) `Explore` subagents read the narrowed files and return file:line conclusions — expensive models (Plan, reviewer, Oracle) see only the synthesized subset, never raw search output. " : "Search strategy (cheapest first): LEXICAL `code` search (mode:\"lexical\"/\"exact\", plus Grep/Glob) for symbols, filenames, errors, routes, flags, and config keys — zero model cost. `Explore` subagents read the narrowed files and return file:line conclusions — expensive models (Plan, reviewer, Oracle) see only the synthesized subset, never raw search output. ";
2914
+ if (opts.fastRuntimeAvailable === false) return `## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)
2379
2915
 
2380
2916
  ${profileLabel} profile runtime wiring is unavailable. Work directly, use only tools actually listed in this session, verify with the repository's relevant build/tests before declaring done, report uncertainty, and do not invent unavailable capabilities.`;
2381
2917
  const peersKey = opts.peersKey ?? opts.groupKeys?.peers ?? "peers";
@@ -2386,32 +2922,29 @@ ${profileLabel} profile runtime wiring is unavailable. Work directly, use only t
2386
2922
  const workerBrowseClause = opts.browseAvailable ? ` \`worker-browse\` runs delegated autonomous browsing tasks through \`mcp__${workersKey}__browse\`.` : "";
2387
2923
  const artifactClause = opts.artifactAvailable ? ` \`mcp__${peersKey}__artifact_*\` provides human review in the artifact panel with auto-open on plan completion.` : "";
2388
2924
  const astraClause = opts.astraAvailable && (opts.profile === "fast" || opts.profile === "cheap1m") ? ` \`mcp__${peersKey}__astra\` (GPT-6 Astra ${astraDescriptor}) is the terminal escalation consultant for the lead only, reserved strictly for the hardest dead ends when direct evidence, Advisor, and Oracle have all failed to produce a defensible path (consulted at most 1-2 times per decision with concise context and specific questions).` : "";
2389
- return `## Operating defaults (apply when the user has not specified otherwise; the user's explicit direction and the domain's own standards always override)
2390
-
2391
- ${profileLabel} launch profile. The lead coordinates execution across specialized native roles: ` + buildNativeReachClauses(opts) + `. In plan mode or when designing changes with complex sequencing, interface contracts, or acceptance criteria, delegate architectural planning to \`Plan\` (in plan mode, produce the plan and acceptance criteria; do not edit files). Discovery rule: delegate to \`Explore\` instead of exploring yourself. For any question spanning more than a couple of files, launch one or more \`Explore\` subagents in parallel with scoped evidence questions and let them return file:line citations; read directly only the small subset you must touch to decide or edit. \`Plan\` follows the same rule when it needs repository facts. Implementation rule: delegate bounded implementation to \`implementer\` whenever a fresh context helps or lead-context pressure matters; brief with outcome, constraints, files in scope, and verification. Review rule: after behavior-changing, cross-boundary, or risk-sensitive implementation, and always after \`implementer\` completes, run relevant build/tests then invoke \`reviewer\` before declaring done. Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. \`Explore\` is cheap and may be launched in parallel across independent discovery questions. Send independent subagent calls in parallel within a single turn. Delegation graph: the lead may invoke all five; \`Plan\` may invoke \`Explore\` and \`reviewer\`; \`implementer\` and \`general-purpose\` may invoke \`reviewer\`; \`Explore\`, \`reviewer\`, and \`worker-browse\` cannot invoke native subagents.
2925
+ return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + (isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly. ` + buildNativeReachClauses(opts) + ". Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` may be used for discovery spanning more than a couple of files; `Plan` for sequencing with complex interfaces or acceptance criteria; `General-Purpose` for mixed multi-step execution; `reviewer` for behavior-changing or risk-sensitive changes. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n" : `${profileLabel} launch profile. The lead coordinates execution across specialized native roles: ` + buildNativeReachClauses(opts) + ". In plan mode or when designing changes with complex sequencing, interface contracts, or acceptance criteria, delegate architectural planning to `Plan` (in plan mode, produce the plan and acceptance criteria; do not edit files). Phase pipeline (budget gather, expensive plan, budget execution, expensive verification). GATHER: delegate to `Explore` instead of exploring yourself. For any question spanning more than a couple of files, launch one or more `Explore` subagents in parallel with scoped evidence questions and let them return file:line citations; read directly only the small subset you must touch to decide or edit. `Plan` follows the same rule when it needs repository facts. PLAN: `Plan` produces handoff-ready steps (ordered, file:line, done conditions, acceptance criteria) for a `General-Purpose` executor that cannot see its reasoning; `Plan` may consult Oracle on unresolved trade-offs and reports any remaining gap to the lead. EXECUTE: delegate mixed multi-step work and Plan handoffs to `General-Purpose` in a fresh context to preserve lead context; brief with outcome, constraints, files in scope, and verification. VERIFY: after behavior-changing, cross-boundary, or risk-sensitive implementation, run relevant build/tests then invoke `reviewer` before declaring done. Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` is cheap and may be launched in parallel across independent discovery questions. Send independent subagent calls in parallel within a single turn. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n") + `Consultation guidance: Follow an evidence-first escalation ladder. Direct empirical evidence (search, code, tests, builds) settles factual questions first. Advisor is an optional, non-binding, lead-only transcript-aware sounding board for trajectory guidance or framing checks (direction, not dictation), and never use it for routine progress, waiting, directly verifiable facts, or completion ritual. \`mcp__${peersKey}__oracle\` is ${oracleDescriptor}, an expert consultant available to the lead and \`Plan\`, preferred over advisor for difficult conceptual, algorithmic, spec/protocol, or architectural tradeoffs evaluated in a self-contained brief; \`reviewer\` and other subagents cannot call Oracle. \`Plan\` may consult Oracle on unresolved trade-offs, and reports any remaining tie-breaking gap to the lead.${astraClause} ` + searchGuidance + `\`mcp__${searchKey}__web\` provides citable sources.${browserClause}${workerBrowseClause}${artifactClause}\n\nPipeline skills (all 200K default context). For non-trivial changes, run in order: (1) \`/gh-gather-context\` BEFORE planning when grounded context is needed: it decomposes the ask, runs lexical search, fans out to bounded Luna-high explore workers, and writes context.md plus context.compact.md; (2) \`/gh-plan\` AFTER context and BEFORE implementation: it ingests the context brief with Sol-medium, produces a scoped modular ordered plan.md with explicit tasks for Luna workers, surfaces open questions, and waits for user approval; (3) \`/gh-implement\` AFTER plan approval: it runs bounded parallel Luna-max task workers in isolated worktrees (each self-tests and self-reviews), aggregates a unified diff, and runs staged review (Luna max, then Sol medium for major issues only). Skip the pipeline for trivial surgical work. Never implement without an approved plan.
2392
2926
 
2393
- Consultation guidance: Follow an evidence-first escalation ladder. Direct empirical evidence (search, code, tests, builds) settles factual questions first. Advisor is an optional, non-binding, lead-only transcript-aware sounding board for trajectory guidance or framing checks (direction, not dictation), and never use it for routine progress, waiting, directly verifiable facts, or completion ritual. \`mcp__${peersKey}__oracle\` is ${oracleDescriptor}, an expert consultant available to the lead and \`Plan\`, preferred over advisor for difficult conceptual, algorithmic, spec/protocol, or architectural tradeoffs evaluated in a self-contained brief; \`reviewer\` and other subagents cannot call Oracle. \`Plan\` may consult Oracle on unresolved trade-offs, and reports any remaining tie-breaking gap to the lead.${astraClause} \`mcp__${searchKey}__code\` provides semantic-first code search and \`mcp__${searchKey}__web\` provides citable sources.${browserClause}${workerBrowseClause}${artifactClause}\n\nVerify claims with concrete repository evidence and tests before declaring work done. User instructions outrank delegation triggers. Stop named teammates when finished.`;
2927
+ Verify claims with concrete repository evidence and tests before declaring work done. User instructions outrank delegation triggers. Stop named teammates when finished.`;
2394
2928
  }
2395
- return "## Operating defaults (apply when the user has not specified otherwise; the user's explicit direction and the domain's own standards always override)\n\nOrchestrate. Delegate research, implementation, review, and large reads to the right subagent, worker, or model. Reach for " + buildNativeReachClauses(opts) + "; worker-* agents for background non-blocking runs; Task subagents for parallel work; peer critics for review. That keeps your own context free to reason and collaborate with the user. Launch independent agents concurrently in a single message rather than serially. Delegation pays when the work is WIDE (many files or sources to sweep) or SLOW, and you need only the conclusion: the main thread is where you think with and respond to the user, and its context window is a finite shared resource. It does NOT pay merely because a sub-question is separable. A narrow, deep question whose whole value is file:line fidelity loses exactly that through a summarization layer, and a sub-question you could answer in one command is cheaper done directly than paying a subagent's startup. Do trivial, surgical, and last-mile work yourself. A named teammate persists after it reports so you can send it follow-ups, and nothing reaps it for you: its idle notice means available, not finished. Stop it once you are done with it.\n\nAdversarial review. The peer critics (`codex_critic` and `codex_reviewer`, `gemini_critic` and `gemini_reviewer`, `opus_critic`, and the `peer-review-coordinator` that fans out to several of them) are fresh-context models, so what they add is a blind spot that whoever produced the work cannot reach by thinking harder about it; prefer a critic from a different lab than the producer, since blind spots correlate within a lab. The `advisor` is a complement and not a substitute: it sees your transcript, so it catches your own drift and momentum, but it inherits your framing, which is exactly what a fresh-context critic does not. They earn their keep on consequential design choices, recommendations, and hard-to-reverse decisions: the cases where plausible alternatives remain and the conclusion rests on judgment rather than on something you can verify directly. That is where confabulation hides, so budget the wait even under delivery pressure. Always consult one when the change touches auth, user input, database queries, crypto, or serialization. They do NOT pay for read-only tracing, ordinary repository lookup, or a conclusion that a focused test, a direct reproduction, or unambiguous code evidence already settles. Asking a critic to re-derive a proven fact returns a confident answer either way, which is ritual skepticism rather than review, and skipping them there is the right call and not a shortcut. Match the lens to the artifact: a strategic critic for plans and trade-offs, a code reviewer for a concrete diff, the coordinator only when the risk warrants several independent lenses. Give whichever you pick the artifact and the constraints and not your rationale, since justification anchors the review and dulls it.\n\nAim high. Default to radical simplicity and a relentless focus on the user's real experience: design for the person and the job to be done, not the demo. Reason about the whole system from first principles, anticipating scale and the long arc rather than patching the surface. Work backwards from the outcome the user actually needs. Question every assumption and prefer what you can derive, reproduce, or test.\n\nEngineering excellence. When making technical decisions, give little weight to development cost; prefer quality, simplicity, robustness, scalability, and long-term maintainability. Fix a bug by first reproducing it end to end, as close to how a real user hits it as you can, so you solve the real problem and not a symptom. When testing a product end to end, be picky about the UI and obsessed with pixel perfection: if something clearly looks off, even when it is unrelated to your task, get it fixed along the way. Hold that same bar for the codebase itself: a lint error, a failing test, or a flaky test is worth fixing the moment you see it, whoever introduced it. Fold it into your current work rather than letting it derail the task the user actually asked for.";
2929
+ return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nOrchestrate. Delegate research, implementation, review, and large reads to the right subagent, worker, or model. Reach for " + buildNativeReachClauses(opts) + "; worker-* agents for background non-blocking runs; Task subagents for parallel work; peer critics for review. That keeps your own context free to reason and collaborate with the user. Launch independent agents concurrently in a single message rather than serially. Delegation pays when the work is WIDE (many files or sources to sweep) or SLOW, and you need only the conclusion: the main thread is where you think with and respond to the user, and its context window is a finite shared resource. It does NOT pay merely because a sub-question is separable. A narrow, deep question whose whole value is file:line fidelity loses exactly that through a summarization layer, and a sub-question you could answer in one command is cheaper done directly than paying a subagent's startup. Do trivial, surgical, and last-mile work yourself. A named teammate persists after it reports so you can send it follow-ups, and nothing reaps it for you: its idle notice means available, not finished. Stop it once you are done with it.\n\nAdversarial review. The peer critics (`codex_critic` and `codex_reviewer`, `gemini_critic` and `gemini_reviewer`, `opus_critic`, and the `peer-review-coordinator` that fans out to several of them) are fresh-context models, so what they add is a blind spot that whoever produced the work cannot reach by thinking harder about it; prefer a critic from a different lab than the producer, since blind spots correlate within a lab. The `advisor` is a complement and not a substitute: it sees your transcript, so it catches your own drift and momentum, but it inherits your framing, which is exactly what a fresh-context critic does not. They earn their keep on consequential design choices, recommendations, and hard-to-reverse decisions: the cases where plausible alternatives remain and the conclusion rests on judgment rather than on something you can verify directly. That is where confabulation hides, so budget the wait even under delivery pressure. Always consult one when the change touches auth, user input, database queries, crypto, or serialization. They do NOT pay for read-only tracing, ordinary repository lookup, or a conclusion that a focused test, a direct reproduction, or unambiguous code evidence already settles. Asking a critic to re-derive a proven fact returns a confident answer either way, which is ritual skepticism rather than review, and skipping them there is the right call and not a shortcut. Match the lens to the artifact: a strategic critic for plans and trade-offs, a code reviewer for a concrete diff, the coordinator only when the risk warrants several independent lenses. Give whichever you pick the artifact and the constraints and not your rationale, since justification anchors the review and dulls it.\n\nAim high. Default to radical simplicity and a relentless focus on the user's real experience: design for the person and the job to be done, not the demo. Reason about the whole system from first principles, anticipating scale and the long arc rather than patching the surface. Work backwards from the outcome the user actually needs. Question every assumption and prefer what you can derive, reproduce, or test.\n\nEngineering excellence. When making technical decisions, give little weight to development cost; prefer quality, simplicity, robustness, scalability, and long-term maintainability. Fix a bug by first reproducing it end to end, as close to how a real user hits it as you can, so you solve the real problem and not a symptom. When testing a product end to end, be picky about the UI and obsessed with pixel perfection: if something clearly looks off, even when it is unrelated to your task, get it fixed along the way. Hold that same bar for the codebase itself: a lint error, a failing test, or a flaky test is worth fixing the moment you see it, whoever introduced it. Fold it into your current work rather than letting it derail the task the user actually asked for.";
2396
2930
  }
2397
2931
  /** The all-available form of the directive. Prefer
2398
2932
  * `buildOperatingDefaultsDirective` on any path that knows which natives
2399
2933
  * resolved; this const is the default for callers and tests that do not model
2400
2934
  * a thin catalog. */
2401
2935
  const OPERATING_DEFAULTS_DIRECTIVE = buildOperatingDefaultsDirective();
2402
- const STANDARD_OPERATING_DEFAULTS_DIGEST = "## Operating defaults (the user's explicit direction and the domain's standards always override)\n\nDelegate when the work is WIDE (many files or sources to sweep) or SLOW and you need only the conclusion, to protect the main thread's finite context and keep it free for reasoning and interacting with the user; prefer parallel delegation for independent work. Do NOT delegate merely because a sub-question is separable: a narrow, deep question whose value is file:line fidelity loses exactly that through a summarization layer, and one answerable in a single command is cheaper done directly. Do trivial, surgical, and last-mile work yourself. Stop a named teammate once you are done with it: it persists for follow-ups, its idle notice means available rather than finished, and nothing reaps it for you.\n\nVerify, do not assert. Run the code, read the file, check the exit code. A claim in prose is worth nothing against state you did not check, and a check that cannot fail proves nothing. Reproduce a bug end to end, the way a real user hits it, before fixing it. Fix a lint error, failing test, or flake the moment you see it, whoever introduced it, without letting it derail the task at hand. Prefer quality and long-term maintainability over development cost.\n\nVerification has a blind spot: a consequential recommendation, a design or trade-off call, or a hard-to-reverse decision that still turns on judgment among plausible alternatives once the direct evidence is in. That is where confabulation hides, so put it past a peer critic before you ship it and budget the wait even under delivery pressure. When a test, a run, a reproduction, or a search would settle the claim, settle it that way instead; a critic asked to re-derive what you can already prove is ritual, not review.\n\nThe agent roster, the tool surface, and the reasoning behind these defaults are in your CLAUDE.md project instructions. Read them when choosing HOW to work; the rules above apply without a lookup.";
2936
+ const STANDARD_OPERATING_DEFAULTS_DIGEST = "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nDelegate when the work is WIDE (many files or sources to sweep) or SLOW and you need only the conclusion, to protect the main thread's finite context and keep it free for reasoning and interacting with the user; prefer parallel delegation for independent work. Do NOT delegate merely because a sub-question is separable: a narrow, deep question whose value is file:line fidelity loses exactly that through a summarization layer, and one answerable in a single command is cheaper done directly. Do trivial, surgical, and last-mile work yourself. Stop a named teammate once you are done with it: it persists for follow-ups, its idle notice means available rather than finished, and nothing reaps it for you.\n\nVerify, do not assert. Run the code, read the file, check the exit code. A claim in prose is worth nothing against state you did not check, and a check that cannot fail proves nothing. Reproduce a bug end to end, the way a real user hits it, before fixing it. Fix a lint error, failing test, or flake the moment you see it, whoever introduced it, without letting it derail the task at hand. Prefer quality and long-term maintainability over development cost.\n\nVerification has a blind spot: a consequential recommendation, a design or trade-off call, or a hard-to-reverse decision that still turns on judgment among plausible alternatives once the direct evidence is in. That is where confabulation hides, so put it past a peer critic before you ship it and budget the wait even under delivery pressure. When a test, a run, a reproduction, or a search would settle the claim, settle it that way instead; a critic asked to re-derive what you can already prove is ritual, not review.\n\nThe agent roster, the tool surface, and the reasoning behind these defaults are in your CLAUDE.md project instructions. Read them when choosing HOW to work; the rules above apply without a lookup.";
2403
2937
  function buildOperatingDefaultsDigest(opts = {}) {
2404
- if (opts.profile === "max") return "## Operating defaults (the user's explicit direction and the domain's standards always override)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Do narrow, obvious, surgical, and single-command work directly; delegate bounded work that is broad, slow, context-heavy, or independently valuable. " + MAX_PARALLELISM_RULE + " Give delegated workstreams non-overlapping scopes and state the outcome, constraints, evidence, and verification expected.\n\nSynthesize and verify: run the code, inspect outputs, and check tests. Agent count and agreement are not evidence. Use a fresh-context peer only for consequential judgment that remains after direct checks, and use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding, primary-lead-only counsel for one focused consequential uncertainty; it is not an approval or completion gate.";
2405
- if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest") {
2938
+ if (opts.profile === "max") return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Do narrow, obvious, surgical, and single-command work directly; delegate bounded work that is broad, slow, context-heavy, or independently valuable. " + MAX_PARALLELISM_RULE + " Give delegated workstreams non-overlapping scopes and state the outcome, constraints, evidence, and verification expected.\n\nPipeline skills (all 200K default): `/gh-gather-context` before planning when context is needed, `/gh-plan` before implementation (waits for user approval), `/gh-implement` after approval with bounded parallel workers and staged review. Skip for trivial work.\n\nSynthesize and verify: run the code, inspect outputs, and check tests. Agent count and agreement are not evidence. Use a fresh-context peer only for consequential judgment that remains after direct checks, and use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding, primary-lead-only counsel for one focused consequential uncertainty; it is not an approval or completion gate.";
2939
+ if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest" || opts.profile === "balanced") {
2406
2940
  const isCheap = opts.profile === "cheap" || opts.profile === "cheap1m";
2407
2941
  const isCheapest = opts.profile === "cheapest";
2408
- const profileLabel = isCheapest ? "Cheapest" : isCheap ? "Cheap" : "Fast";
2409
- const oracleDescriptor = isCheapest ? "(GPT-5.6 Sol 200K/high, lead and Plan)" : isCheap ? "(Grok 4.6 200K/medium, lead and Plan)" : "(Opus 5 1M/high, lead and Plan)";
2942
+ const isBalanced = opts.profile === "balanced";
2943
+ const profileLabel = isCheapest ? "Cheapest" : isCheap ? "Cheap" : isBalanced ? "Balanced" : "Fast";
2944
+ const oracleDescriptor = isCheapest ? "(GPT-5.6 Sol 200K/high, lead and Plan)" : isCheap || isBalanced ? "(Grok 4.6 200K/medium, lead and Plan)" : "(Opus 5 1M/high, lead and Plan)";
2410
2945
  const advisorDescriptor = isCheapest ? "(Gemini/high, lead-only)" : "(Sol/high, lead-only)";
2411
2946
  const astraStep = opts.astraAvailable && (opts.profile === "fast" || opts.profile === "cheap1m") ? `; (4) \`astra\` (GPT-6 Astra 200K/${isCheap ? "medium" : "high"}, lead-only) only as a last resort when direct evidence, Advisor, and Oracle cannot produce a defensible path (at most 1-2 calls per decision).` : ".";
2412
- return `## Operating defaults (the user's explicit direction and the domain's standards always override)
2413
-
2414
- ${profileLabel} launch profile. The lead coordinates execution across specialized roles: delegate broad discovery to \`Explore\` in parallel and do not sweep the repo yourself (read directly only files you will act on); delegate to \`Plan\` in plan mode or when structuring complex multi-step sequencing (\`Plan\` is an advisory planning capability, not an approval gate); delegate bounded implementation to \`implementer\` in a fresh context to preserve lead context (\`general-purpose\` for mixed multi-step execution); delegate to \`reviewer\` after behavior-changing or risk-sensitive implementation, and always after \`implementer\` completes, to verify correctness before declaring done; handle trivial and surgical edits directly. Send independent subagent calls in parallel within a single turn. Stop named teammates when finished.\n\nVerify claims against real evidence: run relevant commands and tests. Follow a disciplined consultation ladder for unresolved decisions: (1) direct code inspection, search, builds, and tests settle factual questions; (2) \`advisor\` ` + advisorDescriptor + " for transcript-aware framing checks or trajectory guidance; (3) `oracle` " + oracleDescriptor + " for self-contained technical/architectural trade-offs" + astraStep;
2947
+ return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + (isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly: use \`Explore\` for discovery spanning more than a couple of files, \`Plan\` in plan mode or for complex sequencing, \`General-Purpose\` for mixed multi-step execution, and \`reviewer\` for behavior-changing or risk-sensitive changes. Handle trivial and surgical edits directly. Stop named teammates when finished.\n\n` : `${profileLabel} launch profile. The lead coordinates execution across specialized roles: delegate broad discovery to \`Explore\` in parallel and do not sweep the repo yourself (read directly only files you will act on); delegate to \`Plan\` in plan mode or when structuring complex multi-step sequencing (\`Plan\` is an advisory planning capability, not an approval gate, and writes handoff-ready steps for \`General-Purpose\`); delegate mixed multi-step execution and Plan handoffs to \`General-Purpose\` in a fresh context; delegate to \`reviewer\` after behavior-changing or risk-sensitive implementation to verify correctness before declaring done; handle trivial and surgical edits directly. Send independent subagent calls in parallel within a single turn. Stop named teammates when finished.\n\n`) + "Pipeline skills (all 200K default): `/gh-gather-context` before planning when context is needed, `/gh-plan` before implementation (waits for user approval), `/gh-implement` after approval with bounded parallel workers and staged review. Skip for trivial work.\n\nVerify claims against real evidence: run relevant commands and tests. Follow a disciplined consultation ladder for unresolved decisions: (1) direct code inspection, search, builds, and tests settle factual questions; (2) `advisor` " + advisorDescriptor + " for transcript-aware framing checks or trajectory guidance; (3) `oracle` " + oracleDescriptor + " for self-contained technical/architectural trade-offs" + astraStep;
2415
2948
  }
2416
2949
  return STANDARD_OPERATING_DEFAULTS_DIGEST;
2417
2950
  }
@@ -2880,9 +3413,18 @@ async function writeInjectedSkill(name, md) {
2880
3413
  * `/gh-orchestrate`, `/gh-floor-keeper`, `/gh-first-mate`). See
2881
3414
  * `docs/floor-raising-agent-surface.md`.
2882
3415
  */
3416
+ /** Pipeline skills for pinned profiles (all 200K default context). */
3417
+ const PIPELINE_SKILLS = [
3418
+ GATHER_CONTEXT_SKILL,
3419
+ PLAN_SKILL,
3420
+ IMPLEMENT_SKILL
3421
+ ];
2883
3422
  /** All injected skills, in dependency order (research underpins the others). */
2884
3423
  const INJECTED_SKILLS = [
2885
3424
  RESEARCH_SKILL,
3425
+ GATHER_CONTEXT_SKILL,
3426
+ PLAN_SKILL,
3427
+ IMPLEMENT_SKILL,
2886
3428
  ORCHESTRATE_SKILL,
2887
3429
  FLOOR_KEEPER_SKILL,
2888
3430
  WORKER_SKILL,
@@ -2892,8 +3434,14 @@ const INJECTED_SKILLS = [
2892
3434
  FIRST_MATE_CONDUCT_SKILL
2893
3435
  ];
2894
3436
  function injectedSkillsForLaunch(selection) {
2895
- if (selection.profileId === "fast" || selection.profileId === "cheap" || selection.profileId === "cheap1m" || selection.profileId === "cheapest") return [];
2896
- if (selection.profileId === "max") return selection.firstMateEnabled ? INJECTED_SKILLS.filter((skill) => skill.name.startsWith("gh-first-mate")) : [];
3437
+ if (isPipelineSkillProfile(selection.profileId)) {
3438
+ if (selection.profileId === "max") {
3439
+ const pipeline = PIPELINE_SKILLS.slice();
3440
+ if (selection.firstMateEnabled) return [...pipeline, ...INJECTED_SKILLS.filter((skill) => skill.name.startsWith("gh-first-mate"))];
3441
+ return pipeline;
3442
+ }
3443
+ return PIPELINE_SKILLS.slice();
3444
+ }
2897
3445
  if (!selection.workerSkillsActive) return [];
2898
3446
  return INJECTED_SKILLS.filter((skill) => selection.firstMateEnabled || !skill.name.startsWith("gh-first-mate"));
2899
3447
  }
@@ -2964,4 +3512,4 @@ async function injectAttributionSuppressionIntoSettingsFile(settingsPath) {
2964
3512
  //#endregion
2965
3513
  export { writePeerMcpRuntimeFiles as C, workersKeyOf as S, sanitizeServeSettingsEnv as _, appendPeerAwarenessToMirroredClaudeMd as a, resolveCodexCliBackend as b, buildOperatingDefaultsDirective as c, prependStyleDirectiveToMirroredClaudeMd as d, buildArtifactReviewSkill as f, planModeAllowRules as g, injectAllowRules as h, writeInjectedSkill as i, prependArtifactPanelDirectiveToMirroredClaudeMd as l, configureServeDefaultPermissionMode as m, INJECTED_SKILLS as n, appendToolbeltAwarenessToMirroredClaudeMd as o, SEAMLESS_BUILTIN_TOOLS as p, injectedSkillsForLaunch as r, buildOperatingDefaultsDigest as s, injectAttributionSuppressionIntoSettingsFile as t, prependOperatingDefaultsToMirroredClaudeMd as u, BUILTIN_SUBAGENT_DEFINITIONS as v, resolveGroupKeysFromMirror as x, injectPeerMcpIntoMirror as y };
2966
3514
 
2967
- //# sourceMappingURL=attribution-settings-z-U2L7ab.js.map
3515
+ //# sourceMappingURL=attribution-settings-sTUXqFzx.js.map