github-router 0.3.323 → 0.3.325

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/dist/{attribution-settings-CX-kIrT8.js → attribution-settings-IbcZgAEf.js} +604 -11
  2. package/dist/attribution-settings-IbcZgAEf.js.map +1 -0
  3. package/dist/browser-ext/manifest.json +1 -1
  4. package/dist/{claude-Cq1xGzbI.js → claude-CTG8JDGA.js} +17 -8
  5. package/dist/claude-CTG8JDGA.js.map +1 -0
  6. package/dist/{codex-INt1hc_5.js → codex-C_FwSkkR.js} +2 -2
  7. package/dist/{codex-INt1hc_5.js.map → codex-C_FwSkkR.js.map} +1 -1
  8. package/dist/hooks.mjs +26 -0
  9. package/dist/hooks.sha256 +1 -1
  10. package/dist/{internal-first-mate-guard-mvfZal_M.js → internal-first-mate-guard-Cj_g7RFr.js} +2 -2
  11. package/dist/{internal-first-mate-guard-mvfZal_M.js.map → internal-first-mate-guard-Cj_g7RFr.js.map} +1 -1
  12. package/dist/{internal-first-mate-guard-Dvd1vZ4l.js → internal-first-mate-guard-ZqPHrYlh.js} +1 -1
  13. package/dist/{internal-worker-guard-BSoS4wzV.js → internal-worker-guard-Dd14aDXQ.js} +2 -2
  14. package/dist/{internal-worker-guard-BSoS4wzV.js.map → internal-worker-guard-Dd14aDXQ.js.map} +1 -1
  15. package/dist/main.js +6 -6
  16. package/dist/{serve-TiUhvpVp.js → serve-CJclFGKH.js} +6 -5
  17. package/dist/serve-CJclFGKH.js.map +1 -0
  18. package/dist/{server-setup-CgbHmupe.js → server-setup-BfZEQOXB.js} +53 -2
  19. package/dist/server-setup-BfZEQOXB.js.map +1 -0
  20. package/dist/{start-DNzOFabp.js → start-kLRzQRdu.js} +2 -2
  21. package/dist/{start-DNzOFabp.js.map → start-kLRzQRdu.js.map} +1 -1
  22. package/dist/{worker-dispatch-DwS99DXz.js → worker-dispatch-CcQygiAA.js} +27 -1
  23. package/dist/{worker-dispatch-DwS99DXz.js.map → worker-dispatch-CcQygiAA.js.map} +1 -1
  24. package/package.json +1 -1
  25. package/dist/attribution-settings-CX-kIrT8.js.map +0 -1
  26. package/dist/claude-Cq1xGzbI.js.map +0 -1
  27. package/dist/serve-TiUhvpVp.js.map +0 -1
  28. package/dist/server-setup-CgbHmupe.js.map +0 -1
@@ -1,12 +1,12 @@
1
1
  import { Ln as BALANCED_PROFILE_NATIVE_AGENT_NAMES, Nn as oneMContextDisabled, Pn as withOneMSuffix, Rn as BALANCED_PROFILE_NATIVE_EFFORTS, Un as CHEAPEST_PROFILE_NATIVE_AGENT_NAMES, Wn as CHEAPEST_PROFILE_NATIVE_EFFORTS, Xn as CHEAP_PROFILE_NATIVE_EFFORTS, Yn as CHEAP_PROFILE_NATIVE_AGENT_NAMES, a as buildAgentPrompt, fn as CONDENSED_OPERATING_SEQUENCE, l as maxPersonasFor, n as MCP_GROUPS, pn as DEFINITION_OF_GREATNESS, t as GROUP_META, u as personasFor } from "./peer-mcp-personas-DcAErb_d.js";
2
2
  import { a as isUnderClaudeConfigMirror, f as writeRuntimeFileSecure, t as PATHS } from "./paths-BnZwolac.js";
3
- import { C as CHEAP_REVIEWER_ALIAS_ID, S as CHEAP_PLAN_ALIAS_ID, _ as CHEAPEST_GENERAL_PURPOSE_ALIAS_ID, b as CHEAP_EXPLORE_ALIAS_ID, f as BALANCED_EXPLORE_ALIAS_ID, g as CHEAPEST_EXPLORE_ALIAS_ID, h as BALANCED_REVIEWER_ALIAS_ID, m as BALANCED_PLAN_ALIAS_ID, p as BALANCED_GENERAL_PURPOSE_ALIAS_ID, v as CHEAPEST_PLAN_ALIAS_ID, w as LUNA_SCOUT_ALIAS_ID, x as CHEAP_GENERAL_PURPOSE_ALIAS_ID, y as CHEAPEST_REVIEWER_ALIAS_ID } from "./server-setup-CgbHmupe.js";
3
+ import { C as CHEAP_REVIEWER_ALIAS_ID, S as CHEAP_PLAN_ALIAS_ID, _ as CHEAPEST_GENERAL_PURPOSE_ALIAS_ID, b as CHEAP_EXPLORE_ALIAS_ID, f as BALANCED_EXPLORE_ALIAS_ID, g as CHEAPEST_EXPLORE_ALIAS_ID, h as BALANCED_REVIEWER_ALIAS_ID, m as BALANCED_PLAN_ALIAS_ID, p as BALANCED_GENERAL_PURPOSE_ALIAS_ID, v as CHEAPEST_PLAN_ALIAS_ID, w as LUNA_SCOUT_ALIAS_ID, x as CHEAP_GENERAL_PURPOSE_ALIAS_ID, y as CHEAPEST_REVIEWER_ALIAS_ID } from "./server-setup-BfZEQOXB.js";
4
4
  import { c as FAST_PROFILE_MODELS, d as FAST_PROFILE_NATIVE_MODELS, l as FAST_PROFILE_NATIVE_AGENT_NAMES, u as FAST_PROFILE_NATIVE_EFFORTS } from "./fast-profile-contract-DmsyQQrw.js";
5
5
  import { C as MAX_PARALLELISM_RULE, S as MAX_COORDINATOR_PROMPT, T as maxNativePrompt, a as MAX_PROFILE_NATIVE_AGENT_NAMES, i as MAX_PROFILE_MODELS, o as MAX_PROFILE_NATIVE_EFFORTS, s as MAX_PROFILE_NATIVE_MODELS, w as maxNativeDescription, x as MAX_COORDINATOR_DESCRIPTION } from "./max-profile-contract-Bl7I_EOC.js";
6
6
  import "./self-invocation-opSXHEs0.js";
7
7
  import { a as STRIPPED_AUTH_ROUTING_ENV_KEYS, n as buildCodexProviderConfigFlags } from "./provision-B-a5KUCL.js";
8
8
  import { n as buildWorkspaceHeaderHelperCommand } from "./mcp-workspace-header-BIW0SRgs.js";
9
- import { a as dispatcherAgentName, c as dispatcherTools, n as activeDispatchModes, o as dispatcherDescription, s as dispatcherPrompt } from "./worker-dispatch-DwS99DXz.js";
9
+ import { a as dispatcherAgentName, c as dispatcherTools, n as activeDispatchModes, o as dispatcherDescription, s as dispatcherPrompt } from "./worker-dispatch-CcQygiAA.js";
10
10
  import consola from "consola";
11
11
  import path from "node:path";
12
12
  import { randomBytes } from "node:crypto";
@@ -2069,6 +2069,203 @@ Return a compact final checkpoint:
2069
2069
  `
2070
2070
  };
2071
2071
  //#endregion
2072
+ //#region src/lib/injected-skills/gather-context-skill.ts
2073
+ const GATHER_CONTEXT_SKILL = {
2074
+ name: "gh-gather-context",
2075
+ md: `---
2076
+ name: gh-gather-context
2077
+ description: Bounded context gathering for non-trivial asks: decomposes the ask, runs lexical code searches to identify relevant files, dispatches bounded parallel explore workers to gather evidence, stitches results into a freshness-stamped context brief plus a compact version. Use when grounded context is needed before planning or changing code.
2078
+ user-invocable: true
2079
+ ---
2080
+
2081
+ # gh-gather-context: bounded context gathering
2082
+
2083
+ Use this skill when a non-trivial ask needs grounded context before planning.
2084
+ All reasoning runs at the 200K default window: the lead and every explore
2085
+ worker use the Luna model at high effort with bare slugs (no 1M accounting).
2086
+ Output is a durable full brief plus a compact downstream version.
2087
+
2088
+ ## Hard bounds
2089
+
2090
+ - Maximum rounds: 3.
2091
+ - Maximum parallel explore workers per round: 6.
2092
+ - Maximum lexical searches per round: 10.
2093
+ - Maximum follow-up reads per round: 5.
2094
+ - Terminate at the first of saturation or a cap.
2095
+ - On cap-hit, return with open unknowns flagged as residual. Do not loop forever.
2096
+
2097
+ ## Evidence tags
2098
+
2099
+ Use these exact tags on every finding and claim:
2100
+
2101
+ - verified-executable: reproduced the symptom, ran the failing test, or ran a check that directly proves the claim. This is the only deterministic confidence tag.
2102
+ - verified-source: read the actual source, config, logs, docs, or primary artifact and cited the relevant locations. This is model-mediated and can still be wrong.
2103
+ - cross-lab-agreed: a different-lab reviewer independently agreed with the claim. This reduces correlated blind spots but is advisory.
2104
+ - unverified: plausible but not confirmed; treat as residual risk.
2105
+
2106
+ ## Procedure
2107
+
2108
+ 1. Restate the ask and define the research target.
2109
+ - Identify whether this is a bug, feature, refactor, incident, or design question.
2110
+ - Name the expected downstream consumer: planner, implementer, or user.
2111
+
2112
+ 2. Decompose the ask into searchable entities.
2113
+ - Extract symbols, filenames, error strings, routes, flags, config keys, and types.
2114
+ - Define what must be true for a correct implementation.
2115
+
2116
+ 3. Run lexical search first, in parallel, in a single turn.
2117
+ - Use mcp__search__code lexically for exact symbols, filenames, errors, routes, flags, and config keys.
2118
+ - Use mcp__search__code semantically only to find concepts, then refine to lexical.
2119
+ - Use git log and git blame when authorship, regression timing, or intent matters.
2120
+ - Use mcp__search__web for upstream APIs, package behavior, protocol docs, or public issues.
2121
+
2122
+ 4. Decompose into bounded explore workers.
2123
+ - Cluster search results into at most 6 coherent investigation areas.
2124
+ - For each area, write a narrow brief: the specific question, the expected artifact, and the files to focus on.
2125
+ - Dispatch ALL explore workers in a single turn via the Agent tool (subagent_type worker-explore). Each runs read-only at the 200K default window and returns a summary with an evidence table and file:line citations. Pass maxWallClockMs 180000 on every worker call so a hung worker is reaped after 3 minutes instead of blocking its slot.
2126
+ - Keep worker results summarized; do not paste every detail into the main context.
2127
+
2128
+ 5. Stitch and verify.
2129
+ - Collect all explore results and deduplicate file references.
2130
+ - Run at most 5 targeted follow-up reads for gaps, in parallel.
2131
+ - Dispatch the worker-review subagent (via the Agent tool, maxWallClockMs 300000) to confirm source-reading for load-bearing claims.
2132
+ - Form a root-cause hypothesis or integration map, and state what would falsify it.
2133
+
2134
+ 6. Run a completeness pass.
2135
+ - Ask: what do we still not know?
2136
+ - Ask: what claim, if false, would break the conclusion?
2137
+ - Ask: have we checked primary sources for every load-bearing claim?
2138
+ - If no material unknowns remain and the root cause is at least verified-source, stop for saturation.
2139
+
2140
+ 7. Persist outputs under .github-router/context/<slug>/ and close the stage.
2141
+ - context.md: full brief with the ask decomposition, searches run, worker reports, evidence table, hypothesis, freshness metadata (HEAD commit, working-tree diff hash, timestamp, repo path), residuals, and full citations.
2142
+ - context.compact.md: downstream consumable with a one-paragraph ask summary, key files with one-line purposes, critical constraints (APIs, types, patterns, forbidden changes), integration seams, and residual risks.
2143
+ - .complete: stage-completion marker written ONLY after every dispatched explore worker has returned or been recorded as stopped, and both briefs are on disk. No downstream stage (/gh-plan, /gh-implement, /gh-swe-pipeline stage 2+) may start until this marker exists.
2144
+ - Downstream phases read by pointer and check freshness instead of re-injecting the whole brief.
2145
+
2146
+ ## Waterfall rule
2147
+
2148
+ This stage owns all explore workers until it completes. Do NOT return while
2149
+ any explore worker is still running unless you explicitly record it as
2150
+ stopped (saturation reached or its area superseded) with the reason. If the
2151
+ brief already saturates the ask, stop remaining workers first (no further
2152
+ follow-ups; treat partial output as superseded), then write the marker. A
2153
+ plan built on shifting evidence wastes more than a stopped worker costs.
2154
+
2155
+ ## Return format
2156
+
2157
+ Return a compact brief, not the whole dump:
2158
+
2159
+ - Context files: paths to context.md and context.compact.md.
2160
+ - Freshness: HEAD commit, diff hash, timestamp.
2161
+ - Termination: saturated or cap-hit; if cap-hit, name the cap.
2162
+ - Summary: 3-8 bullets with confidence tags.
2163
+ - Evidence table: claim, tag, primary source or command, reviewer status.
2164
+ - Residual unknowns: explicit list, or none.
2165
+ - Downstream guidance: recommended next action and what must be rechecked if the tree changes.
2166
+
2167
+ ## Non-goals
2168
+
2169
+ - Do not present verified-source or cross-lab-agreed as deterministic.
2170
+ - Do not hide open unknowns because the answer looks useful.
2171
+ - Do not keep searching after the cap.
2172
+ - Do not paste the entire persisted brief into later turns unless the user asks.
2173
+ - Do not finish without the .complete marker: a marker-less brief is not a completed stage.
2174
+ `
2175
+ };
2176
+ //#endregion
2177
+ //#region src/lib/injected-skills/implement-skill.ts
2178
+ const IMPLEMENT_SKILL = {
2179
+ name: "gh-implement",
2180
+ md: `---
2181
+ name: gh-implement
2182
+ description: Parallel implementation of an approved plan using bounded Luna workers with isolated worktrees: each worker implements its task, self-tests, self-reviews, and returns a patch; the lead aggregates into a unified diff, runs staged review, and returns the final diff with a report. Use when a user-approved plan is ready for execution.
2183
+ user-invocable: true
2184
+ ---
2185
+
2186
+ # gh-implement: bounded parallel implementation with staged review
2187
+
2188
+ Use this skill only after /gh-plan produced a user-approved plan.md. All
2189
+ implementation runs at the 200K default window: the lead and every task worker
2190
+ use the Luna model at max effort with bare slugs (no 1M accounting). Review is
2191
+ staged: a Luna max pass first, then a Sol medium pass only for major issues.
2192
+
2193
+ ## Hard bounds
2194
+
2195
+ - Maximum concurrent implement workers: 8.
2196
+ - Maximum retries per task: 2.
2197
+ - Maximum review-fix cycles: 2.
2198
+ - Worktrees are auto-removed on success and retained on failure for debugging.
2199
+
2200
+ ## Stage gate 0 (do this BEFORE any implementation)
2201
+
2202
+ 1. Verify the plan stage is finished: .github-router/plans/<slug>/plan.md AND
2203
+ its .complete marker both exist, with an explicit user-approval record. If
2204
+ any is missing, STOP: do not implement an unapproved plan. Finish or
2205
+ re-invoke planning first.
2206
+ 2. Check freshness: if HEAD or the working-tree diff hash moved since the
2207
+ plan was approved, re-verify stale load-bearing assumptions before
2208
+ dispatching workers.
2209
+ 3. Stop the previous stage: send no further follow-ups to any lingering plan
2210
+ workers (worker-plan follow-ups) and record them as stopped with the
2211
+ reason. Implementing while planning still runs builds on a moving target
2212
+ and wastes both stages. Only advance once every plan worker has returned
2213
+ or is recorded as stopped.
2214
+
2215
+ ## Procedure
2216
+
2217
+ 1. Parse the approved plan.
2218
+ - Read plan.md fully.
2219
+ - Group tasks by parallelGroup; order groups by dependency.
2220
+ - For each group, prepare an isolated git worktree per task plus a narrow task brief (task spec, relevant context excerpt, acceptance criteria, verification commands).
2221
+
2222
+ 2. Dispatch bounded implement workers, one parallel batch per group.
2223
+ - Dispatch ALL tasks in the group in a single turn via the Agent tool (subagent_type worker-implement, with worktree isolation, maxWallClockMs 600000 per task so a hung worker is reaped after 10 minutes instead of blocking its slot).
2224
+ - Each worker runs at the 200K default window and must self-contain its work:
2225
+ a. Implement the change.
2226
+ b. Run the task verification commands (tests, typecheck, lint).
2227
+ c. Self-review against the acceptance criteria.
2228
+ d. Fix any self-found issues (at most 2 internal fix cycles).
2229
+ e. Return the patch plus test results and self-review notes.
2230
+ - Do NOT dispatch the same task twice (no dedup exists); a retry is a new dispatch only after a recorded failure.
2231
+ - For a big artifact, have the worker write it to a file and return the path.
2232
+
2233
+ 3. Aggregate and validate.
2234
+ - Collect all patches and apply them sequentially to the main worktree (or merge the worktrees).
2235
+ - Run the full relevant validation: test suite, typecheck, and lint.
2236
+ - If any task fails validation, route it back to an implement worker (at most 2 retries per task). If it still fails, checkpoint with the failure as residual risk instead of pretending it is solved.
2237
+
2238
+ 4. Run staged review.
2239
+ - Pass 1 (always): dispatch the worker-review subagent (via the Agent tool, maxWallClockMs 300000) over the unified diff for correctness against acceptance criteria, code quality and consistency, security and performance regressions, and test coverage. Categorize findings as minor (style, nits) or major (logic, architecture).
2240
+ - If pass 1 finds no major issues, finish here.
2241
+ - Pass 2 (major issues only): dispatch a fix worker (via the Agent tool, maxWallClockMs 300000) at the 200K default window using the Sol model at medium effort with the flagged areas, the failing checks, and the pass-1 findings. It returns fixed patches or an explicit escalate-to-user with reasons.
2242
+
2243
+ 5. Finalize.
2244
+ - Apply any review fixes and re-run full validation.
2245
+ - Produce the unified diff for the whole plan.
2246
+ - Write .github-router/plans/<slug>/implementation-report.md with task completion status, test results summary, review findings and resolutions, the final diff path, and residual risks.
2247
+
2248
+ ## Return format
2249
+
2250
+ Return:
2251
+
2252
+ - Unified diff path.
2253
+ - Implementation report path.
2254
+ - Task completion status per task id.
2255
+ - Test, typecheck, and lint results.
2256
+ - Review summary (pass 1 findings; pass 2 findings and fixes if used).
2257
+ - Final residual risks and next action.
2258
+
2259
+ ## Non-goals
2260
+
2261
+ - Do not start without a user-approved plan.md plus its .complete approval record.
2262
+ - Do not start while plan workers still run; stop them first.
2263
+ - Do not serialize work that has no data dependency; independent tasks in a group run concurrently.
2264
+ - Do not nest workflow invocations: workers are internal sessions and must not re-invoke /gh-implement (or any /gh-* pipeline skill).
2265
+ - Do not claim completeness when retries or review cycles are exhausted with open failures.
2266
+ `
2267
+ };
2268
+ //#endregion
2072
2269
  //#region src/lib/injected-skills/orchestrate-skill.ts
2073
2270
  const ORCHESTRATE_SKILL = {
2074
2271
  name: "gh-orchestrate",
@@ -2207,6 +2404,110 @@ Return:
2207
2404
  `
2208
2405
  };
2209
2406
  //#endregion
2407
+ //#region src/lib/injected-skills/plan-skill.ts
2408
+ const PLAN_SKILL = {
2409
+ name: "gh-plan",
2410
+ md: `---
2411
+ name: gh-plan
2412
+ description: Holistic planning from gathered context: ingests the context brief, creates a scoped modular ordered implementation plan with explicit tasks for cheap implementation workers, surfaces open questions for user approval, and persists plan.md. Use when a non-trivial change needs a reviewed plan before implementation.
2413
+ user-invocable: true
2414
+ ---
2415
+
2416
+ # gh-plan: holistic planning for cheap implementation
2417
+
2418
+ Use this skill after /gh-gather-context (or when equivalent context is already
2419
+ available) and before any implementation. The planner runs at the 200K default
2420
+ window using the Sol model at medium effort. The plan must be scoped, modular,
2421
+ non-overlapping, and ordered, with enough detail for Luna implementation
2422
+ workers to execute each task in isolation. User approval is mandatory before
2423
+ implementation.
2424
+
2425
+ ## Prerequisites
2426
+
2427
+ - A freshness-stamped context brief from /gh-gather-context, or equivalent context.
2428
+ - Read context.compact.md first; read context.md sections on demand (residual unknowns, evidence table).
2429
+
2430
+ ## Stage gate 0 (do this BEFORE any planning)
2431
+
2432
+ 1. Verify the context stage is finished: .github-router/context/<slug>/.complete
2433
+ exists. If the marker is missing, STOP: do not plan on a partial brief.
2434
+ Finish or re-invoke gathering first.
2435
+ 2. Check freshness: if HEAD or the working-tree diff hash moved since the
2436
+ brief's stamp, re-verify stale load-bearing claims before using them.
2437
+ 3. Stop the previous stage: send no further follow-ups to any lingering
2438
+ gather-context explore workers and record them as stopped with the reason.
2439
+ Planning while explore workers still run builds on shifting evidence and
2440
+ wastes both stages. Only advance once every gather worker has returned or
2441
+ is recorded as stopped.
2442
+
2443
+ ## Hard bounds
2444
+
2445
+ - Maximum tasks: 20.
2446
+ - Maximum parallel groups: 5.
2447
+ - Keep planner input well under the 200K window (target at most around 150K tokens of context) so there is headroom for reasoning and output. This is self-discipline, not an enforced cap: prefer the compact brief and read full sections only on demand.
2448
+
2449
+ ## Procedure
2450
+
2451
+ 1. Ingest context.
2452
+ - Read context.compact.md fully.
2453
+ - Read the evidence table and residual unknowns from context.md.
2454
+ - Identify acceptance criteria, constraints, integration seams, and forbidden changes.
2455
+
2456
+ 2. Build a blind-spot table before decomposing.
2457
+ - Wrong-spec risk: judgment-only, mitigated only by user-blessed acceptance criteria.
2458
+ - Root-cause risk: executable-checkable if reproduced or covered by a failing test; otherwise advisory.
2459
+ - Integration risk: usually source-verified plus tests where possible.
2460
+ - Regression risk: executable-checkable when tests, typecheck, or lint cover it.
2461
+ - Review risk: advisory cross-lab review reduces correlated blind spots.
2462
+ - Concurrency or merge risk: source-verified and sometimes executable-checkable.
2463
+ - Missing-test risk: executable-checkable only after a test exists and runs.
2464
+ - Tag every blind spot as executable-checkable or judgment-only.
2465
+
2466
+ 3. Decompose into minimal safe increments.
2467
+ - Each task touches a single file or a tightly coupled file group.
2468
+ - Each task states input artifacts, output artifact, acceptance criteria, verification commands, and rollback concern.
2469
+ - Order tasks by dependency (topological sort); tasks with no data dependency share a parallel group.
2470
+ - Keep each task small enough for one Luna worker at the 200K window (target at most 50K context tokens of relevant files per task).
2471
+ - If the ask needs discovery follow-ups, delegate them to worker-explore background subagents (via the Agent tool, maxWallClockMs 180000) rather than bloating the plan.
2472
+
2473
+ 4. Surface open questions before finalizing.
2474
+ - Ask about ambiguous acceptance criteria, design decisions with multiple valid approaches, risk tolerance, and test strategy.
2475
+ - Present a short candidate list for confirmation where possible.
2476
+
2477
+ 5. Persist the plan to .github-router/plans/<slug>/plan.md.
2478
+ - Ask summary and user-blessed acceptance criteria.
2479
+ - Blind-spot table with executable-checkable or judgment-only tags.
2480
+ - Ordered task list with ids, files, dependencies, parallel groups, acceptance criteria, verification commands, rollback concerns, and estimated context tokens.
2481
+ - Open questions and user answers.
2482
+ - Cost estimate: task count, parallel groups, and context tokens.
2483
+ - Residual risks.
2484
+
2485
+ 6. Checkpoint with the user and wait for explicit approval.
2486
+ - Present the goal, acceptance criteria, task-to-group map, per-task blind spot killed, residual risks, and cost estimate.
2487
+ - If the user rejects scope or cost, downshift to the smallest plan that kills the important blind spots.
2488
+ - Do not proceed to implementation without approval.
2489
+ - Write .github-router/plans/<slug>/.complete ONLY after explicit user approval, recording the approval in the marker. A plan without an approval record is NOT complete, and no downstream stage (/gh-implement, /gh-swe-pipeline stage 3) may start without it.
2490
+
2491
+ ## Return format
2492
+
2493
+ Return:
2494
+
2495
+ - Plan file: path to the durable plan.md.
2496
+ - Task count and parallel groups.
2497
+ - Open questions and user answers.
2498
+ - Cost estimate.
2499
+ - Residual risks and next action (implementation only after approval).
2500
+
2501
+ ## Non-goals
2502
+
2503
+ - Do not edit implementation files while planning; in plan mode, produce the plan and acceptance criteria only.
2504
+ - Do not present judgment-only conclusions as executable guarantees.
2505
+ - Do not hide open unknowns because the plan looks complete.
2506
+ - Do not start planning while gather-context workers still run; stop them first.
2507
+ - Do not dispatch implement workers from planning: stages never overlap.
2508
+ `
2509
+ };
2510
+ //#endregion
2210
2511
  //#region src/lib/injected-skills/research-skill.ts
2211
2512
  const RESEARCH_SKILL = {
2212
2513
  name: "gh-research",
@@ -2318,6 +2619,117 @@ Return a compact brief, not the whole research dump:
2318
2619
  `
2319
2620
  };
2320
2621
  //#endregion
2622
+ //#region src/lib/injected-skills/swe-pipeline-skill.ts
2623
+ const SWE_PIPELINE_SKILL = {
2624
+ name: "gh-swe-pipeline",
2625
+ md: `---
2626
+ name: gh-swe-pipeline
2627
+ description: Strict sequential SWE pipeline for non-trivial code changes: runs gather-context to completion, then plan with user approval, then implement with staged review. Each stage waits for the previous to finish fully, stopping leftover workers before advancing. Use when the user wants the full structured engineering workflow in one command.
2628
+ user-invocable: true
2629
+ ---
2630
+
2631
+ # gh-swe-pipeline: strict sequential SWE workflow
2632
+
2633
+ Use this skill when the user invokes /gh-swe-pipeline for a non-trivial code
2634
+ change. It coordinates the three pipeline stages in STRICT SEQUENCE. No two
2635
+ stages ever overlap: each stage runs to completion, its workers are all
2636
+ finished or explicitly stopped, and its completion artifact exists before the
2637
+ next stage starts.
2638
+
2639
+ All work runs at the 200K default window with bare slugs (no 1M accounting):
2640
+ gather-context uses Luna high, plan uses Sol medium, implement and review
2641
+ pass 1 use Luna max, review pass 2 uses Sol medium.
2642
+
2643
+ ## The one rule
2644
+
2645
+ WATERFALL ONLY. Never start a stage while the previous stage still has running
2646
+ workers or an unwritten completion artifact. If a stage already has enough
2647
+ evidence to proceed, FIRST stop every still-running worker from the previous
2648
+ stage (let their maxWallClockMs reap them or send no further follow-ups and
2649
+ treat their partial output as superseded), record what was stopped and why,
2650
+ THEN advance. Overlapping stages waste money and produce plans built on
2651
+ shifting evidence. This is the failure the pipeline exists to prevent.
2652
+
2653
+ ## Stage 0: triage (no workers)
2654
+
2655
+ 1. Restate the ask in one sentence.
2656
+ 2. Decide trivial versus non-trivial. Trivial (typo, one-line config read,
2657
+ obvious three-line fix, pure explanation) SKIPS the pipeline: say why and
2658
+ do the work directly. Do not pay orchestration cost as ritual.
2659
+ 3. For non-trivial work, derive a run slug and create
2660
+ .github-router/swe/<slug>/run.md with the ask, the triage verdict, and
2661
+ per-stage status (pending, running, complete, skipped).
2662
+
2663
+ ## Stage 1: gather context (to completion)
2664
+
2665
+ 1. Invoke the gh-gather-context skill and WAIT for its full return. Do not
2666
+ plan, sketch tasks, or dispatch plan workers while it runs.
2667
+ 2. Its completion artifact is
2668
+ .github-router/context/<slug>/context.md plus context.compact.md and a
2669
+ .complete marker. If the marker is missing, the stage is NOT complete:
2670
+ keep waiting or re-invoke; never advance on a partial brief.
2671
+ 3. Early-stop rule: if the returned brief already saturates the ask (root
2672
+ cause at least verified-source, no material unknowns), stop any
2673
+ still-running explore workers (no follow-ups; record them as stopped),
2674
+ accept the brief, and advance. Do not keep searching after saturation.
2675
+ 4. Cap-hit rule: if the brief reports cap-hit with residuals, surface the
2676
+ residuals in run.md and ask the user whether to proceed to planning with
2677
+ the gap or to spend one more bounded round. Do not silently treat a
2678
+ cap-hit brief as complete.
2679
+
2680
+ ## Stage 2: plan (to user approval)
2681
+
2682
+ 1. Precondition check BEFORE invoking gh-plan: the context .complete marker
2683
+ exists AND every gather-context explore worker has returned or been
2684
+ recorded as stopped. If either is false, do not invoke planning. Fix
2685
+ stage 1 first.
2686
+ 2. Invoke the gh-plan skill and WAIT for its full return. Do not dispatch
2687
+ implement workers, sketch diffs, or edit implementation files while it
2688
+ runs. Plan mode means plan and acceptance criteria only.
2689
+ 3. Its completion artifact is .github-router/plans/<slug>/plan.md with a
2690
+ user-approval record (.complete marker written only after explicit
2691
+ approval). A plan without explicit user approval is NOT complete.
2692
+ 4. Present the goal, acceptance criteria, task-to-group map, residual risks,
2693
+ and cost estimate. Wait for explicit approval. If the user rejects scope
2694
+ or cost, downshift to the smallest plan that kills the important blind
2695
+ spots and re-seek approval. NEVER advance to implementation without
2696
+ approval recorded in run.md.
2697
+
2698
+ ## Stage 3: implement (to reviewed diff)
2699
+
2700
+ 1. Precondition check BEFORE invoking gh-implement: plan.md exists, its
2701
+ .complete marker exists, and run.md records explicit user approval. If
2702
+ any is missing, do not invoke implementation. Fix stage 2 first.
2703
+ 2. Stop any lingering plan workers (worker-plan follow-ups) before the
2704
+ first implement dispatch; record them as stopped.
2705
+ 3. Invoke the gh-implement skill and WAIT for its full return: unified diff,
2706
+ implementation report, test/typecheck/lint results, and review summary.
2707
+ 4. If retries or review cycles exhaust with open failures, checkpoint with
2708
+ the failure as residual risk instead of pretending it is solved.
2709
+
2710
+ ## Return format
2711
+
2712
+ Return:
2713
+
2714
+ - Run file: path to .github-router/swe/<slug>/run.md with per-stage status.
2715
+ - Context files: paths to context.md and context.compact.md, freshness
2716
+ (HEAD commit, diff hash, timestamp), termination (saturated or cap-hit).
2717
+ - Plan file: path to plan.md, task count, parallel groups, open questions
2718
+ and user answers, cost estimate, approval record.
2719
+ - Implement output: unified diff path, report path, per-task status, test
2720
+ results, review summary.
2721
+ - Residual risks and next action.
2722
+
2723
+ ## Non-goals
2724
+
2725
+ - Do not run stages in parallel or overlap workers across stages.
2726
+ - Do not advance past a missing completion artifact or a missing approval.
2727
+ - Do not nest pipeline invocations: workers are internal sessions and must
2728
+ not re-invoke /gh-swe-pipeline or any /gh-* pipeline skill.
2729
+ - Do not present judgment-only conclusions as executable guarantees.
2730
+ `
2731
+ };
2732
+ //#endregion
2321
2733
  //#region src/lib/injected-skills/worker-skill.ts
2322
2734
  /**
2323
2735
  * The `/gh-worker` skill: the operating model for the NON-BLOCKING workers
@@ -2378,6 +2790,88 @@ The dispatcher calls the worker once and relays its result verbatim.
2378
2790
  `
2379
2791
  };
2380
2792
  //#endregion
2793
+ //#region src/lib/skill-model-contract.ts
2794
+ /**
2795
+ * Universal skill model contract for the `/gh-gather-context`, `/gh-plan`,
2796
+ * and `/gh-implement` pipeline skills.
2797
+ *
2798
+ * ALL roles run at the 200K DEFAULT context window (bare slugs, no `[1m]`
2799
+ * accounting bracket) on every profile. This module is deliberately
2800
+ * dependency-free so profile contracts, launch validation, worker dispatch,
2801
+ * and the injected skill bodies can all import the same literals without
2802
+ * cycles.
2803
+ *
2804
+ * Model choices (per pipeline design):
2805
+ * - gatherContext lead + explore agents: Luna, high effort
2806
+ * - plan lead: Sol, medium effort
2807
+ * - implement lead + task agents: Luna, max effort
2808
+ * - review pass 1: Luna, max effort
2809
+ * - review pass 2 (major issues only): Sol, medium effort
2810
+ */
2811
+ const SKILL_LUNA_MODEL_ID = "gpt-5.6-luna";
2812
+ const SKILL_SOL_MODEL_ID = "gpt-5.6-sol";
2813
+ Object.freeze({
2814
+ gatherContext: Object.freeze({
2815
+ lead: SKILL_LUNA_MODEL_ID,
2816
+ exploreAgent: SKILL_LUNA_MODEL_ID,
2817
+ leadEffort: "high",
2818
+ agentEffort: "high"
2819
+ }),
2820
+ plan: Object.freeze({
2821
+ lead: SKILL_SOL_MODEL_ID,
2822
+ leadEffort: "medium"
2823
+ }),
2824
+ implement: Object.freeze({
2825
+ lead: SKILL_LUNA_MODEL_ID,
2826
+ taskAgent: SKILL_LUNA_MODEL_ID,
2827
+ leadEffort: "max",
2828
+ agentEffort: "max"
2829
+ }),
2830
+ review: Object.freeze({
2831
+ pass1: Object.freeze({
2832
+ model: SKILL_LUNA_MODEL_ID,
2833
+ effort: "max"
2834
+ }),
2835
+ pass2: Object.freeze({
2836
+ model: SKILL_SOL_MODEL_ID,
2837
+ effort: "medium"
2838
+ })
2839
+ })
2840
+ });
2841
+ Object.freeze({
2842
+ gatherContext: Object.freeze({
2843
+ maxRounds: 3,
2844
+ maxExploreAgentsPerRound: 6,
2845
+ maxLexicalSearchesPerRound: 10,
2846
+ maxFollowUpReadsPerRound: 5
2847
+ }),
2848
+ plan: Object.freeze({
2849
+ maxTasks: 20,
2850
+ maxParallelGroups: 5
2851
+ }),
2852
+ implement: Object.freeze({
2853
+ maxConcurrentAgents: 8,
2854
+ maxRetriesPerTask: 2,
2855
+ maxReviewFixCycles: 2
2856
+ })
2857
+ });
2858
+ /**
2859
+ * Profiles that receive the pipeline skills. Every pinned profile gets
2860
+ * them; `standard` is intentionally excluded (it keeps the existing
2861
+ * research/orchestrate/worker surface).
2862
+ */
2863
+ const PIPELINE_SKILL_PROFILES = [
2864
+ "fast",
2865
+ "max",
2866
+ "cheap",
2867
+ "cheap1m",
2868
+ "cheapest",
2869
+ "balanced"
2870
+ ];
2871
+ function isPipelineSkillProfile(profileId) {
2872
+ return PIPELINE_SKILL_PROFILES.includes(profileId);
2873
+ }
2874
+ //#endregion
2381
2875
  //#region src/lib/injected-skills/artifact-review-skill.ts
2382
2876
  function buildArtifactReviewSkill(peersKey = "peers") {
2383
2877
  const toolPrefix = `mcp__${peersKey}__artifact_`;
@@ -2492,6 +2986,75 @@ You are running inside an ai-or-die tab, so the \`${toolPrefix}*\` tools drive a
2492
2986
  * dashes and does not mention any Claude / Anthropic attribution.
2493
2987
  */
2494
2988
  const STYLE_DIRECTIVE = "Write concisely without losing detail. Use a natural human voice. Avoid em dashes. Do not attribute work to Claude, AI, LLM, or Anthropic anywhere (commits, PRs, issues, code, comments, docs).";
2989
+ /**
2990
+ * Operating-defaults directive injected at the TOP of the mirrored CLAUDE.md.
2991
+ * The main agent's system prompt (`--append-system-prompt`) gets
2992
+ * OPERATING_DEFAULTS_DIGEST instead, with this full statement available through
2993
+ * CLAUDE.md. Three defaults, layered under the user's own
2994
+ * direction and the domain's standards as addons (on a direct conflict the
2995
+ * user's direction wins, then the domain standard, then the default below):
2996
+ *
2997
+ * 1. Orchestrate (strong default): delegate the heavy / parallel /
2998
+ * context-heavy work to the right subagent / worker / model, keeping the
2999
+ * main context free to reason and collaborate with the user, while still
3000
+ * doing trivial / surgical / last-mile work directly (delegating that
3001
+ * would only add relay-fidelity loss + latency).
3002
+ * 2. Adversarial review: WHEN a peer critic earns its keep and, equally
3003
+ * important, when reaching for one is ritual rather than review. Same
3004
+ * failure shape the delegation default had: "consult a critic for
3005
+ * non-trivial changes" is unfalsifiable in advance, so it collapses into
3006
+ * either never (four of four unprimed agents) or always (worse than
3007
+ * never). The discriminator is whether the conclusion still turns on
3008
+ * judgment once the direct evidence is in: a consequential recommendation
3009
+ * cannot be run, which is exactly where confabulation hides, while a
3010
+ * tracing question a search already proved gains nothing from a second
3011
+ * model re-deriving it. The roster, the lens-to-artifact match, the
3012
+ * advisor-complements-rather-than-substitutes distinction, and the
3013
+ * do-not-anchor-the-critic rule live here; the digest carries only the
3014
+ * trigger and the ritual exclusion.
3015
+ * 3. Excellence lens: the principles stated plainly and concretely (radical
3016
+ * simplicity + real-user focus; whole-system first-principles thinking that
3017
+ * anticipates scale; work back from the customer outcome). Named exemplars
3018
+ * were dropped per the injected-surface review: a named entity is a dense,
3019
+ * high-variance vector that pulls in persona mannerisms at top salience, and
3020
+ * the guidance favors specific functional framing over comparison, so
3021
+ * specificity carries the vividness instead.
3022
+ * 4. Engineering excellence: quality / robustness / maintainability over
3023
+ * development cost; reproduce a bug end-to-end (as a real user hits it)
3024
+ * before fixing so the fix targets the real cause; a pixel-perfect UI bar;
3025
+ * and fix any lint error / test failure / flake on sight, whoever caused it,
3026
+ * folded into the current work rather than derailing the user's task (the
3027
+ * scope guardrail keeps proactive quality from becoming yak-shaving). The
3028
+ * digest carries a one-line form; the full statement lives here so it does
3029
+ * not cost the context window every turn.
3030
+ *
3031
+ * Self-referentially compliant with the style directive: no em dashes, no
3032
+ * Claude / Anthropic attribution.
3033
+ *
3034
+ * Availability-aware: four of the natives (`scout`, `implementer-fast`,
3035
+ * `reviewer-fast`, and `general-purpose-fast`) are DROPPED rather than
3036
+ * downgraded when no model in
3037
+ * their chain resolves, so naming them unconditionally here would tell the lead to delegate
3038
+ * to an agent that has no `.md` file and is absent from the Task
3039
+ * `subagent_type` enum. Build the directive with
3040
+ * `buildOperatingDefaultsDirective` and the same availability booleans used for
3041
+ * the `.md` generation and the awareness snippet; the exported const below is
3042
+ * the all-available form, kept for callers and tests that do not model a thin
3043
+ * catalog.
3044
+ */
3045
+ /**
3046
+ * Pipeline skills (`/gh-gather-context`, `/gh-plan`, `/gh-implement`,
3047
+ * `/gh-swe-pipeline`) injected ONLY for `--swe` launches on pinned profiles
3048
+ * (fast, max, cheap, cheap1m, cheapest, balanced). Standard is intentionally
3049
+ * excluded. Without `--swe` the skill files are not written and this text is
3050
+ * not referenced. All skill models run at the 200K default window (bare
3051
+ * slugs, no 1M accounting): gatherContext uses Luna high for the lead and
3052
+ * every explore worker (bounded: 3 rounds, 6 workers per round); plan uses
3053
+ * Sol medium; implement uses Luna max for the lead and every task worker
3054
+ * (bounded: 8 concurrent, 2 retries per task, isolated worktrees); review is
3055
+ * staged Luna max then Sol medium for major issues only.
3056
+ */
3057
+ const PIPELINE_SKILLS_AWARENESS = "Pipeline skills (all 200K default context). Prefer `/gh-swe-pipeline`: it runs the stages below in strict sequence, each completing before the next starts, with user approval before implementation. Stages in order: (1) `/gh-gather-context` BEFORE planning when grounded context is needed: decomposes the ask, runs lexical search, fans out to bounded Luna-high explore workers, writes context.md plus context.compact.md; (2) `/gh-plan` AFTER context and BEFORE implementation: ingests the brief with Sol-medium, produces a scoped modular ordered plan.md, surfaces open questions, waits for user approval; (3) `/gh-implement` AFTER plan approval: bounded parallel Luna-max task workers in isolated worktrees (each self-tests and self-reviews), aggregates a unified diff, runs staged review (Luna max, then Sol medium for major issues only). Skip for trivial work. Never implement without an approved plan. Never overlap stages.";
2495
3058
  /** Oxford-comma join: "a", "a and b", "a, b, and c". */
2496
3059
  function joinClauses(parts) {
2497
3060
  if (parts.length <= 1) return parts[0] ?? "";
@@ -2561,7 +3124,8 @@ function buildOperatingDefaultsDirective(opts = {}) {
2561
3124
  if (opts.profile === "max") {
2562
3125
  const peersKey = opts.peersKey ?? opts.groupKeys?.peers ?? "peers";
2563
3126
  const artifactClause = opts.artifactAvailable ? ` Live human review is available in the artifact panel via \`mcp__${peersKey}__artifact_*\`.` : "";
2564
- return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Handle narrow, obvious, surgical, and single-command work directly; delegate a bounded workstream when it is broad, slow, context-heavy, or benefits from a genuinely independent perspective. " + MAX_PARALLELISM_RULE + " Brief each role with the desired outcome, relevant context, constraints, expected evidence, and verification. Use the roster as complementary capabilities, never as a required Explore → Plan → implement → review sequence. Avoid overlapping assignments and do not ask several models the same generic question. Synthesize results against the repository and executable checks: model agreement is not verification. Use one fresh-context peer only when a consequential judgment remains after direct checks; use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding counsel for one focused consequential uncertainty that evidence and the appropriate roles cannot settle; it has no approval or workflow authority, and a further consultation requires materially new or conflicting evidence." + artifactClause;
3127
+ const pipelineClause = opts.sweEnabled === false ? "" : `\n\n${PIPELINE_SKILLS_AWARENESS}`;
3128
+ return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Handle narrow, obvious, surgical, and single-command work directly; delegate a bounded workstream when it is broad, slow, context-heavy, or benefits from a genuinely independent perspective. " + MAX_PARALLELISM_RULE + " Brief each role with the desired outcome, relevant context, constraints, expected evidence, and verification. Use the roster as complementary capabilities, never as a required Explore → Plan → implement → review sequence. Avoid overlapping assignments and do not ask several models the same generic question. Synthesize results against the repository and executable checks: model agreement is not verification. Use one fresh-context peer only when a consequential judgment remains after direct checks; use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding counsel for one focused consequential uncertainty that evidence and the appropriate roles cannot settle; it has no approval or workflow authority, and a further consultation requires materially new or conflicting evidence." + artifactClause + pipelineClause;
2565
3129
  }
2566
3130
  if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest" || opts.profile === "balanced") {
2567
3131
  const isCheap = opts.profile === "cheap" || opts.profile === "cheap1m";
@@ -2582,7 +3146,9 @@ ${profileLabel} profile runtime wiring is unavailable. Work directly, use only t
2582
3146
  const workerBrowseClause = opts.browseAvailable ? ` \`worker-browse\` runs delegated autonomous browsing tasks through \`mcp__${workersKey}__browse\`.` : "";
2583
3147
  const artifactClause = opts.artifactAvailable ? ` \`mcp__${peersKey}__artifact_*\` provides human review in the artifact panel with auto-open on plan completion.` : "";
2584
3148
  const astraClause = opts.astraAvailable && (opts.profile === "fast" || opts.profile === "cheap1m") ? ` \`mcp__${peersKey}__astra\` (GPT-6 Astra ${astraDescriptor}) is the terminal escalation consultant for the lead only, reserved strictly for the hardest dead ends when direct evidence, Advisor, and Oracle have all failed to produce a defensible path (consulted at most 1-2 times per decision with concise context and specific questions).` : "";
2585
- return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + (isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly. ` + buildNativeReachClauses(opts) + ". Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` may be used for discovery spanning more than a couple of files; `Plan` for sequencing with complex interfaces or acceptance criteria; `General-Purpose` for mixed multi-step execution; `reviewer` for behavior-changing or risk-sensitive changes. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n" : `${profileLabel} launch profile. The lead coordinates execution across specialized native roles: ` + buildNativeReachClauses(opts) + ". In plan mode or when designing changes with complex sequencing, interface contracts, or acceptance criteria, delegate architectural planning to `Plan` (in plan mode, produce the plan and acceptance criteria; do not edit files). Phase pipeline (budget gather, expensive plan, budget execution, expensive verification). GATHER: delegate to `Explore` instead of exploring yourself. For any question spanning more than a couple of files, launch one or more `Explore` subagents in parallel with scoped evidence questions and let them return file:line citations; read directly only the small subset you must touch to decide or edit. `Plan` follows the same rule when it needs repository facts. PLAN: `Plan` produces handoff-ready steps (ordered, file:line, done conditions, acceptance criteria) for a `General-Purpose` executor that cannot see its reasoning; `Plan` may consult Oracle on unresolved trade-offs and reports any remaining gap to the lead. EXECUTE: delegate mixed multi-step work and Plan handoffs to `General-Purpose` in a fresh context to preserve lead context; brief with outcome, constraints, files in scope, and verification. VERIFY: after behavior-changing, cross-boundary, or risk-sensitive implementation, run relevant build/tests then invoke `reviewer` before declaring done. Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` is cheap and may be launched in parallel across independent discovery questions. Send independent subagent calls in parallel within a single turn. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n") + `Consultation guidance: Follow an evidence-first escalation ladder. Direct empirical evidence (search, code, tests, builds) settles factual questions first. Advisor is an optional, non-binding, lead-only transcript-aware sounding board for trajectory guidance or framing checks (direction, not dictation), and never use it for routine progress, waiting, directly verifiable facts, or completion ritual. \`mcp__${peersKey}__oracle\` is ${oracleDescriptor}, an expert consultant available to the lead and \`Plan\`, preferred over advisor for difficult conceptual, algorithmic, spec/protocol, or architectural tradeoffs evaluated in a self-contained brief; \`reviewer\` and other subagents cannot call Oracle. \`Plan\` may consult Oracle on unresolved trade-offs, and reports any remaining tie-breaking gap to the lead.${astraClause} ` + searchGuidance + `\`mcp__${searchKey}__web\` provides citable sources.${browserClause}${workerBrowseClause}${artifactClause}\n\nVerify claims with concrete repository evidence and tests before declaring work done. User instructions outrank delegation triggers. Stop named teammates when finished.`;
3149
+ const pipeline = isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly. ` + buildNativeReachClauses(opts) + ". Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` may be used for discovery spanning more than a couple of files; `Plan` for sequencing with complex interfaces or acceptance criteria; `General-Purpose` for mixed multi-step execution; `reviewer` for behavior-changing or risk-sensitive changes. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n" : `${profileLabel} launch profile. The lead coordinates execution across specialized native roles: ` + buildNativeReachClauses(opts) + ". In plan mode or when designing changes with complex sequencing, interface contracts, or acceptance criteria, delegate architectural planning to `Plan` (in plan mode, produce the plan and acceptance criteria; do not edit files). Phase pipeline (budget gather, expensive plan, budget execution, expensive verification). GATHER: delegate to `Explore` instead of exploring yourself. For any question spanning more than a couple of files, launch one or more `Explore` subagents in parallel with scoped evidence questions and let them return file:line citations; read directly only the small subset you must touch to decide or edit. `Plan` follows the same rule when it needs repository facts. PLAN: `Plan` produces handoff-ready steps (ordered, file:line, done conditions, acceptance criteria) for a `General-Purpose` executor that cannot see its reasoning; `Plan` may consult Oracle on unresolved trade-offs and reports any remaining gap to the lead. EXECUTE: delegate mixed multi-step work and Plan handoffs to `General-Purpose` in a fresh context to preserve lead context; brief with outcome, constraints, files in scope, and verification. VERIFY: after behavior-changing, cross-boundary, or risk-sensitive implementation, run relevant build/tests then invoke `reviewer` before declaring done. Handle trivial, surgical, single-file, or single-command tasks directly; you do not need to justify skipping delegation. `Explore` is cheap and may be launched in parallel across independent discovery questions. Send independent subagent calls in parallel within a single turn. Delegation graph: the lead may invoke all four; `Plan` may invoke `Explore` and `reviewer`; `General-Purpose` may invoke `reviewer`; `Explore`, `reviewer`, and `worker-browse` cannot invoke native subagents.\n\n";
3150
+ const swePipelineClause = opts.sweEnabled === false ? "" : `${PIPELINE_SKILLS_AWARENESS}\n\n`;
3151
+ return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + pipeline + `Consultation guidance: Follow an evidence-first escalation ladder. Direct empirical evidence (search, code, tests, builds) settles factual questions first. Advisor is an optional, non-binding, lead-only transcript-aware sounding board for trajectory guidance or framing checks (direction, not dictation), and never use it for routine progress, waiting, directly verifiable facts, or completion ritual. \`mcp__${peersKey}__oracle\` is ${oracleDescriptor}, an expert consultant available to the lead and \`Plan\`, preferred over advisor for difficult conceptual, algorithmic, spec/protocol, or architectural tradeoffs evaluated in a self-contained brief; \`reviewer\` and other subagents cannot call Oracle. \`Plan\` may consult Oracle on unresolved trade-offs, and reports any remaining tie-breaking gap to the lead.${astraClause} ` + searchGuidance + `\`mcp__${searchKey}__web\` provides citable sources.${browserClause}${workerBrowseClause}${artifactClause}\n\n` + swePipelineClause + "Verify claims with concrete repository evidence and tests before declaring work done. User instructions outrank delegation triggers. Stop named teammates when finished.";
2586
3152
  }
2587
3153
  return "## Operating defaults (these layer with the user's explicit direction and the domain's own standards as addons, not replacements: follow all three together; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nOrchestrate. Delegate research, implementation, review, and large reads to the right subagent, worker, or model. Reach for " + buildNativeReachClauses(opts) + "; worker-* agents for background non-blocking runs; Task subagents for parallel work; peer critics for review. That keeps your own context free to reason and collaborate with the user. Launch independent agents concurrently in a single message rather than serially. Delegation pays when the work is WIDE (many files or sources to sweep) or SLOW, and you need only the conclusion: the main thread is where you think with and respond to the user, and its context window is a finite shared resource. It does NOT pay merely because a sub-question is separable. A narrow, deep question whose whole value is file:line fidelity loses exactly that through a summarization layer, and a sub-question you could answer in one command is cheaper done directly than paying a subagent's startup. Do trivial, surgical, and last-mile work yourself. A named teammate persists after it reports so you can send it follow-ups, and nothing reaps it for you: its idle notice means available, not finished. Stop it once you are done with it.\n\nAdversarial review. The peer critics (`codex_critic` and `codex_reviewer`, `gemini_critic` and `gemini_reviewer`, `opus_critic`, and the `peer-review-coordinator` that fans out to several of them) are fresh-context models, so what they add is a blind spot that whoever produced the work cannot reach by thinking harder about it; prefer a critic from a different lab than the producer, since blind spots correlate within a lab. The `advisor` is a complement and not a substitute: it sees your transcript, so it catches your own drift and momentum, but it inherits your framing, which is exactly what a fresh-context critic does not. They earn their keep on consequential design choices, recommendations, and hard-to-reverse decisions: the cases where plausible alternatives remain and the conclusion rests on judgment rather than on something you can verify directly. That is where confabulation hides, so budget the wait even under delivery pressure. Always consult one when the change touches auth, user input, database queries, crypto, or serialization. They do NOT pay for read-only tracing, ordinary repository lookup, or a conclusion that a focused test, a direct reproduction, or unambiguous code evidence already settles. Asking a critic to re-derive a proven fact returns a confident answer either way, which is ritual skepticism rather than review, and skipping them there is the right call and not a shortcut. Match the lens to the artifact: a strategic critic for plans and trade-offs, a code reviewer for a concrete diff, the coordinator only when the risk warrants several independent lenses. Give whichever you pick the artifact and the constraints and not your rationale, since justification anchors the review and dulls it.\n\nAim high. Default to radical simplicity and a relentless focus on the user's real experience: design for the person and the job to be done, not the demo. Reason about the whole system from first principles, anticipating scale and the long arc rather than patching the surface. Work backwards from the outcome the user actually needs. Question every assumption and prefer what you can derive, reproduce, or test.\n\nEngineering excellence. When making technical decisions, give little weight to development cost; prefer quality, simplicity, robustness, scalability, and long-term maintainability. Fix a bug by first reproducing it end to end, as close to how a real user hits it as you can, so you solve the real problem and not a symptom. When testing a product end to end, be picky about the UI and obsessed with pixel perfection: if something clearly looks off, even when it is unrelated to your task, get it fixed along the way. Hold that same bar for the codebase itself: a lint error, a failing test, or a flaky test is worth fixing the moment you see it, whoever introduced it. Fold it into your current work rather than letting it derail the task the user actually asked for.";
2588
3154
  }
@@ -2593,7 +3159,8 @@ ${profileLabel} profile runtime wiring is unavailable. Work directly, use only t
2593
3159
  const OPERATING_DEFAULTS_DIRECTIVE = buildOperatingDefaultsDirective();
2594
3160
  const STANDARD_OPERATING_DEFAULTS_DIGEST = "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nDelegate when the work is WIDE (many files or sources to sweep) or SLOW and you need only the conclusion, to protect the main thread's finite context and keep it free for reasoning and interacting with the user; prefer parallel delegation for independent work. Do NOT delegate merely because a sub-question is separable: a narrow, deep question whose value is file:line fidelity loses exactly that through a summarization layer, and one answerable in a single command is cheaper done directly. Do trivial, surgical, and last-mile work yourself. Stop a named teammate once you are done with it: it persists for follow-ups, its idle notice means available rather than finished, and nothing reaps it for you.\n\nVerify, do not assert. Run the code, read the file, check the exit code. A claim in prose is worth nothing against state you did not check, and a check that cannot fail proves nothing. Reproduce a bug end to end, the way a real user hits it, before fixing it. Fix a lint error, failing test, or flake the moment you see it, whoever introduced it, without letting it derail the task at hand. Prefer quality and long-term maintainability over development cost.\n\nVerification has a blind spot: a consequential recommendation, a design or trade-off call, or a hard-to-reverse decision that still turns on judgment among plausible alternatives once the direct evidence is in. That is where confabulation hides, so put it past a peer critic before you ship it and budget the wait even under delivery pressure. When a test, a run, a reproduction, or a search would settle the claim, settle it that way instead; a critic asked to re-derive what you can already prove is ritual, not review.\n\nThe agent roster, the tool surface, and the reasoning behind these defaults are in your CLAUDE.md project instructions. Read them when choosing HOW to work; the rules above apply without a lookup.";
2595
3161
  function buildOperatingDefaultsDigest(opts = {}) {
2596
- if (opts.profile === "max") return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Do narrow, obvious, surgical, and single-command work directly; delegate bounded work that is broad, slow, context-heavy, or independently valuable. " + MAX_PARALLELISM_RULE + " Give delegated workstreams non-overlapping scopes and state the outcome, constraints, evidence, and verification expected.\n\nSynthesize and verify: run the code, inspect outputs, and check tests. Agent count and agreement are not evidence. Use a fresh-context peer only for consequential judgment that remains after direct checks, and use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding, primary-lead-only counsel for one focused consequential uncertainty; it is not an approval or completion gate.";
3162
+ const digestPipelineSentence = opts.sweEnabled === false ? "" : "Pipeline skills (all 200K default): `/gh-gather-context` before planning when context is needed, `/gh-plan` before implementation (waits for user approval), `/gh-implement` after approval with bounded parallel workers and staged review. Skip for trivial work.\n\n";
3163
+ if (opts.profile === "max") return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\nMax launch profile. The lead owns the outcome. Start with direct repository or runtime evidence. Do narrow, obvious, surgical, and single-command work directly; delegate bounded work that is broad, slow, context-heavy, or independently valuable. " + MAX_PARALLELISM_RULE + " Give delegated workstreams non-overlapping scopes and state the outcome, constraints, evidence, and verification expected.\n\n" + digestPipelineSentence + "Synthesize and verify: run the code, inspect outputs, and check tests. Agent count and agreement are not evidence. Use a fresh-context peer only for consequential judgment that remains after direct checks, and use the coordinator only when several distinct lenses could change the decision. Advisor is optional, non-binding, primary-lead-only counsel for one focused consequential uncertainty; it is not an approval or completion gate.";
2597
3164
  if (opts.profile === "fast" || opts.profile === "cheap" || opts.profile === "cheap1m" || opts.profile === "cheapest" || opts.profile === "balanced") {
2598
3165
  const isCheap = opts.profile === "cheap" || opts.profile === "cheap1m";
2599
3166
  const isCheapest = opts.profile === "cheapest";
@@ -2602,7 +3169,7 @@ function buildOperatingDefaultsDigest(opts = {}) {
2602
3169
  const oracleDescriptor = isCheapest ? "(GPT-5.6 Sol 200K/high, lead and Plan)" : isCheap || isBalanced ? "(Grok 4.6 200K/medium, lead and Plan)" : "(Opus 5 1M/high, lead and Plan)";
2603
3170
  const advisorDescriptor = isCheapest ? "(Gemini/high, lead-only)" : "(Sol/high, lead-only)";
2604
3171
  const astraStep = opts.astraAvailable && (opts.profile === "fast" || opts.profile === "cheap1m") ? `; (4) \`astra\` (GPT-6 Astra 200K/${isCheap ? "medium" : "high"}, lead-only) only as a last resort when direct evidence, Advisor, and Oracle cannot produce a defensible path (at most 1-2 calls per decision).` : ".";
2605
- return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + (isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly: use \`Explore\` for discovery spanning more than a couple of files, \`Plan\` in plan mode or for complex sequencing, \`General-Purpose\` for mixed multi-step execution, and \`reviewer\` for behavior-changing or risk-sensitive changes. Handle trivial and surgical edits directly. Stop named teammates when finished.\n\n` : `${profileLabel} launch profile. The lead coordinates execution across specialized roles: delegate broad discovery to \`Explore\` in parallel and do not sweep the repo yourself (read directly only files you will act on); delegate to \`Plan\` in plan mode or when structuring complex multi-step sequencing (\`Plan\` is an advisory planning capability, not an approval gate, and writes handoff-ready steps for \`General-Purpose\`); delegate mixed multi-step execution and Plan handoffs to \`General-Purpose\` in a fresh context; delegate to \`reviewer\` after behavior-changing or risk-sensitive implementation to verify correctness before declaring done; handle trivial and surgical edits directly. Send independent subagent calls in parallel within a single turn. Stop named teammates when finished.\n\n`) + "Verify claims against real evidence: run relevant commands and tests. Follow a disciplined consultation ladder for unresolved decisions: (1) direct code inspection, search, builds, and tests settle factual questions; (2) `advisor` " + advisorDescriptor + " for transcript-aware framing checks or trajectory guidance; (3) `oracle` " + oracleDescriptor + " for self-contained technical/architectural trade-offs" + astraStep;
3172
+ return "## Operating defaults (these layer with the user's direction and the domain's standards as addons: follow all three; on a direct conflict the user's direction wins, then the domain standard, then the default below)\n\n" + (isCheapest ? `${profileLabel} launch profile. The lead owns the outcome and handles straightforward work directly: use \`Explore\` for discovery spanning more than a couple of files, \`Plan\` in plan mode or for complex sequencing, \`General-Purpose\` for mixed multi-step execution, and \`reviewer\` for behavior-changing or risk-sensitive changes. Handle trivial and surgical edits directly. Stop named teammates when finished.\n\n` : `${profileLabel} launch profile. The lead coordinates execution across specialized roles: delegate broad discovery to \`Explore\` in parallel and do not sweep the repo yourself (read directly only files you will act on); delegate to \`Plan\` in plan mode or when structuring complex multi-step sequencing (\`Plan\` is an advisory planning capability, not an approval gate, and writes handoff-ready steps for \`General-Purpose\`); delegate mixed multi-step execution and Plan handoffs to \`General-Purpose\` in a fresh context; delegate to \`reviewer\` after behavior-changing or risk-sensitive implementation to verify correctness before declaring done; handle trivial and surgical edits directly. Send independent subagent calls in parallel within a single turn. Stop named teammates when finished.\n\n`) + digestPipelineSentence + "Verify claims against real evidence: run relevant commands and tests. Follow a disciplined consultation ladder for unresolved decisions: (1) direct code inspection, search, builds, and tests settle factual questions; (2) `advisor` " + advisorDescriptor + " for transcript-aware framing checks or trajectory guidance; (3) `oracle` " + oracleDescriptor + " for self-contained technical/architectural trade-offs" + astraStep;
2606
3173
  }
2607
3174
  return STANDARD_OPERATING_DEFAULTS_DIGEST;
2608
3175
  }
@@ -3068,12 +3635,28 @@ async function writeInjectedSkill(name, md) {
3068
3635
  * Injected-skill registry: the floor-raising / controller skills the `claude`
3069
3636
  * launcher materializes into the per-launch `CLAUDE_CONFIG_DIR` mirror so the
3070
3637
  * spawned Claude Code session discovers them (`/gh-research`,
3071
- * `/gh-orchestrate`, `/gh-floor-keeper`, `/gh-first-mate`). See
3072
- * `docs/floor-raising-agent-surface.md`.
3638
+ * `/gh-orchestrate`, `/gh-floor-keeper`, `/gh-first-mate`, plus the `--swe`
3639
+ * pipeline skills `/gh-gather-context`, `/gh-plan`, `/gh-implement`,
3640
+ * `/gh-swe-pipeline`). See `docs/floor-raising-agent-surface.md`.
3641
+ */
3642
+ /**
3643
+ * Pipeline skills for `--swe` launches on pinned profiles (all 200K default
3644
+ * context). The orchestrator (`/gh-swe-pipeline`) runs the other three in
3645
+ * strict sequence; the three stages stay individually invokable for users who
3646
+ * only want one stage.
3073
3647
  */
3648
+ const PIPELINE_SKILLS = [
3649
+ GATHER_CONTEXT_SKILL,
3650
+ PLAN_SKILL,
3651
+ IMPLEMENT_SKILL,
3652
+ SWE_PIPELINE_SKILL
3653
+ ];
3074
3654
  /** All injected skills, in dependency order (research underpins the others). */
3075
3655
  const INJECTED_SKILLS = [
3076
3656
  RESEARCH_SKILL,
3657
+ GATHER_CONTEXT_SKILL,
3658
+ PLAN_SKILL,
3659
+ IMPLEMENT_SKILL,
3077
3660
  ORCHESTRATE_SKILL,
3078
3661
  FLOOR_KEEPER_SKILL,
3079
3662
  WORKER_SKILL,
@@ -3083,8 +3666,18 @@ const INJECTED_SKILLS = [
3083
3666
  FIRST_MATE_CONDUCT_SKILL
3084
3667
  ];
3085
3668
  function injectedSkillsForLaunch(selection) {
3086
- if (selection.profileId === "fast" || selection.profileId === "cheap" || selection.profileId === "cheap1m" || selection.profileId === "cheapest" || selection.profileId === "balanced") return [];
3087
- if (selection.profileId === "max") return selection.firstMateEnabled ? INJECTED_SKILLS.filter((skill) => skill.name.startsWith("gh-first-mate")) : [];
3669
+ if (isPipelineSkillProfile(selection.profileId)) {
3670
+ if (!selection.sweEnabled) {
3671
+ if (selection.profileId === "max" && selection.firstMateEnabled) return INJECTED_SKILLS.filter((skill) => skill.name.startsWith("gh-first-mate"));
3672
+ return [];
3673
+ }
3674
+ if (selection.profileId === "max") {
3675
+ const pipeline = PIPELINE_SKILLS.slice();
3676
+ if (selection.firstMateEnabled) return [...pipeline, ...INJECTED_SKILLS.filter((skill) => skill.name.startsWith("gh-first-mate"))];
3677
+ return pipeline;
3678
+ }
3679
+ return PIPELINE_SKILLS.slice();
3680
+ }
3088
3681
  if (!selection.workerSkillsActive) return [];
3089
3682
  return INJECTED_SKILLS.filter((skill) => selection.firstMateEnabled || !skill.name.startsWith("gh-first-mate"));
3090
3683
  }
@@ -3155,4 +3748,4 @@ async function injectAttributionSuppressionIntoSettingsFile(settingsPath) {
3155
3748
  //#endregion
3156
3749
  export { writePeerMcpRuntimeFiles as C, workersKeyOf as S, sanitizeServeSettingsEnv as _, appendPeerAwarenessToMirroredClaudeMd as a, resolveCodexCliBackend as b, buildOperatingDefaultsDirective as c, prependStyleDirectiveToMirroredClaudeMd as d, buildArtifactReviewSkill as f, planModeAllowRules as g, injectAllowRules as h, writeInjectedSkill as i, prependArtifactPanelDirectiveToMirroredClaudeMd as l, configureServeDefaultPermissionMode as m, INJECTED_SKILLS as n, appendToolbeltAwarenessToMirroredClaudeMd as o, SEAMLESS_BUILTIN_TOOLS as p, injectedSkillsForLaunch as r, buildOperatingDefaultsDigest as s, injectAttributionSuppressionIntoSettingsFile as t, prependOperatingDefaultsToMirroredClaudeMd as u, BUILTIN_SUBAGENT_DEFINITIONS as v, resolveGroupKeysFromMirror as x, injectPeerMcpIntoMirror as y };
3157
3750
 
3158
- //# sourceMappingURL=attribution-settings-CX-kIrT8.js.map
3751
+ //# sourceMappingURL=attribution-settings-IbcZgAEf.js.map