github-router 0.3.206 → 0.3.211

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  "manifest_version": 3,
3
3
  "name": "github-router browser bridge",
4
4
  "short_name": "gh-router-browser",
5
- "version": "0.3.206",
5
+ "version": "0.3.211",
6
6
  "description": "Bridge between Claude (via github-router /mcp) and the browser. Implements tab control, navigation, clicks, form fill, downloads, screenshots, devtools eval. Blocks navigation to chrome://settings.",
7
7
  "key": "MIIBIjANBgkqhkiG9w0BAQEFAAOCAQ8AMIIBCgKCAQEAqJElxuBlonBS3TVW9FJN0mGTtShB3L1hoaYf6k39SOr1ogGYmF90EjRxy1i21k9wQQjPf26bcBu/9X67KrQjQV0uB38CaNukgiSeoLjfptN811u+PJHx6BP+jx3Qa6/3VenNPxHC8WEU0GXql8QSjIHEyCwKb6fMASXOK94JyB5Ywov2x8mt/+9ncqBBBMVzf6r5Sagy4PL1XnryLsuADD/vOEkPet8wXgH/Oj7v5tTsQQZ7U1JT51PoDs2BFnXc5v3TkVgZwd32k3ONh+nkDw1Hof+4zwUGOyJE6eMrlYzRlKM4Qxdf9JpavQvqfieAbTRWcyKeclnHeoIfE7cDBQIDAQAB",
8
8
  "background": {
@@ -1,4 +1,4 @@
1
- import { B as runWorkerAgent, F as EXPLORE_DEFAULT_MODEL, I as IMPLEMENT_DEFAULT_MODEL, L as PLAN_DEFAULT_MODEL, N as BROWSE_DEFAULT_MODEL, P as DEFAULT_MODEL, R as REVIEW_DEFAULT_MODEL, V as withNoOutputRetry, z as appendPlanReminder } from "./peer-mcp-personas-C3ii7ZqP.js";
1
+ import { B as runWorkerAgent, F as EXPLORE_DEFAULT_MODEL, I as IMPLEMENT_DEFAULT_MODEL, L as PLAN_DEFAULT_MODEL, N as BROWSE_DEFAULT_MODEL, P as DEFAULT_MODEL, R as REVIEW_DEFAULT_MODEL, V as withNoOutputRetry, z as appendPlanReminder } from "./peer-mcp-personas-CxpFD-rW.js";
2
2
  import "./paths-BO22pMUb.js";
3
3
  import "./lifecycle-D-1CYr1Y.js";
4
4
  import "./lifecycle-bPdiXjYB.js";
package/dist/main.js CHANGED
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { $ as buildAdvisorStream, $t as setupCopilotToken, A as trustRepo, At as readResponseBodyCapped, Bt as buildWorkspaceHeaderJson, C as fileLastPromptStore, Ct as getTokenCount, D as repoRoot, Dt as createResponses, E as repoFingerprint, Et as pickEndpoint, Ft as extractTarGzMember, G as toolbeltEnabled, Gt as DEFAULT_CODEX_MODEL_FALLBACKS, H as buildEnv, Ht as toolbeltPathOverride, It as extractZipMember, J as TOOLBELT_TOOLS, Jt as UPSTREAM_INACTIVITY_TIMEOUT_MS, K as toolbeltSkipSet, Kt as DEFAULT_PORT, Lt as shouldUseInsecureTls, M as resolveSealedGate, Mt as provisionBrowserAssets, Nt as hasSupportedBrowserInstalled, O as stopGateEnabledForRepo, Ot as createChatCompletions, Pt as provisionAndIndexColbert, Q as ADVISOR_TOOL_INSTRUCTIONS, Qt as withInstallLock, Rt as ArtifactClient, S as fileFindingsStore, St as createMessages, T as isSubagentContext, Tt as resolveMcpToolTimeoutMs, U as availableToolCommands, Ut as DEFAULT_CLAUDE_MODEL_FALLBACKS, Vt as collapsePathKeys, W as buildToolbeltAwareness, Wt as DEFAULT_CODEX_MODEL, X as searchWeb, Xt as pickClaudeDefault, Y as assetFor, Yt as generateRandomPort, Z as ADVISOR_INTERNAL_TOOL_NAME, Zt as getPackageVersion, _ as stopGateDisabled, _n as copilotBaseUrl, _t as nativeSubagentModel, a as buildPeerAwarenessSnippet, an as cacheVSCodeVersion, at as logStreamError, b as stopReviewEnabled, bn as state, bt as shimDefaultsToXhigh, c as personasFor, cn as resolveCodexModel, ct as handleMcpDelete, d as buildStopHookCommand, dn as getModels, dt as artifactToolsEnabled, en as setupGitHubAgentToken, et as injectAdvisorTool, f as captureLaunchBaseline, fn as getGitHubUser, ft as browseAgentEnabled, g as launchBaselineKey, gn as GITHUB_API_BASE_URL, gt as geminiAvailable, h as injectStopHookIntoSettingsFile, hn as forwardError, ht as fleetToolsEnabled, i as buildAgentPrompt, in as cacheModels, it as isControllerClosedError, j as liveExec, jt as parseJsonOrDiagnose, k as stopReviewStateDir, kt as MAX_RESPONSE_BODY_BYTES, l as buildArtifactOpenHookCommand, ln as resolveModel, lt as handleMcpPost, m as fileBlockBudget, mn as HTTPError, mt as browserToolsEnabled, n as MCP_GROUPS, nn as tryRefreshAndRetry, nt as buildAnthropicErrorEvent, o as buildPeerAwarenessSummary, on as filterBetaHeader, ot as readIteratorWithTimeout, p as decideStopHook, pn as fetchWithTransientRetry, pt as browserCompoundToolsEnabled, q as vscodeRipgrepPath, qt as UPSTREAM_FETCH_TIMEOUT_MS, r as assertMcpToolSurfaceConsistent, rn as cacheCopilotVersion, rt as buildOpenAIErrorEvent, s as enumerateInjectedMcpToolNames, sn as isNullish, st as relayAnthropicStream, t as GROUP_META, tn as setupGitHubToken, tt as isAdvisorRequested, u as buildSessionBindHookCommand, un as sleep$1, ut as agentToolsEnabled, v as stopGateId, vn as copilotHeaders, vt as standInToolEnabled, w as fileReviewDebounce, wt as assembleResponsesPayload, x as fileBaselineStore, xt as countTokens, y as stopGatePlanMode, yn as githubHeaders, yt as workerToolsEnabled, zt as buildWorkspaceHeaderHelperCommand } from "./peer-mcp-personas-C3ii7ZqP.js";
2
+ import { $ as buildAdvisorStream, $t as getPackageVersion, A as trustRepo, At as readResponseBodyCapped, Bt as ArtifactClient, C as fileLastPromptStore, Ct as getTokenCount, D as repoRoot, Dt as createResponses, E as repoFingerprint, Et as pickEndpoint, Ft as extractTarGzMember, G as toolbeltEnabled, Gt as DEFAULT_CLAUDE_MODEL_FALLBACKS, H as buildEnv, Ht as buildWorkspaceHeaderJson, It as extractZipMember, J as TOOLBELT_TOOLS, Jt as DEFAULT_PORT, K as toolbeltSkipSet, Kt as DEFAULT_CODEX_MODEL, Lt as CONDENSED_OPERATING_SEQUENCE, M as resolveSealedGate, Mt as provisionBrowserAssets, Nt as hasSupportedBrowserInstalled, O as stopGateEnabledForRepo, Ot as createChatCompletions, Pt as provisionAndIndexColbert, Q as ADVISOR_TOOL_INSTRUCTIONS, Qt as pickClaudeDefault, Rt as DEFINITION_OF_GREATNESS, S as fileFindingsStore, Sn as state, St as createMessages, T as isSubagentContext, Tt as resolveMcpToolTimeoutMs, U as availableToolCommands, Ut as collapsePathKeys, Vt as buildWorkspaceHeaderHelperCommand, W as buildToolbeltAwareness, Wt as toolbeltPathOverride, X as searchWeb, Xt as UPSTREAM_INACTIVITY_TIMEOUT_MS, Y as assetFor, Yt as UPSTREAM_FETCH_TIMEOUT_MS, Z as ADVISOR_INTERNAL_TOOL_NAME, Zt as generateRandomPort, _ as stopGateDisabled, _n as forwardError, _t as nativeSubagentModel, a as buildPeerAwarenessSnippet, an as cacheCopilotVersion, at as logStreamError, b as stopReviewEnabled, bn as copilotHeaders, bt as shimDefaultsToXhigh, c as personasFor, cn as filterBetaHeader, ct as handleMcpDelete, d as buildStopHookCommand, dn as resolveModel, dt as artifactToolsEnabled, en as withInstallLock, et as injectAdvisorTool, f as captureLaunchBaseline, fn as sleep$1, ft as browseAgentEnabled, g as launchBaselineKey, gn as HTTPError, gt as geminiAvailable, h as injectStopHookIntoSettingsFile, hn as fetchWithTransientRetry, ht as fleetToolsEnabled, i as buildAgentPrompt, in as tryRefreshAndRetry, it as isControllerClosedError, j as liveExec, jt as parseJsonOrDiagnose, k as stopReviewStateDir, kt as MAX_RESPONSE_BODY_BYTES, l as buildArtifactOpenHookCommand, ln as isNullish, lt as handleMcpPost, m as fileBlockBudget, mn as getGitHubUser, mt as browserToolsEnabled, n as MCP_GROUPS, nn as setupGitHubAgentToken, nt as buildAnthropicErrorEvent, o as buildPeerAwarenessSummary, on as cacheModels, ot as readIteratorWithTimeout, p as decideStopHook, pn as getModels, pt as browserCompoundToolsEnabled, q as vscodeRipgrepPath, qt as DEFAULT_CODEX_MODEL_FALLBACKS, r as assertMcpToolSurfaceConsistent, rn as setupGitHubToken, rt as buildOpenAIErrorEvent, s as enumerateInjectedMcpToolNames, sn as cacheVSCodeVersion, st as relayAnthropicStream, t as GROUP_META, tn as setupCopilotToken, tt as isAdvisorRequested, u as buildSessionBindHookCommand, un as resolveCodexModel, ut as agentToolsEnabled, v as stopGateId, vn as GITHUB_API_BASE_URL, vt as standInToolEnabled, w as fileReviewDebounce, wt as assembleResponsesPayload, x as fileBaselineStore, xn as githubHeaders, xt as countTokens, y as stopGatePlanMode, yn as copilotBaseUrl, yt as workerToolsEnabled, zt as shouldUseInsecureTls } from "./peer-mcp-personas-CxpFD-rW.js";
3
3
  import { a as isUnderClaudeConfigMirror, d as writeRuntimeFileSecure, i as ensurePaths, o as removeOwnClaudeConfigMirror, r as ensureClaudeConfigMirror, t as PATHS, u as writeArtifactCredsToMirror } from "./paths-BO22pMUb.js";
4
4
  import { c as killManagedTree, d as runCommandCapture, f as runCommandVoid, l as parseBoolEnv, s as killChildProcessTree, u as resolveExecutable } from "./lifecycle-D-1CYr1Y.js";
5
5
  import { a as sweepRegistry } from "./lifecycle-bPdiXjYB.js";
@@ -2894,7 +2894,7 @@ async function discoverGateCommands(cwd, opts) {
2894
2894
  if (files.length === 0) return null;
2895
2895
  let result;
2896
2896
  try {
2897
- const { runWorkerAgent } = await import("./engine-DZjN7BuD.js");
2897
+ const { runWorkerAgent } = await import("./engine-DFuchRoh.js");
2898
2898
  result = await runWorkerAgent({
2899
2899
  mode: "explore",
2900
2900
  workspace: root,
@@ -3149,11 +3149,166 @@ function buildFirstMateGuardHookCommand(execPath, entry) {
3149
3149
  return `${JSON.stringify(execPath)} ${JSON.stringify(entry)} internal-first-mate-guard`;
3150
3150
  }
3151
3151
 
3152
+ //#endregion
3153
+ //#region src/lib/injected-skills/first-mate-conduct-skill.ts
3154
+ const FIRST_MATE_CONDUCT_SKILL = {
3155
+ name: "gh-first-mate-conduct",
3156
+ md: `---
3157
+ name: gh-first-mate-conduct
3158
+ description: Fleet conductor for first-mate — one durable heartbeat loop drives a FLEET of per-repo CEO meta-subagents to greatness. Arms the deterministic loop, sweeps the whole portfolio once, fans out a fresh CEO subagent per repo that needs judgment, batches their verdicts, couriers human decisions, and re-arms. Use when first-mate should be the default durable driver for one or many repos.
3159
+ user-invocable: true
3160
+ ---
3161
+
3162
+ # gh-first-mate-conduct: the deterministic-loop fleet conductor
3163
+
3164
+ You are the **fleet conductor**. One correctly-armed heartbeat loop, run by you (the main session), drives a fleet of GitHub repos — each by its own per-repo CEO meta-subagent — to greatness. This is the ONLY way one instance drives many repos durably: only the main REPL can arm a durable cron, so you own the heartbeat; the per-repo CEOs own judgment; the durable ledger + strategy store own memory. You hold almost nothing in context.
3165
+
3166
+ You carry the whole brain by REFERENCE, not by memorizing missions: the CEO/CTO/CPO operating protocol (\`/gh-first-mate-operate\`), the per-CEO driving loop (\`/gh-first-mate\`), and the definition of repo greatness (below). Your context each wake is only the compact board + each CEO's compact return — never diffs, logs, or transcripts.
3167
+
3168
+ ## The loop each wake (arm it right, then fan out)
3169
+
3170
+ 1. **One global sweep.** Call \`mcp__first-mate__advance\` ONCE with no \`mission_id\` — it sweeps the whole portfolio and returns \`board\`, \`needsModel[]\`, \`needsHuman[]\`, \`nextWakeSeconds\`. Tier1 auto-answers have ALREADY fired inside the tool for the safe \`author_fix\`/\`answer_agent_question\`/\`decompose\` envelope, so the cheap mechanical loop is handled for free and most wakes surface little. Do NOT disable tier1 — it is what keeps you from spawning a CEO for trivia.
3171
+ 2. **Partition by mission/repo.** Group the residual \`needsModel\` (escalated \`review_plan\`/\`judge_review\` + anything tier1 declined) and open \`needsHuman\` and active board rows by \`missionId\`.
3172
+ 3. **Surgical per-repo CEO fan-out.** For each mission that has an open judgment \`needsModel\`, OR a strategic checkpoint due (a phase to advance, a greatness item to verify, a pre-registered kill/pivot threshold reached), spawn a **fresh** CEO meta-subagent — Agent tool, in PARALLEL (multiple Agent calls in one message), capped at a few per wake (fleet fan-out cap; the MCP inflight budget is shared). A mission whose agents are grinding with no \`needsModel\` and no checkpoint due needs ZERO CEO spawns this wake. Hand each CEO its brief (below).
3173
+ 4. **Batch + apply + courier.** Collect each CEO's returned \`model_answers\`; call \`mcp__first-mate__advance\` ONCE more with all of them (\`model_answers: [...]\`) to apply — the CONDUCTOR owns the single drive lease, CEOs never call \`advance\` themselves (that would contend the lease). Courier every \`needsHuman\` packet to the user (open \`packetHtmlPath\` in the artifact panel); never decide merges/abandons yourself.
3174
+ 5. **Re-arm ONE heartbeat** from the MINIMUM \`nextWakeSeconds\` across the portfolio (the tightest cadence wins, so an imminent-work repo tightens the whole fleet).
3175
+
3176
+ ## The CEO spawn brief (hand this to each fresh per-repo CEO)
3177
+
3178
+ > You are the CEO of repo <owner/name> (mission <id>), spawned fresh for one turn with COMPLETE authority over that repo on the GitHub platform. Do NOT arm a heartbeat (you are a subagent — you cannot) and do NOT call \`mcp__first-mate__advance\` (the conductor applies your verdicts). Steps: (1) \`mcp__first-mate__read_strategy({mission_id})\` + \`mcp__first-mate__mission_status({mission_id})\` to re-hydrate strategy + state; (2) follow \`/gh-first-mate\` (per-CEO driving) + \`/gh-first-mate-operate\` (CEO protocol) + the greatness bar — you DELEGATE all buildable work (code, docs, README/website content, UI, CI, tests) to cloud-agent units and do little yourself; your hands are for orchestration, verification, and decisions, not building; and you VERIFY every user-viewable surface (product UI, README-as-rendered, Pages, docs, release, og-card) by VIEWING the rendered pixels (\`mcp__browser__*\` / screenshots), never guessing from code; (3) for each of YOUR \`needsModel\` requests, VERIFY the deliverable against external evidence (delegate heavy reads to worker-explore/worker-review to stay context-thin) and produce the typed verdict (decompose with disjoint \`fileScopes\`, review_plan, judge_review, author_fix, answer_agent_question); (4) \`mcp__first-mate__write_strategy({mission_id, currentPhase, activeBet, greatnessChecklist, decisionLog:[one entry], nextStrategicAction})\` to persist your strategy delta; (5) RETURN a compact \`{ model_answers:[{requestId,verdict}], needsHuman:[…to courier], strategy_written:true }\` — no prose, no diffs.
3179
+
3180
+ Because each CEO is FRESH per wake, its strategic continuity comes ONLY from the strategy store — so a rich \`write_strategy\` (phase, pre-registered bet + thresholds, greatness checklist with evidence handles, an append-only decision-log entry, what-was-tried) is what stops the next wake's CEO from drifting or re-litigating a dead end.
3181
+
3182
+ ## Self-driving heartbeat (arm / disarm — you own the ONLY one)
3183
+
3184
+ ONE durable cron, marker \`[fm-heartbeat]\`, is the dead-man's-switch that survives idle, compaction, restart, and /clear. There is exactly ONE first-mate heartbeat regardless of which driver skill armed it; manage it create-fresh-then-reap-the-rest:
3185
+
3186
+ Arm (nextWakeSeconds is a number):
3187
+ 1. Cadence bucket from nextWakeSeconds (fixed cron, no time math): \`<=120 → "1-59/2 * * * *"\`; \`<=600 → "2,7,12,17,22,27,32,37,42,47,52,57 * * * *"\`; else \`"3,13,23,33,43,53 * * * *"\`.
3188
+ 2. CronCreate the new job (durable:true, recurring:true, the chosen cron, prompt \`"/gh-first-mate-conduct [fm-heartbeat] wake the fleet, sweep, fan out CEOs, apply verdicts, reschedule."\`) and capture its id.
3189
+ 3. CronList, then CronDelete every job whose prompt contains \`[fm-heartbeat]\` EXCEPT the id you just created — converges to exactly one, reaps duplicates/old-version orphans (including a stray standalone \`/gh-first-mate\` heartbeat), resets the 7-day expiry.
3190
+
3191
+ Disarm (nextWakeSeconds is null AND no pending needsHuman): CronList and CronDelete every \`[fm-heartbeat]\` job; report the fleet is idle and resumes when a mission is next started/advanced.
3192
+
3193
+ MCP unavailable / not \`--agents\`: do not advance; reap all \`[fm-heartbeat]\` jobs and report "re-run under \`github-router claude --agents\`." No scheduler tool: report the next wake is in nextWakeSeconds and stop.
3194
+
3195
+ ## Context discipline & report
3196
+
3197
+ The ledger is durable memory for unit state; the strategy store is durable memory for CEO strategy; your context is neither. Never read a full diff/log/transcript. Report compactly from the board: per mission — id, repos, phase counts, blocked count, the greatness-checklist progress (leading done + which LAGGING signals moved), needsHuman awaiting the user, and the next wake. A repo is only "great" when a LAGGING signal has moved, never when leading boxes are merely ticked.
3198
+
3199
+ ## Definition of greatness (the bar every repo is driven toward)
3200
+
3201
+ ${DEFINITION_OF_GREATNESS}
3202
+ `
3203
+ };
3204
+
3205
+ //#endregion
3206
+ //#region src/lib/injected-skills/first-mate-operate-skill.ts
3207
+ const FIRST_MATE_OPERATE_SKILL = {
3208
+ name: "gh-first-mate-operate",
3209
+ md: `---
3210
+ name: gh-first-mate-operate
3211
+ description: Operator-facing CEO/CTO/CPO operating protocol for autonomously driving a product with first-mate — shape each mission from a real struggling moment, make acceptance criteria externally verifiable, sequence discovery through growth, and escalate launch, spend, and pricing to the human. Use when deciding WHAT product work first-mate should drive, not only how to execute it.
3212
+ user-invocable: true
3213
+ ---
3214
+
3215
+ # gh-first-mate-operate: drive a product as CEO + CTO + CPO
3216
+
3217
+ You are the CEO of the product. The GitHub cloud coding agents are your team — they carry the CTO/CPO/engineering execution roles (scaffolded into each repo). Your job is to think like a CEO and get real, verified work out of that team: decide the product direction (the niche, the riskiest assumption, the MVP scope, when to launch, what to measure, what to iterate), turn each decision into a scoped mission, drive the agents to deliver it, and hold the result to evidence.
3218
+
3219
+ You do not write the product code. You orchestrate: shape missions, review plans, answer the team's questions fast so they never idle, verify deliverables, and sequence the whole effort toward an outcome. Seed the team's playbook once with \`mcp__first-mate__scaffold_repo\` (it commits \`docs/playbook/README.md\` plus the \`ceo\`/\`cto\`/\`cpo\` role agents the cloud agents read); then use THIS protocol to run the company.
3220
+
3221
+ ## Delegate the work — your hands are for orchestration, not building
3222
+
3223
+ You are the orchestrator; the GitHub cloud coding agents are the workers. The whole point of first-mate is to get work OUT of the team — so DELEGATE, and do little yourself.
3224
+
3225
+ - **Everything BUILDABLE is a cloud-agent mission/unit, never hand-written by you:** product code, bug fixes, README/docs/website CONTENT, CI/workflows, tests, the UI. Dispatch it via \`mcp__first-mate__start_mission\` / \`add_units\` / \`decompose\` — do NOT open an editor and write it yourself.
3226
+ - **Your OWN hands are only for:** strategy & decisions, decomposition, answering the team (plan review, questions, fix instructions, judge verdicts), VERIFYING deliverables (browse / read / screenshot — **observe, never build**), persisting strategy, and GitHub-**platform** governance via \`gh\` (merge, release, branch protection, secrets, labels, issues, triage). That is orchestration, not building.
3227
+ - **Resist the "I'll just quickly fix it myself" urge — it is the failure mode.** A quick hand-edit feels faster but doesn't scale, produces no reusable team capability, and isn't your job. If you catch yourself about to write code, docs, or UI: STOP and dispatch it as a scoped unit instead. The only exception is a genuine one-line last-mile config the cloud-agent loop cannot reach — and even then, prefer a unit.
3228
+
3229
+ ## Drive the team (get work out of them)
3230
+
3231
+ - **Verify, never trust "done".** Every deliverable clears an external checkpoint — a real HTTP 200, green CI, an observed analytics event, a real survey N — or it is not done. Reject self-reported completion and send it back with a concrete gap.
3232
+ - **Keep the team unblocked and busy.** A blocked agent produces nothing: answer \`answer_agent_question\` promptly, dispatch independent units in parallel, and re-steer a stalled or underdelivering agent instead of waiting. Idle or looping agents are wasted throughput.
3233
+ - **Set the bar as acceptance criteria.** The mission's acceptance criteria = the phase's externally verifiable checkpoint. Vague criteria produce vague work; make the bar reproducible.
3234
+ - **Own the P&L of attention.** Kill low-value missions, double down on what moves the outcome metric, and escalate only the genuinely human-gated calls (launch to real channels, spend, pricing, merges).
3235
+
3236
+ ## The one rule that makes autonomy safe
3237
+
3238
+ Every phase advances only on an EXTERNALLY VERIFIABLE checkpoint — a real HTTP 200, a green CI run, an observed analytics event, or a real survey sample size — never a self-reported "done". Autonomous agents fail or hallucinate "done" a large fraction of the time, so an unverified claim is not progress. Encode the checkpoint as the mission's acceptance criteria and refuse to advance without the evidence.
3239
+
3240
+ ## Operating loop (OODA inside Build-Measure-Learn)
3241
+
3242
+ - Inner loop, each turn: OBSERVE fresh evidence (issues, mentions, downloads, analytics), ORIENT against the current job/segment/assumptions, DECIDE one reversible next action against a pre-set threshold, ACT by delegating a scoped mission to the cloud agents.
3243
+ - Outer loop, each phase: build the smallest testable increment, measure externally observable behavior, learn against the pre-registered threshold, then persist or pivot. Do not enter the next phase until its checkpoint is independently reproducible.
3244
+
3245
+ ## Shaping a mission by phase
3246
+
3247
+ When you call \`mcp__first-mate__start_mission\` (or \`mcp__first-mate__add_units\`), set the fields from the CURRENT phase:
3248
+
3249
+ - **goal**: the phase objective, grounded in a real struggling moment — not a feature wish.
3250
+ - **acceptance_criteria**: the phase's externally verifiable exit checkpoint, stated as evidence a reviewer can reproduce (e.g. "cold-start quickstart under five minutes, timed from a fresh checkout, recorded in the PR"; "Sean Ellis survey with N≥40 responses and ≥40% 'very disappointed'").
3251
+ - **house_rules**: any hard constraint (privacy, license, brand, spend limit).
3252
+ - Unlock real parallelism the right way: give each independent unit a DISJOINT \`fileScopes\` allowlist so their builds run CONCURRENTLY (the controller proves non-overlap and dispatches in parallel up to the mission's \`maxConcurrentBuilds\`), and/or split fully independent workstreams into SEPARATE missions (the build gate is per-mission, so N missions build N units at once). \`max_in_flight_per_provider\` is a global provider cap, NOT the build-concurrency lever — disjoint \`fileScopes\` and separate missions are. Never race overlapping work on the same files.
3253
+
3254
+ Let the controller drive decomposition and steering (see \`gh-first-mate\`); this skill decides the PHASE and the checkpoint, not the controller mechanics.
3255
+
3256
+ ## Phased sequence (shared with the scaffolded playbook)
3257
+
3258
+ ${CONDENSED_OPERATING_SEQUENCE}
3259
+
3260
+ ## Definition of greatness (the shipping bar every repo must clear)
3261
+
3262
+ Driving a product to greatness is not just shipping features — the repo itself must clear a verifiable shipping-infrastructure bar. Drive each repo toward it and gate each item on real, third-party-checkable evidence (a green check, a \`gh api\`, a \`curl\`, a \`cosign verify\`), never a self-report. This is the SAME bar the scaffolded playbook and the eval use.
3263
+
3264
+ ${DEFINITION_OF_GREATNESS}
3265
+
3266
+ ## Iterate to polish — never guess the UI, drive it and view the pixels
3267
+
3268
+ UI/UX quality (and every user-viewable surface) is DRIVEN, SEEN, and ITERATED, never designed once and asserted. For ANY unit that touches the product UI, the README, the website, docs, the release page, or the og-card, require the cloud agent to — and verify yourself by — VIEW the rendered result before "done":
3269
+ 1. Build & run the real artifact (the deployed Pages / live URL, or a dev server) — a local happy-path screenshot is not the product.
3270
+ 2. Drive every state & flow with \`mcp__browser__*\` (navigate/act/observe): first-run/empty, real input, forced error, success, edge/overflow — not just the ideal state.
3271
+ 3. Screenshot the matrix: each state at mobile/tablet/desktop × light+dark, honoring \`prefers-color-scheme\` and \`prefers-reduced-motion\`. These pixels, not the code, are the evidence.
3272
+ 4. Critique the pixels against the professional bar (Pillar D of the greatness definition); for each defect name the concrete problem AND the screenshot it is in. Rank by severity.
3273
+ 5. Fix the top defects (dispatch to the implementer role; prefer token/design-system fixes over one-off patches).
3274
+ 6. Re-drive & re-capture (repeat 2–5) until the vision rubric is clean and the deterministic gates (visual-regression / axe / contrast / CWV) are green. Update \`toHaveScreenshot\` baselines only on an intentional, reviewed change.
3275
+ 7. Lock it in: commit the visual-regression baselines + axe/Lighthouse/CWV CI so polish can't silently regress.
3276
+
3277
+ The no-guessing rule: never mark a user-viewable surface — including the README as GitHub renders it — "done" from code/markdown review alone; back every claim with a screenshot of the actual running/rendered artifact.
3278
+
3279
+ ## Every iteration must move the end-user experience forward — and regress nothing
3280
+
3281
+ This is the objective the whole loop is tuned to, and the operating expression of the CEO eval (\`docs/first-mate-ceo-eval-framework.md\`, the End-User Experience Delta lens): after each iteration (or a bounded set), measure what VERIFIABLY improved for the END USER, and confirm nothing regressed.
3282
+
3283
+ - **Drive the pinned user journeys before AND after.** Keep a pinned set of golden journeys per repo (the aha path + key flows) plus user-facing facets — journey completion, time-to-first-value, capability, performance, accessibility, user-facing correctness, the rendered README/Pages/docs/Release, state robustness. Capture them on the inherited state, then again after the iteration, by DRIVING the real artifact — never inferred from code or a 200.
3284
+ - **Ship only a net improvement.** An iteration is done only when it shows at least one MATERIAL, verified user-facing improvement (past a pre-set threshold, tied to a real user job) AND zero detected unintended regression under the coverage you actually ran. Report coverage honestly: "no regression detected in the pinned journeys," never an absolute "zero."
3285
+ - **A detected regression is stop-the-line.** A journey that worked now failing, a metric crossing its budget the wrong way, a removed/broken capability, a new user-hitting bug, a broken rendered surface, or a slower quickstart CAPS the iteration: fix the regression before the improvement counts. Recovering a regression you caused only retires the debt; it is not new progress.
3286
+ - **Intended user-facing changes are approved tradeoffs, not free.** A deliberate deprecation / redesign / pivot is recorded as an approved tradeoff (old baseline preserved, migration path, human approval) — never silently relabeled "not a regression."
3287
+ - **Necessary invisible work is legitimate but bounded.** Security hardening, migrations, and refactors that de-risk future UX are enabling-investment, not churn — but bounded: after ~2 consecutive iterations with no user-facing delta (or ~a third of a milestone's budget), the next iteration must show a material user improvement or you flag the strand.
3288
+ - **The agent must not own the ruler.** The journey manifest + measurement harness are yours (or human-blessed), not authored by the cloud agent being measured; a change to them is reviewed and logged, like a merge approval.
3289
+
3290
+ ## Anti-patterns (hard stops)
3291
+
3292
+ - **Over-building without distribution:** run a reachability/channel test before extending product scope. "Build it and they will come" is not a plan.
3293
+ - **Hallucinated progress:** require real evidence (HTTP 200, green CI, observed analytics, real survey N); never convert activity or a narrative into completion.
3294
+ - **Viral ≠ product-market fit:** attention, stars, and shares do not replace the Sean Ellis threshold plus a flattening retention curve.
3295
+ - **Metrics after the fact:** pre-register kill/pivot/continue thresholds before collecting results.
3296
+
3297
+ ## Escalate to the human (never decide autonomously)
3298
+
3299
+ Hard authority limits: launching to real external channels, any spend or paid acquisition, setting or changing pricing, issuing discounts, entering contracts, expanding privileges, and any regulated/legal/privacy commitment require an explicit human boundary or approval. Within those limits, proceed on best judgment and record assumptions rather than pausing. Merge approval and abandonment remain human-gated per \`gh-first-mate\`.
3300
+
3301
+ ## Report
3302
+
3303
+ Report the current phase, its checkpoint and whether it is met with reproducible evidence, the active mission(s) and their phase-appropriate acceptance criteria, and the next decision or escalation.
3304
+ `
3305
+ };
3306
+
3152
3307
  //#endregion
3153
3308
  //#region src/lib/injected-skills/first-mate-setup-skill.ts
3154
3309
  const FIRST_MATE_SETUP_SKILL = {
3155
3310
  name: "gh-first-mate-scaffold",
3156
- md: "---\nname: gh-first-mate-scaffold\ndescription: Scaffolds a repo-geared agentic-dev foundation through first-mate: seeds guidance files, role agents, ADRs, changelog, learnings, PR template, test instructions, Copilot setup, and CI through a scaffold branch and PR. Use before the first build wave on an owned repository.\nuser-invocable: true\n---\n\n# gh-first-mate-scaffold\n\nInvoke the `scaffold_repo` MCP tool (`mcp__first-mate__scaffold_repo`) before the first build wave on an owned repository. The goal is not generic TODO stubs; it is a repo-geared foundation that GitHub agents, local agents, reviewers, and CI can read.\n\n## What it seeds\n\n- `AGENTS.md` / `CLAUDE.md` / `GEMINI.md` / `.github/copilot-instructions.md` — identical guidance with overview, detected stack, commands, hard DoD gate, primary OS, conventions, structure, decisions/memory, handoff, testing, and gotchas.\n- `.github/agents/{planner,implementer,reviewer,researcher,tester}.md` mirrored into `.claude/agents/` — role agents with frontmatter, cold-start contract, method, quality bar, output contract, and self-reminder.\n- `docs/adrs/0000-template.md` plus `docs/adr/0001-record-architecture-decisions.md` — Nygard-style decision record foundation.\n- `LEARNINGS.md`, `CHANGELOG.md`, `docs/history/0000-template.md`, `docs/plans/README.md`, and `docs/research/README.md` — durable memory, history, plans, and research conventions.\n- `.github/pull_request_template.md` — summary, type, failure-modes-considered-and-tested, and DoD checklist.\n- `.github/instructions/tests.instructions.md` — path-scoped test guidance filled from detected framework/dir/glob where possible.\n- `.github/workflows/copilot-setup-steps.yml` and starter `.github/workflows/ci.yml` — detected toolchain setup with stable quality-gate job names.\n\nIt does not seed factory-protocol or `docs/factory/` files. Orchestration remains outside the product repo in first-mate.\n\n## Usage\n\n```\nmcp__first-mate__scaffold_repo({ repo: \"owner/repo\" })\nmcp__first-mate__scaffold_repo({ repo: \"owner/repo\", mode: \"enhance\" })\nmcp__first-mate__scaffold_repo({\n repo: \"owner/repo\",\n mode: \"add-missing-only\",\n detection_overrides: { primary_os: \"windows-latest\", test_command: \"npm test\" }\n})\n```\n\nModes:\n\n- `add-missing-only` (default): seed absent files and skip present files.\n- `enhance`: for guidance files, ADR index, changelog, and learnings, append only missing `##` sections; never rewrite existing prose. Other present files are skipped.\n- `overwrite-approved`: replace existing files only when explicitly approved.\n\nAlways inspect the returned per-file report and PR. A no-op result means the repo already has the foundation or has no missing enhanceable sections.\n"
3311
+ md: "---\nname: gh-first-mate-scaffold\ndescription: Scaffolds a repo-geared agentic-dev foundation through first-mate: seeds guidance files, role agents, ADRs, changelog, learnings, PR template, test instructions, Copilot setup, and CI through a scaffold branch and PR. Use before the first build wave on an owned repository.\nuser-invocable: true\n---\n\n# gh-first-mate-scaffold\n\nInvoke the `scaffold_repo` MCP tool (`mcp__first-mate__scaffold_repo`) before the first build wave on an owned repository. The goal is not generic TODO stubs; it is a repo-geared foundation that GitHub agents, local agents, reviewers, and CI can read.\n\n## What it seeds\n\n- `AGENTS.md` / `CLAUDE.md` / `GEMINI.md` / `.github/copilot-instructions.md` — identical guidance with overview, detected stack, commands, hard DoD gate, primary OS, conventions, structure, decisions/memory, handoff, testing, and gotchas.\n- `.github/agents/{ceo,cto,cpo,planner,implementer,reviewer,researcher,tester}.md` mirrored into `.claude/agents/` — C-suite operator hats plus execution roles, each with frontmatter, cold-start contract, method, quality bar, output contract, and self-reminder.\n- `docs/playbook/README.md` — the autonomous DISCOVER → NICHE → POSITION → SCOPE → BUILD → LAUNCH → MEASURE → ITERATE → GROW protocol, externally verifiable phase gates, OODA / Build-Measure-Learn governance, and authority limits.\n- `docs/adrs/0000-template.md` plus `docs/adr/0001-record-architecture-decisions.md` — Nygard-style decision record foundation.\n- `LEARNINGS.md`, `CHANGELOG.md`, `docs/history/0000-template.md`, `docs/plans/README.md`, and `docs/research/README.md` — durable memory, history, plans, and research conventions.\n- `.github/pull_request_template.md` — summary, type, failure-modes-considered-and-tested, and DoD checklist.\n- `.github/instructions/tests.instructions.md` — path-scoped test guidance filled from detected framework/dir/glob where possible.\n- `.github/workflows/copilot-setup-steps.yml` and starter `.github/workflows/ci.yml` — detected toolchain setup with stable quality-gate job names.\n\nIt does not seed factory-protocol or `docs/factory/` files. Orchestration remains outside the product repo in first-mate.\n\n## Usage\n\n```\nmcp__first-mate__scaffold_repo({ repo: \"owner/repo\" })\nmcp__first-mate__scaffold_repo({ repo: \"owner/repo\", mode: \"enhance\" })\nmcp__first-mate__scaffold_repo({\n repo: \"owner/repo\",\n mode: \"add-missing-only\",\n detection_overrides: { primary_os: \"windows-latest\", test_command: \"npm test\" }\n})\n```\n\nModes:\n\n- `add-missing-only` (default): seed absent files and skip present files.\n- `enhance`: for guidance files, product playbook, ADR index, changelog, and learnings, append only missing `##` sections; never rewrite existing prose. Other present files are skipped.\n- `overwrite-approved`: replace existing files only when explicitly approved.\n\nAlways inspect the returned per-file report and PR. A no-op result means the repo already has the foundation or has no missing enhanceable sections.\n"
3157
3312
  };
3158
3313
 
3159
3314
  //#endregion
@@ -3172,6 +3327,17 @@ Use this skill when the user wants first-mate to drive GitHub cloud coding agent
3172
3327
  The first-mate controller is the durable system of record: missions, units, decisions, handles, and controller state live in its registry and ledger.
3173
3328
  Your job is to run the thin protocol, not to hold the mission in context.
3174
3329
 
3330
+ ## You are the CEO
3331
+
3332
+ You are the CEO of the product. The GitHub cloud coding agents are your team; your job is to get real, verified work out of them and drive the product to an outcome — not to write the code yourself. Operate like a CEO every turn:
3333
+
3334
+ - **Drive results, not activity.** Hold every deliverable to external evidence (a real HTTP 200, green CI, an observed metric, a real survey N). Never accept a self-reported "done" — autonomous agents fail or hallucinate completion a large fraction of the time.
3335
+ - **Set clear expectations.** Every mission's acceptance criteria IS the bar: a phase's externally verifiable checkpoint, stated so a reviewer can reproduce it.
3336
+ - **Keep the team unblocked and busy.** Answer agent questions fast (the controller surfaces \`answer_agent_question\`), dispatch independent units in parallel, and re-steer a stalled or weak agent promptly rather than letting it idle.
3337
+ - **Own the outcome.** Sequence missions toward the product result (niche → MVP → launch → traction), kill low-value work, and iterate on evidence. Think in bets — hypothesis, metric, threshold — and delegate execution to the cloud-agent team.
3338
+
3339
+ For the full operating protocol — discovery, positioning, MVP scope, launch, measure, iterate, grow — invoke \`/gh-first-mate-operate\`.
3340
+
3175
3341
  ## Foundation-first mandate
3176
3342
 
3177
3343
  Before the first build wave on an owned repository, run \`mcp__first-mate__scaffold_repo\` and verify the PR landed or is already present. The scaffold must seed a repo-geared foundation that GitHub agents and CI can read: guidance, role agents, ADRs, changelog, learnings, PR template, test instructions, Copilot setup, and CI. Do not seed factory-protocol files into product repos; first-mate is the external orchestrator.
@@ -3180,7 +3346,7 @@ Use \`mode: "add-missing-only"\` for new repos, \`mode: "enhance"\` when a repo
3180
3346
 
3181
3347
  ## Scoped-work discipline
3182
3348
 
3183
- Well-scoped, testable work items succeed; vague meta-work fails. Discovery/decompose must emit concrete units with acceptance criteria, expected evidence, and dependencies. Keep one active build unit per concern. Parallelism is for read-only producers (research, review, planning) and independent units only, not for racing broad implementation waves.
3349
+ Well-scoped, testable work items succeed; vague meta-work fails. Discovery/decompose must emit concrete units with acceptance criteria, expected evidence, and dependencies. **Parallelize deliberately.** Independent build units that declare DISJOINT \`fileScopes\` build CONCURRENTLY the controller proves independence (no unmet deps + non-overlapping declared scopes) and dispatches them in parallel up to the mission's \`maxConcurrentBuilds\` cap; units with overlapping or undeclared scope serialize their builds (the safe default). For fully independent workstreams you can also run SEPARATE missions — the build gate is per-mission, so N missions build N units at once. Never race overlapping work on the same files.
3184
3350
 
3185
3351
  Judgment and merge policy: merge remains human-gated, evidence-gated, and head/base-bound. Use the best available model tier for plan review, judgment, and merge decisions; never cheap out on plan/judge/merge calls.
3186
3352
 
@@ -3221,7 +3387,7 @@ Keep verdicts small and typed to the request kind.
3221
3387
 
3222
3388
  Use the request's kind and payload as the contract:
3223
3389
 
3224
- - decompose: split a unit-less active mission into dispatchable units. Return { units: [{ title, repo?, agent?, dependsOn?, model? }] }. \`dependsOn\` entries are 0-based indices into the same units list. Emit once per unit-less active mission; the controller creates durable unit ids and will not ask again after units exist.
3390
+ - decompose: split a unit-less active mission into dispatchable units. Return { units: [{ title, repo?, agent?, dependsOn?, model?, fileScopes? }] }. \`dependsOn\` entries are 0-based indices into the same units list. **\`fileScopes\` is the parallelism lever** — declare each unit's disjoint file allowlist (paths or \`dir/**\` prefixes it may touch) so the controller can prove independence and build those units CONCURRENTLY; units with overlapping or absent scopes serialize their builds. Emit once per unit-less active mission; the controller creates durable unit ids and will not ask again after units exist.
3225
3391
  - review_plan: review the plan against the mission goal, acceptance criteria, and house rules. Return { decision: "approve" } when the plan is good enough to implement, or { decision: "refine", instruction: "..." } with a short actionable refinement.
3226
3392
  - answer_agent_question: answer only from the acceptance criteria and supplied context. Return { answer: "..." }. If the answer is not derivable, do not invent policy; escalate by leaving a short answer that says what the human must decide.
3227
3393
  - author_fix: author a concise fix instruction for the cloud agent. Return { instruction: "..." } with the failure, expected behavior, and any bounded check to run.
@@ -4372,7 +4538,9 @@ const INJECTED_SKILLS = [
4372
4538
  FLOOR_KEEPER_SKILL,
4373
4539
  WORKER_SKILL,
4374
4540
  FIRST_MATE_SKILL,
4375
- FIRST_MATE_SETUP_SKILL
4541
+ FIRST_MATE_SETUP_SKILL,
4542
+ FIRST_MATE_OPERATE_SKILL,
4543
+ FIRST_MATE_CONDUCT_SKILL
4376
4544
  ];
4377
4545
 
4378
4546
  //#endregion
@@ -5054,7 +5222,7 @@ function initProxyFromEnv() {
5054
5222
  //#endregion
5055
5223
  //#region package.json
5056
5224
  var name = "github-router";
5057
- var version$1 = "0.3.206";
5225
+ var version$1 = "0.3.211";
5058
5226
 
5059
5227
  //#endregion
5060
5228
  //#region src/lib/approval.ts
@@ -8620,7 +8788,7 @@ function getCodexEnvVars(serverUrl) {
8620
8788
  //#endregion
8621
8789
  //#region src/claude.ts
8622
8790
  function isFirstMateSkillName(name$1) {
8623
- return name$1 === "gh-first-mate" || name$1 === "gh-first-mate-scaffold";
8791
+ return name$1 === "gh-first-mate" || name$1 === "gh-first-mate-scaffold" || name$1 === "gh-first-mate-operate" || name$1 === "gh-first-mate-conduct";
8624
8792
  }
8625
8793
  const claudeArgs = {
8626
8794
  ...sharedServerArgs,