@yagni-app/code-staging 0.1.0-staging.997.1 → 0.2.0-staging.1025.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +58 -9
  2. package/dist/claudeCompat.d.ts +36 -5
  3. package/dist/claudeCompat.js +85 -23
  4. package/dist/claudePlugins.d.ts +109 -0
  5. package/dist/claudePlugins.js +336 -0
  6. package/dist/cli.js +14 -4
  7. package/dist/crashReport.d.ts +135 -0
  8. package/dist/crashReport.js +291 -0
  9. package/dist/doctor.d.ts +21 -0
  10. package/dist/doctor.js +52 -0
  11. package/dist/extension/askAdvisorTool.js +7 -1
  12. package/dist/extension/bless.js +16 -3
  13. package/dist/extension/boostCommand.d.ts +144 -0
  14. package/dist/extension/boostCommand.js +263 -0
  15. package/dist/extension/branding.d.ts +31 -0
  16. package/dist/extension/branding.js +37 -0
  17. package/dist/extension/chipEditor.js +7 -3
  18. package/dist/extension/claudeRules.d.ts +54 -0
  19. package/dist/extension/claudeRules.js +180 -0
  20. package/dist/extension/config.d.ts +61 -0
  21. package/dist/extension/config.js +86 -0
  22. package/dist/extension/costHud.d.ts +128 -15
  23. package/dist/extension/costHud.js +189 -19
  24. package/dist/extension/crashReport.d.ts +89 -0
  25. package/dist/extension/crashReport.js +241 -0
  26. package/dist/extension/index.d.ts +43 -4
  27. package/dist/extension/index.js +241 -32
  28. package/dist/extension/initPass.d.ts +65 -47
  29. package/dist/extension/initPass.js +145 -145
  30. package/dist/extension/mcpTools.d.ts +57 -0
  31. package/dist/extension/mcpTools.js +132 -0
  32. package/dist/extension/pipeline/eval.d.ts +42 -5
  33. package/dist/extension/pipeline/eval.js +44 -0
  34. package/dist/extension/pipeline/goCommand.d.ts +18 -0
  35. package/dist/extension/pipeline/goCommand.js +139 -26
  36. package/dist/extension/pipeline/goCompareCommand.d.ts +18 -8
  37. package/dist/extension/pipeline/goCompareCommand.js +42 -23
  38. package/dist/extension/pipeline/orchestrator.js +9 -0
  39. package/dist/extension/pipeline/runCostTable.d.ts +37 -0
  40. package/dist/extension/pipeline/runCostTable.js +165 -0
  41. package/dist/extension/pipeline/runState.d.ts +19 -0
  42. package/dist/extension/pipeline/runState.js +11 -0
  43. package/dist/extension/pipeline/runner.d.ts +19 -0
  44. package/dist/extension/pipeline/runner.js +13 -1
  45. package/dist/extension/pipeline/scrubSecrets.js +2 -2
  46. package/dist/extension/pipeline/stages.d.ts +3 -1
  47. package/dist/extension/pipeline/stages.js +3 -1
  48. package/dist/extension/pipeline/types.d.ts +7 -4
  49. package/dist/extension/pipeline/verify.js +6 -1
  50. package/dist/extension/pipeline/worktree.js +3 -1
  51. package/dist/extension/provider.d.ts +7 -1
  52. package/dist/extension/provider.js +8 -1
  53. package/dist/extension/recall.js +5 -2
  54. package/dist/extension/rerouteNotice.d.ts +42 -0
  55. package/dist/extension/rerouteNotice.js +67 -0
  56. package/dist/extension/sessionRuns.d.ts +45 -0
  57. package/dist/extension/sessionRuns.js +77 -0
  58. package/dist/extension/subagents.d.ts +17 -7
  59. package/dist/extension/subagents.js +52 -7
  60. package/dist/launch.d.ts +17 -3
  61. package/dist/launch.js +22 -9
  62. package/dist/login.d.ts +7 -0
  63. package/dist/login.js +3 -1
  64. package/package.json +2 -2
@@ -45,15 +45,32 @@ export declare const JUDGMENT_STAGE_IDS: readonly ["plan", "review"];
45
45
  * moves exactly one variable.
46
46
  */
47
47
  export declare function withJudgmentTier(stages: PipelineStage[] | undefined, tier: ModelTier): PipelineStage[];
48
+ /**
49
+ * The map stage's model choice IS the map-tier decision. Kept as data,
50
+ * mirroring {@link JUDGMENT_STAGE_IDS}, so the transform below and any future
51
+ * lane agree on what "map tier" means. Currently a single stage, but an array
52
+ * so a future lane that widens "map" to more than one stage is a data change.
53
+ */
54
+ export declare const MAP_STAGE_IDS: readonly ["map"];
55
+ /**
56
+ * Set the map stage to a given tier, leaving every other stage untouched.
57
+ * PURE: returns new stage objects, mirroring {@link withJudgmentTier}'s shape.
58
+ *
59
+ * This is the seam for the Task 13 gate: does map on `standard` produce a
60
+ * meaningfully better repo brief than map on `efficient`, or is the cheaper
61
+ * tier good enough for a step whose job is compression, not judgment.
62
+ */
63
+ export declare function withMapTier(stages: PipelineStage[] | undefined, tier: ModelTier): PipelineStage[];
48
64
  /**
49
65
  * Which lane an outcome belongs to.
50
66
  *
51
- * Two comparisons ride this engine today: grounded-vs-blind (does grounding
52
- * beat guessing) and peak-vs-advanced judgment (does the stronger model on
53
- * plan+review produce better changes). Both are two-lane A/Bs scored the same
54
- * way, so the lane is DATA rather than a hardcoded union.
67
+ * Three comparisons ride this engine today: grounded-vs-blind (does grounding
68
+ * beat guessing), peak-vs-advanced judgment (does the stronger model on
69
+ * plan+review produce better changes), and standard-vs-efficient map (does
70
+ * the pricier tier on the map stage produce better changes). All are two-lane
71
+ * A/Bs scored the same way, so the lane is DATA rather than a hardcoded union.
55
72
  */
56
- export type Lane = "grounded" | "blind" | "peak_judgment" | "advanced_judgment";
73
+ export type Lane = "grounded" | "blind" | "peak_judgment" | "advanced_judgment" | "map_standard" | "map_efficient";
57
74
  /** A lane's id plus how it is named in the report. */
58
75
  export interface LaneSpec {
59
76
  id: Lane;
@@ -81,6 +98,23 @@ export declare const JUDGMENT_LANES: {
81
98
  readonly label: "Advanced judgment";
82
99
  };
83
100
  };
101
+ /**
102
+ * The map-tier question (Task 13): standard on map against efficient on map.
103
+ * `standard` is the treatment (the pricier, currently-shipped lane) so the
104
+ * verdict answers "does paying for standard over efficient on map pay off",
105
+ * matching how {@link JUDGMENT_LANES} always puts the more expensive lane in
106
+ * `treatment`.
107
+ */
108
+ export declare const MAP_TIER_LANES: {
109
+ readonly treatment: {
110
+ readonly id: "map_standard";
111
+ readonly label: "Map standard";
112
+ };
113
+ readonly baseline: {
114
+ readonly id: "map_efficient";
115
+ readonly label: "Map efficient";
116
+ };
117
+ };
84
118
  /**
85
119
  * A lane's business-fit score, produced by running the GROUNDED
86
120
  * review_business_match against that lane's final diff (so both lanes are judged
@@ -139,6 +173,7 @@ export interface ComparisonCopy {
139
173
  }
140
174
  export declare const GROUNDED_COPY: ComparisonCopy;
141
175
  export declare const JUDGMENT_COPY: ComparisonCopy;
176
+ export declare const MAP_TIER_COPY: ComparisonCopy;
142
177
  /**
143
178
  * Render the comparative report a human reads. Leads with the headline conflict
144
179
  * delta (the GATE number), then each lane's stop reason, finding tally, and fit
@@ -165,4 +200,6 @@ export declare function makeComparisonEval(seams: ComparisonSeams, lanes: {
165
200
  export declare function makeGroundedVsBlindEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
166
201
  /** The judgment-tier question: peak on plan+review against advanced (YAG-380). */
167
202
  export declare function makeJudgmentTierEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
203
+ /** The map-tier question: standard on map against efficient on map (Task 13). */
204
+ export declare function makeMapTierEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
168
205
  //# sourceMappingURL=eval.d.ts.map
@@ -61,6 +61,25 @@ export function withJudgmentTier(stages = selectStages("full"), tier) {
61
61
  const judgment = new Set(JUDGMENT_STAGE_IDS);
62
62
  return stages.map((s) => (judgment.has(s.id) ? { ...s, model: tier } : s));
63
63
  }
64
+ /**
65
+ * The map stage's model choice IS the map-tier decision. Kept as data,
66
+ * mirroring {@link JUDGMENT_STAGE_IDS}, so the transform below and any future
67
+ * lane agree on what "map tier" means. Currently a single stage, but an array
68
+ * so a future lane that widens "map" to more than one stage is a data change.
69
+ */
70
+ export const MAP_STAGE_IDS = ["map"];
71
+ /**
72
+ * Set the map stage to a given tier, leaving every other stage untouched.
73
+ * PURE: returns new stage objects, mirroring {@link withJudgmentTier}'s shape.
74
+ *
75
+ * This is the seam for the Task 13 gate: does map on `standard` produce a
76
+ * meaningfully better repo brief than map on `efficient`, or is the cheaper
77
+ * tier good enough for a step whose job is compression, not judgment.
78
+ */
79
+ export function withMapTier(stages = selectStages("full"), tier) {
80
+ const mapStages = new Set(MAP_STAGE_IDS);
81
+ return stages.map((s) => (mapStages.has(s.id) ? { ...s, model: tier } : s));
82
+ }
64
83
  /** The grounding GATE: the real /go against a grounding-suppressed twin. */
65
84
  export const GROUNDED_LANES = {
66
85
  treatment: { id: "grounded", label: "Grounded" },
@@ -71,6 +90,17 @@ export const JUDGMENT_LANES = {
71
90
  treatment: { id: "peak_judgment", label: "Peak judgment" },
72
91
  baseline: { id: "advanced_judgment", label: "Advanced judgment" },
73
92
  };
93
+ /**
94
+ * The map-tier question (Task 13): standard on map against efficient on map.
95
+ * `standard` is the treatment (the pricier, currently-shipped lane) so the
96
+ * verdict answers "does paying for standard over efficient on map pay off",
97
+ * matching how {@link JUDGMENT_LANES} always puts the more expensive lane in
98
+ * `treatment`.
99
+ */
100
+ export const MAP_TIER_LANES = {
101
+ treatment: { id: "map_standard", label: "Map standard" },
102
+ baseline: { id: "map_efficient", label: "Map efficient" },
103
+ };
74
104
  export const GROUNDED_COPY = {
75
105
  title: "Grounded vs blind eval",
76
106
  deltaLabel: "FIT delta (blind conflicts minus grounded conflicts)",
@@ -91,6 +121,16 @@ export const JUDGMENT_COPY = {
91
121
  : "Peak and advanced judgment tied on flagged conflicts for this ticket.",
92
122
  footer: "Report only. This comparison is never wired to model selection, tiers, or routing. Only plan and review differ between the lanes, so the delta isolates the judgment tier. One ticket is an anecdote; run several before concluding.",
93
123
  };
124
+ export const MAP_TIER_COPY = {
125
+ title: "Standard vs efficient map eval",
126
+ deltaLabel: "FIT delta (efficient-map conflicts minus standard-map conflicts)",
127
+ verdict: (favor) => favor > 0
128
+ ? `Standard on map REDUCED business-fit conflicts by ${favor} on this ticket.`
129
+ : favor < 0
130
+ ? `Standard on map showed ${Math.abs(favor)} MORE flagged conflict${Math.abs(favor) === 1 ? "" : "s"} on this ticket. The map tier upgrade did not pay for itself here.`
131
+ : "Standard and efficient map tied on flagged conflicts for this ticket.",
132
+ footer: "Report only. This comparison is never wired to model selection, tiers, or routing. Only the map stage differs between the lanes, so the delta isolates the map tier. One ticket is an anecdote; run several before concluding.",
133
+ };
94
134
  const blocking = (findings) => findings.filter((f) => f.severity === "critical" || f.severity === "high").length;
95
135
  /**
96
136
  * Render the comparative report a human reads. Leads with the headline conflict
@@ -179,4 +219,8 @@ export function makeGroundedVsBlindEval(seams) {
179
219
  export function makeJudgmentTierEval(seams) {
180
220
  return makeComparisonEval(seams, JUDGMENT_LANES, JUDGMENT_COPY);
181
221
  }
222
+ /** The map-tier question: standard on map against efficient on map (Task 13). */
223
+ export function makeMapTierEval(seams) {
224
+ return makeComparisonEval(seams, MAP_TIER_LANES, MAP_TIER_COPY);
225
+ }
182
226
  //# sourceMappingURL=eval.js.map
@@ -58,6 +58,8 @@
58
58
  * is fail-soft, so recording can never break `/go`.
59
59
  */
60
60
  import type { ExtensionAPI, ExtensionCommandContext } from "@earendil-works/pi-coding-agent";
61
+ import type { SpendResponse } from "../costHud.js";
62
+ import { type CrashReporter } from "../crashReport.js";
61
63
  import { runFinish as defaultRunFinish } from "./finish.js";
62
64
  import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
63
65
  import { makeRunSession as defaultMakeRunSession } from "./runSession.js";
@@ -97,12 +99,28 @@ export interface RegisterGoDeps {
97
99
  runFinish?: typeof defaultRunFinish;
98
100
  /** Injectable journal read for cross-process liveness (default: the file checkpoint store). */
99
101
  loadJournal?: (sessionKey: string) => CheckpointRecord[];
102
+ /**
103
+ * Task 8: fetch the server-authoritative per-stage spend for one /go run
104
+ * (`GET /api/yagni-code/spend?runId=`, wired in index.ts reusing Task 7's
105
+ * fetch machinery). Absent (no dep wired), throwing, or resolving null all
106
+ * fall back to the local `runCostNote` client estimate — see
107
+ * `resolveCostText` below. No default here: unlike every other network dep
108
+ * on this interface, index.ts is the ONLY real wiring (there is nothing
109
+ * sensible to default to without a baseUrl/token), and every test that
110
+ * cares injects its own fake.
111
+ */
112
+ fetchRunSpend?: (runId: string, signal?: AbortSignal) => Promise<SpendResponse | null>;
100
113
  /** Injectable fs existence check (worktree adoption on resume). */
101
114
  exists?: (path: string) => boolean;
102
115
  /** Clock seam for registry rows + staleness. */
103
116
  now?: () => number;
104
117
  baseUrl?: string;
105
118
  getToken?: () => string | undefined;
119
+ /**
120
+ * Crash reporter for the run's terminal catch (a THROWN error, not a
121
+ * verdict-shaped failure). Fail-soft and fire-and-forget by contract.
122
+ */
123
+ reportCrash?: CrashReporter;
106
124
  }
107
125
  /**
108
126
  * Per-run UI key prefix: the widget/status keys are `yagni-go:<runShortId>` so
@@ -65,8 +65,11 @@ import { eventToLine } from "./activity.js";
65
65
  import { ActivityFeed, SPINNER_FRAMES } from "./activityFeed.js";
66
66
  import { RunState } from "./runState.js";
67
67
  import { aggregateRunUsage } from "./budget.js";
68
+ import { formatRunCostTable } from "./runCostTable.js";
68
69
  import { makeCombinedCheckpointStore, makeFileCheckpointStore, makePiJournalCheckpointStore, } from "./checkpoint.js";
69
70
  import { getToken as defaultGetToken, resolveBaseUrl } from "../config.js";
71
+ import { makeCrashReporter } from "../crashReport.js";
72
+ import { scrubSecrets } from "./scrubSecrets.js";
70
73
  import { isDesktopSurface } from "../surface.js";
71
74
  import { runFinish as defaultRunFinish, verifyTrailerValue, } from "./finish.js";
72
75
  import { GO_USAGE, parseGoArgs } from "./goFlags.js";
@@ -75,6 +78,7 @@ import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
75
78
  import { planResume } from "./resume.js";
76
79
  import { activeRunCount, beginRun, classifyRunLiveness, findActiveRunByTicket, isRunInFlight, isTerminalStatus, lastJournalTs, loadRegistryRows, MAX_CONCURRENT_RUNS, settleRun, trackRunPromise, worktreesDir, } from "./runRegistry.js";
77
80
  import { makeRunSession as defaultMakeRunSession } from "./runSession.js";
81
+ import { recordSessionRun } from "../sessionRuns.js";
78
82
  import { resolveTicketBrief as defaultResolveTicketBrief } from "./ticketResolution.js";
79
83
  import { snapshotWorkspace as defaultSnapshotWorkspace } from "./workspace.js";
80
84
  import { parseChangedPaths } from "./verify.js";
@@ -168,9 +172,9 @@ function finishNote(result, finish) {
168
172
  return "";
169
173
  return `\n\n${parts.join("\n")}`;
170
174
  }
171
- /** Blocking-finding detail appended to the DURABLE handoff (spec: surface findings, not counts). */
172
- function formatHandoff(result, ticket) {
173
- const summary = formatSummary(result);
175
+ /** Blocking-finding detail appended to the DURABLE handoff (spec: surface findings, not counts). `costText` — see `resolveCostText` (Task 8). */
176
+ function formatHandoff(result, ticket, costText) {
177
+ const summary = formatSummaryForChat(result, costText);
174
178
  const degraded = `${degradedLensNote(result)}${verifyNote(result)}`;
175
179
  const blocking = result.findings.filter((f) => f.severity === "critical" || f.severity === "high");
176
180
  if (blocking.length === 0)
@@ -235,23 +239,41 @@ function outcomeFor(stopReason) {
235
239
  return "failed";
236
240
  return "completed"; // clean | round_cap are both real completions
237
241
  }
238
- /**
239
- * The bare "what was done" recap — stop reason + the review tally — with NO
240
- * "YAGNI Code /go finished:" prefix. This is what rides `finish` as the
241
- * `run_outcome` activity body. The Work-page timeline already labels that beat
242
- * "Run outcome", so prefixing here would double up into "Run outcome: YAGNI
243
- * Code /go finished: ...". `formatSummary` re-adds the prefix for the standalone
244
- * toast + scrollback line, where the context is useful on its own.
245
- */
246
- function recapBody(result) {
242
+ /** Stop reason + the review tally, with NO cost segment and no prefix — the shared core of `recapBody`/`recapBodyForChat` below. */
243
+ function tallyLine(result) {
247
244
  const rounds = result.rounds.length;
248
245
  const total = result.findings.length;
249
246
  const blocking = result.findings.filter((f) => f.severity === "critical" || f.severity === "high").length;
250
247
  const reason = STOP_REASON_COPY[result.stopReason];
251
248
  return (`${reason}. ` +
252
249
  `${rounds} review round${rounds === 1 ? "" : "s"}, ` +
253
- `${total} finding${total === 1 ? "" : "s"} (${blocking} blocking).` +
254
- runCostNote(result));
250
+ `${total} finding${total === 1 ? "" : "s"} (${blocking} blocking).`);
251
+ }
252
+ /**
253
+ * The bare "what was done" recap — stop reason + the review tally + the LOCAL
254
+ * cost estimate — with NO "YAGNI Code /go finished:" prefix. This is what
255
+ * rides `finish` as the `run_outcome` activity body (session.finish's
256
+ * `outcomeSummary`), which must stay synchronous: it is recorded as part of
257
+ * ending the run and must never wait on the Task 8 server spend fetch. The
258
+ * chat-facing summary (`recapBodyForChat`, below) is the one that prefers the
259
+ * server-priced table when it is available. The Work-page timeline already
260
+ * labels this beat "Run outcome", so prefixing here would double up into "Run
261
+ * outcome: YAGNI Code /go finished: ...". `formatSummaryForChat` re-adds the
262
+ * prefix for the standalone toast + scrollback line, where the context is
263
+ * useful on its own.
264
+ */
265
+ function recapBody(result) {
266
+ return `${tallyLine(result)}${runCostNote(result)}`;
267
+ }
268
+ /**
269
+ * Chat-facing recap: the same tally as `recapBody`, but the cost segment is
270
+ * `costText` — precomputed by `resolveCostText` (Task 8), which is either the
271
+ * multi-line server-priced table or the labeled local-estimate fallback.
272
+ * Kept distinct from `recapBody` because `costText` requires an async fetch
273
+ * that `recapBody`'s caller (session.finish's outcomeSummary) cannot wait on.
274
+ */
275
+ function recapBodyForChat(result, costText) {
276
+ return `${tallyLine(result)}${costText}`;
255
277
  }
256
278
  /**
257
279
  * What the run actually cost.
@@ -271,14 +293,62 @@ function runCostNote(result) {
271
293
  return ` Run cost: $${usage.cost.toFixed(2)} over ${tokens} tokens.`;
272
294
  }
273
295
  /**
274
- * One-line, human-readable outcome for the session + a UI toast. This is the
275
- * DURABLE record (sent via `sendUserMessage`, fired even on headless runs where
276
- * no `ActivityFeed` is constructed), so it is intentionally kept separate from
277
- * `ActivityFeed.finalLine` (the feed's compact, UI-only outcome glyph-line)
278
- * rather than the two sharing one outcome-copy source.
296
+ * Task 8: resolve the cost segment for the CHAT-FACING summary (the toast +
297
+ * the durable handoff) — the server-priced per-stage table
298
+ * (`formatRunCostTable`) when this run is tracked and the fetch succeeds,
299
+ * else today's local `runCostNote` estimate, explicitly labeled "(client
300
+ * estimate)" so it is never mistaken for the priced total. Mirrors
301
+ * `registerCostCommand`'s own fail-soft preference order in costHud.ts.
302
+ *
303
+ * Never throws and never blocks long: `fetchRunSpend` (index.ts) already
304
+ * carries its own request timeout (same 5s ceiling as /cost's fetches), and
305
+ * this wraps the ENTIRE body — not just the fetch — in try/catch. That is
306
+ * deliberate, not belt-and-braces: the caller (see `runToCompletion` below)
307
+ * starts this promise and leaves it un-awaited for a while so session.finish
308
+ * never waits on it, so a throw ANYWHERE in here between promise creation and
309
+ * that later `await` — including from `runCostNote`/`formatRunCostTable` on a
310
+ * malformed result, not just from the fetch itself — would otherwise surface
311
+ * as a guaranteed unhandled rejection during that gap, independent of
312
+ * whatever the caller does with the eventual value.
313
+ *
314
+ * An empty-rows response is treated the same as "unavailable" (mirrors
315
+ * registerCostCommand's `serverLooksStale` guard): a run that billed real
316
+ * spend but whose rows have not landed yet would otherwise render a
317
+ * misleading "$0.00 (server)" table instead of the honest local estimate.
279
318
  */
280
- function formatSummary(result) {
281
- return `YAGNI Code /go finished: ${recapBody(result)}`;
319
+ async function resolveCostText(result, runId, fetchRunSpend, signal) {
320
+ try {
321
+ if (runId && fetchRunSpend) {
322
+ try {
323
+ const spend = await fetchRunSpend(runId, signal);
324
+ if (spend && spend.rows.length > 0)
325
+ return formatRunCostTable(spend);
326
+ }
327
+ catch {
328
+ /* fail-soft: fall through to the local client estimate below */
329
+ }
330
+ }
331
+ const local = runCostNote(result);
332
+ return local ? `${local} (client estimate)` : "";
333
+ }
334
+ catch {
335
+ // Truly can't say anything honest (even the local estimate blew up):
336
+ // no cost text at all, matching runCostNote's own "nothing to report" ""
337
+ // convention rather than fabricating a number.
338
+ return "";
339
+ }
340
+ }
341
+ /**
342
+ * One-line-or-more, human-readable outcome for the session + a UI toast. This
343
+ * is the DURABLE record (sent via `sendUserMessage`, fired even on headless
344
+ * runs where no `ActivityFeed` is constructed), so it is intentionally kept
345
+ * separate from `ActivityFeed.finalLine` (the feed's compact, UI-only outcome
346
+ * glyph-line) rather than the two sharing one outcome-copy source. `costText`
347
+ * carries the Task 8 cost segment (`resolveCostText`'s result: the
348
+ * server-priced table, or its labeled local-estimate fallback).
349
+ */
350
+ function formatSummaryForChat(result, costText) {
351
+ return `YAGNI Code /go finished: ${recapBodyForChat(result, costText)}`;
282
352
  }
283
353
  /**
284
354
  * User-facing notice when a /go run is NOT being recorded on the Work page.
@@ -351,8 +421,17 @@ export function registerGoCommand(pi, deps = {}) {
351
421
  const loadJournal = deps.loadJournal ?? ((sessionKey) => makeFileCheckpointStore(sessionKey).load());
352
422
  const exists = deps.exists ?? existsSync;
353
423
  const now = deps.now ?? Date.now;
424
+ // Task 8: no default — see the RegisterGoDeps doc comment for why (only
425
+ // index.ts's real wiring makes sense here; absent, `resolveCostText` falls
426
+ // back to the local client estimate).
427
+ const fetchRunSpend = deps.fetchRunSpend;
354
428
  const resolveTicketBrief = deps.resolveTicketBrief ??
355
429
  ((rawArg) => defaultResolveTicketBrief({ baseUrl: deps.baseUrl ?? resolveBaseUrl(), getToken: deps.getToken ?? defaultGetToken }, rawArg));
430
+ const reportCrash = deps.reportCrash ??
431
+ makeCrashReporter({
432
+ baseUrl: deps.baseUrl ?? resolveBaseUrl(),
433
+ getToken: deps.getToken ?? defaultGetToken,
434
+ });
356
435
  pi.registerCommand("go", {
357
436
  description: "Run the grounded multi-agent pipeline (map → plan → implement → review → fix) on a ticket and produce a reviewed change.",
358
437
  handler: async (args, ctx) => {
@@ -667,6 +746,22 @@ export function registerGoCommand(pi, deps = {}) {
667
746
  ...(repoCtx.repo ? { repo: repoCtx.repo } : {}),
668
747
  ...(repoCtx.branch ? { branch: repoCtx.branch } : {}),
669
748
  };
749
+ // YAG-383: this run's dispatches bill under handle.runId, not this
750
+ // driver session's id (see attributionHeaders in config.ts), so /cost's
751
+ // server-authoritative spend fetch would never see them without being
752
+ // told this run id explicitly. Record it the moment it is known (same
753
+ // gate as the attribution header above: only when the run is tracked).
754
+ if (handle.runId)
755
+ recordSessionRun(handle.runId);
756
+ // Desktop parity (Task 9): the desktop's /spend join needs this same
757
+ // server UUID (`run.runId` on the wire record is the LOCAL 8-char short
758
+ // id used for widget/status keys, never the join key). Stamp it the
759
+ // moment it is known and paint IMMEDIATELY — not the trailing-edge
760
+ // debounce — so a fast-finishing run's last few paints still carry it.
761
+ if (handle.runId && run) {
762
+ run.setServerRunId(handle.runId);
763
+ paint(true);
764
+ }
670
765
  // Mark the journal terminal (fail-soft). A run_finish makes the key read as
671
766
  // a COMPLETED run (so the next /go starts fresh, not a bogus resume); its
672
767
  // absence is exactly what marks a crashed run resumable.
@@ -892,8 +987,9 @@ export function registerGoCommand(pi, deps = {}) {
892
987
  const outcome = outcomeFor(finalResult.stopReason);
893
988
  // The end-of-run recap: "what was done" (rounds, findings, stop reason).
894
989
  // The bare `recapBody` rides `finish` as the run_outcome activity (the
895
- // timeline labels that beat "Run outcome"); the prefixed `formatSummary`
896
- // is the standalone toast + scrollback line shown to the user below.
990
+ // timeline labels that beat "Run outcome"); the prefixed
991
+ // `formatSummaryForChat` is the standalone toast + scrollback line
992
+ // shown to the user below.
897
993
  // Non-clean terminal outcome in worktree mode: preserve the work as a
898
994
  // WIP commit so it is never lost and the worktree becomes removable.
899
995
  // (A clean outcome's commit is the FINISH stage's, captured above.)
@@ -902,7 +998,13 @@ export function registerGoCommand(pi, deps = {}) {
902
998
  // both misattribute the commit and spuriously advance a work-less run).
903
999
  const wipSha = finalResult.stopReason === "clean" ? undefined : await preserveWip(finalResult.stopReason);
904
1000
  const runCommitSha = finishInfo?.commitSha ?? wipSha;
905
- const summary = formatSummary(finalResult);
1001
+ // Task 8: START the server-priced fetch here but do NOT await it —
1002
+ // session.finish (the durable finish POST) and everything gated on
1003
+ // it below (recordRunFinish's journal write, settleRun clearing the
1004
+ // in-flight guard) must land regardless of how long this fetch
1005
+ // takes. Only the chat-facing summary/handoff, built after, needs
1006
+ // the resolved value (awaited just below, right before it is used).
1007
+ const costTextPromise = resolveCostText(finalResult, handle.runId, fetchRunSpend, ctx.signal);
906
1008
  await session.finish({
907
1009
  outcome,
908
1010
  stopReason: outcome === "completed" ? undefined : STOP_REASON_COPY[finalResult.stopReason],
@@ -919,15 +1021,26 @@ export function registerGoCommand(pi, deps = {}) {
919
1021
  ...(runCommitSha ? { commitSha: runCommitSha } : {}),
920
1022
  ...(finishInfo?.prUrl ? { prUrl: finishInfo.prUrl } : {}),
921
1023
  });
1024
+ // Only the chat-facing summary/handoff waits on the cost fetch,
1025
+ // now that the durable finish above has already landed.
1026
+ const costText = await costTextPromise;
1027
+ const summary = formatSummaryForChat(finalResult, costText);
922
1028
  notify(summary, summaryTone(finalResult.stopReason));
923
1029
  // The DURABLE record surfaces the unresolved blocking findings IN FULL
924
1030
  // (file:line + message), so a later turn never has to rediscover them.
925
- await deliverHandoff(`${formatHandoff(finalResult, ticket)}${finishNote(finalResult, finish)}${worktreeNote(worktreePath, repoCtx.branch)}`, summaryTone(finalResult.stopReason));
1031
+ await deliverHandoff(`${formatHandoff(finalResult, ticket, costText)}${finishNote(finalResult, finish)}${worktreeNote(worktreePath, repoCtx.branch)}`, summaryTone(finalResult.stopReason));
926
1032
  }
927
1033
  catch (err) {
928
1034
  clearUI();
929
1035
  const message = err instanceof Error ? err.message : String(err);
930
- await session.finish({ outcome: "failed", stopReason: message });
1036
+ // A THROWN error ending the run is crash-shaped (verdict failures
1037
+ // return results instead) — report it, fire-and-forget. The default
1038
+ // reporter never rejects; the catch guards an injected one.
1039
+ void reportCrash(err, "go", runCwd).catch(() => { });
1040
+ // The stopReason travels to the backend run record; scrub it like
1041
+ // every other captured text (a raw error can echo a connection
1042
+ // string or key).
1043
+ await session.finish({ outcome: "failed", stopReason: scrubSecrets(message) });
931
1044
  recordRunFinish("failed");
932
1045
  const wipSha = await preserveWip("failed");
933
1046
  settleRun(runId, { status: "failed", ...(wipSha ? { commitSha: wipSha } : {}) });
@@ -1,11 +1,15 @@
1
1
  /**
2
- * Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only).
2
+ * Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only),
3
+ * plus two opt-in tier lanes: `--judgment` (peak vs advanced on plan+review) and
4
+ * `--map-tier` (standard vs efficient on map).
3
5
  *
4
6
  * It runs the same ticket through the pipeline TWICE in isolated git worktrees
5
7
  * (so the user's tree is never touched): once grounded (the real /go) and once
6
8
  * blind (grounding tools stripped + grounding-free personas), scores each change
7
9
  * for business fit against this company's recorded judgment via the grounded
8
- * review_business_match endpoint, and prints a comparative report.
10
+ * review_business_match endpoint, and prints a comparative report. The
11
+ * `--judgment` and `--map-tier` modes instead run both lanes fully grounded and
12
+ * vary only their one stage's tier, isolating that upgrade's fit delta.
9
13
  *
10
14
  * The report is the deliverable; nothing here is wired to model selection or
11
15
  * routing. The live double-run is dogfood (it spawns real children + models), so
@@ -16,7 +20,7 @@ import type { ExtensionAPI, ExtensionCommandContext } from "@earendil-works/pi-c
16
20
  import { type ComparisonReport, type Lane } from "./eval.js";
17
21
  import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
18
22
  /** Which two-lane comparison a /go-compare invocation runs. */
19
- export type CompareMode = "grounding" | "judgment";
23
+ export type CompareMode = "grounding" | "judgment" | "map-tier";
20
24
  export interface RegisterGoCompareDeps {
21
25
  /** The whole eval runner, injectable so the handler is unit-tested without spawning. */
22
26
  compare?: (ticket: string, ctx: ExtensionCommandContext, mode: CompareMode) => Promise<ComparisonReport>;
@@ -26,10 +30,15 @@ export interface RegisterGoCompareDeps {
26
30
  git?: GitCommand;
27
31
  }
28
32
  /**
29
- * Parse the leading `--judgment` flag. PURE.
33
+ * Parse the leading `--judgment` / `--map-tier` flag. PURE.
30
34
  *
31
35
  * Default stays `grounding` so the existing command is byte-identical for
32
- * callers who pass only a ticket.
36
+ * callers who pass only a ticket. The two flags are mutually exclusive in
37
+ * practice (only one is ever passed); if both appear, `--judgment` wins, but
38
+ * BOTH regexes are stripped unconditionally regardless of which one decided
39
+ * the mode, so the losing flag can never leak into the ticket ref (e.g.
40
+ * `--judgment --map-tier YAG-1` must yield ticket `"YAG-1"`, not
41
+ * `"--map-tier YAG-1"`).
33
42
  */
34
43
  export declare function parseCompareArgs(args: string): {
35
44
  mode: CompareMode;
@@ -38,9 +47,10 @@ export declare function parseCompareArgs(args: string): {
38
47
  /**
39
48
  * The per-lane stage list.
40
49
  *
41
- * Grounding mode strips the grounding tools from the blind lane. Judgment mode
42
- * runs BOTH lanes fully grounded and varies only the plan/review tier, so the
43
- * delta isolates the judgment upgrade rather than confounding it with grounding.
50
+ * Grounding mode strips the grounding tools from the blind lane. Judgment and
51
+ * map-tier modes run BOTH lanes fully grounded and vary only their one tier
52
+ * (plan/review for judgment, map for map-tier), so the delta isolates that one
53
+ * upgrade rather than confounding it with grounding.
44
54
  */
45
55
  export declare function stagesForLane(lane: Lane): import("./types.js").PipelineStage[];
46
56
  export type GitCommand = (args: string[], cwd: string | undefined, signal?: AbortSignal) => Promise<string>;
@@ -1,11 +1,15 @@
1
1
  /**
2
- * Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only).
2
+ * Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only),
3
+ * plus two opt-in tier lanes: `--judgment` (peak vs advanced on plan+review) and
4
+ * `--map-tier` (standard vs efficient on map).
3
5
  *
4
6
  * It runs the same ticket through the pipeline TWICE in isolated git worktrees
5
7
  * (so the user's tree is never touched): once grounded (the real /go) and once
6
8
  * blind (grounding tools stripped + grounding-free personas), scores each change
7
9
  * for business fit against this company's recorded judgment via the grounded
8
- * review_business_match endpoint, and prints a comparative report.
10
+ * review_business_match endpoint, and prints a comparative report. The
11
+ * `--judgment` and `--map-tier` modes instead run both lanes fully grounded and
12
+ * vary only their one stage's tier, isolating that upgrade's fit delta.
9
13
  *
10
14
  * The report is the deliverable; nothing here is wired to model selection or
11
15
  * routing. The live double-run is dogfood (it spawns real children + models), so
@@ -17,32 +21,38 @@ import { mkdtempSync, rmSync } from "node:fs";
17
21
  import { tmpdir } from "node:os";
18
22
  import { join } from "node:path";
19
23
  import { getToken as defaultGetToken, resolveBaseUrl } from "../config.js";
20
- import { blindStages, makeGroundedVsBlindEval, makeJudgmentTierEval, reportOnlyStages, withJudgmentTier, } from "./eval.js";
24
+ import { blindStages, makeGroundedVsBlindEval, makeJudgmentTierEval, makeMapTierEval, reportOnlyStages, withJudgmentTier, withMapTier, } from "./eval.js";
21
25
  import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
22
26
  /**
23
- * Parse the leading `--judgment` flag. PURE.
27
+ * Parse the leading `--judgment` / `--map-tier` flag. PURE.
24
28
  *
25
29
  * Default stays `grounding` so the existing command is byte-identical for
26
- * callers who pass only a ticket.
30
+ * callers who pass only a ticket. The two flags are mutually exclusive in
31
+ * practice (only one is ever passed); if both appear, `--judgment` wins, but
32
+ * BOTH regexes are stripped unconditionally regardless of which one decided
33
+ * the mode, so the losing flag can never leak into the ticket ref (e.g.
34
+ * `--judgment --map-tier YAG-1` must yield ticket `"YAG-1"`, not
35
+ * `"--map-tier YAG-1"`).
27
36
  */
28
37
  export function parseCompareArgs(args) {
29
- let rest = args.trim();
30
- let mode = "grounding";
31
- // Accept the flag anywhere leading, so `/go-compare --judgment YAG-1` and
38
+ const trimmed = args.trim();
39
+ // Accept either flag anywhere leading, so `/go-compare --judgment YAG-1` and
32
40
  // `/go-compare YAG-1 --judgment` both work.
33
- const flag = /(^|\s)--judgment(\s|$)/;
34
- if (flag.test(rest)) {
35
- mode = "judgment";
36
- rest = rest.replace(flag, " ").trim();
37
- }
38
- return { mode, ticket: rest };
41
+ const judgmentFlag = /(^|\s)--judgment(\s|$)/;
42
+ const mapTierFlag = /(^|\s)--map-tier(\s|$)/;
43
+ const hasJudgment = judgmentFlag.test(trimmed);
44
+ const hasMapTier = mapTierFlag.test(trimmed);
45
+ const mode = hasJudgment ? "judgment" : hasMapTier ? "map-tier" : "grounding";
46
+ const ticket = trimmed.replace(judgmentFlag, " ").replace(mapTierFlag, " ").trim();
47
+ return { mode, ticket };
39
48
  }
40
49
  /**
41
50
  * The per-lane stage list.
42
51
  *
43
- * Grounding mode strips the grounding tools from the blind lane. Judgment mode
44
- * runs BOTH lanes fully grounded and varies only the plan/review tier, so the
45
- * delta isolates the judgment upgrade rather than confounding it with grounding.
52
+ * Grounding mode strips the grounding tools from the blind lane. Judgment and
53
+ * map-tier modes run BOTH lanes fully grounded and vary only their one tier
54
+ * (plan/review for judgment, map for map-tier), so the delta isolates that one
55
+ * upgrade rather than confounding it with grounding.
46
56
  */
47
57
  export function stagesForLane(lane) {
48
58
  switch (lane) {
@@ -52,6 +62,10 @@ export function stagesForLane(lane) {
52
62
  return withJudgmentTier(reportOnlyStages(), "peak");
53
63
  case "advanced_judgment":
54
64
  return withJudgmentTier(reportOnlyStages(), "advanced");
65
+ case "map_standard":
66
+ return withMapTier(reportOnlyStages(), "standard");
67
+ case "map_efficient":
68
+ return withMapTier(reportOnlyStages(), "efficient");
55
69
  default:
56
70
  return reportOnlyStages();
57
71
  }
@@ -94,8 +108,9 @@ function defaultCompare(deps) {
94
108
  const result = await runPipeline(ticket, {
95
109
  cwd: tree,
96
110
  signal: ctx.signal,
97
- // Only the blind lane runs ungrounded. Both judgment lanes stay
98
- // grounded so the delta isolates the tier, not the grounding.
111
+ // Only the blind lane runs ungrounded. Both judgment lanes and both
112
+ // map-tier lanes stay grounded so the delta isolates the one tier
113
+ // under test, not the grounding.
99
114
  grounded: lane !== "blind",
100
115
  stages: stagesForLane(lane),
101
116
  childEnv: evalEnv,
@@ -144,14 +159,16 @@ function defaultCompare(deps) {
144
159
  };
145
160
  const evalRunner = mode === "judgment"
146
161
  ? makeJudgmentTierEval({ runLane, scoreFit })
147
- : makeGroundedVsBlindEval({ runLane, scoreFit });
162
+ : mode === "map-tier"
163
+ ? makeMapTierEval({ runLane, scoreFit })
164
+ : makeGroundedVsBlindEval({ runLane, scoreFit });
148
165
  return evalRunner(ticket);
149
166
  };
150
167
  }
151
168
  export function registerGoCompareCommand(pi, deps = {}) {
152
169
  const compare = deps.compare ?? defaultCompare(deps);
153
170
  pi.registerCommand("go-compare", {
154
- description: "Eval only: run a ticket through two lane configurations in isolated worktrees and report the business-fit delta. Default compares grounded vs grounding-suppressed; --judgment compares peak vs advanced on plan+review. Never affects routing.",
171
+ description: "Eval only: run a ticket through two lane configurations in isolated worktrees and report the business-fit delta. Default compares grounded vs grounding-suppressed; --judgment compares peak vs advanced on plan+review; --map-tier compares standard vs efficient on map. Never affects routing.",
155
172
  handler: async (args, ctx) => {
156
173
  const notify = (message, type) => {
157
174
  if (ctx.hasUI)
@@ -159,7 +176,7 @@ export function registerGoCompareCommand(pi, deps = {}) {
159
176
  };
160
177
  const { mode, ticket } = parseCompareArgs(args);
161
178
  if (!ticket) {
162
- notify("Usage: /go-compare [--judgment] <ticket>", "warning");
179
+ notify("Usage: /go-compare [--judgment | --map-tier] <ticket>", "warning");
163
180
  return;
164
181
  }
165
182
  if (!ctx.isIdle()) {
@@ -168,7 +185,9 @@ export function registerGoCompareCommand(pi, deps = {}) {
168
185
  }
169
186
  notify(mode === "judgment"
170
187
  ? "Running the peak vs advanced judgment eval (two isolated passes). This takes a while."
171
- : "Running the grounded vs blind eval (two isolated passes). This takes a while.", "info");
188
+ : mode === "map-tier"
189
+ ? "Running the standard vs efficient map eval (two isolated passes). This takes a while."
190
+ : "Running the grounded vs blind eval (two isolated passes). This takes a while.", "info");
172
191
  try {
173
192
  const out = await compare(ticket, ctx, mode);
174
193
  await pi.sendUserMessage(out.report);