@yagni-app/code-staging 0.1.0-staging.997.1 → 0.2.0-staging.1025.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +58 -9
- package/dist/claudeCompat.d.ts +36 -5
- package/dist/claudeCompat.js +85 -23
- package/dist/claudePlugins.d.ts +109 -0
- package/dist/claudePlugins.js +336 -0
- package/dist/cli.js +14 -4
- package/dist/crashReport.d.ts +135 -0
- package/dist/crashReport.js +291 -0
- package/dist/doctor.d.ts +21 -0
- package/dist/doctor.js +52 -0
- package/dist/extension/askAdvisorTool.js +7 -1
- package/dist/extension/bless.js +16 -3
- package/dist/extension/boostCommand.d.ts +144 -0
- package/dist/extension/boostCommand.js +263 -0
- package/dist/extension/branding.d.ts +31 -0
- package/dist/extension/branding.js +37 -0
- package/dist/extension/chipEditor.js +7 -3
- package/dist/extension/claudeRules.d.ts +54 -0
- package/dist/extension/claudeRules.js +180 -0
- package/dist/extension/config.d.ts +61 -0
- package/dist/extension/config.js +86 -0
- package/dist/extension/costHud.d.ts +128 -15
- package/dist/extension/costHud.js +189 -19
- package/dist/extension/crashReport.d.ts +89 -0
- package/dist/extension/crashReport.js +241 -0
- package/dist/extension/index.d.ts +43 -4
- package/dist/extension/index.js +241 -32
- package/dist/extension/initPass.d.ts +65 -47
- package/dist/extension/initPass.js +145 -145
- package/dist/extension/mcpTools.d.ts +57 -0
- package/dist/extension/mcpTools.js +132 -0
- package/dist/extension/pipeline/eval.d.ts +42 -5
- package/dist/extension/pipeline/eval.js +44 -0
- package/dist/extension/pipeline/goCommand.d.ts +18 -0
- package/dist/extension/pipeline/goCommand.js +139 -26
- package/dist/extension/pipeline/goCompareCommand.d.ts +18 -8
- package/dist/extension/pipeline/goCompareCommand.js +42 -23
- package/dist/extension/pipeline/orchestrator.js +9 -0
- package/dist/extension/pipeline/runCostTable.d.ts +37 -0
- package/dist/extension/pipeline/runCostTable.js +165 -0
- package/dist/extension/pipeline/runState.d.ts +19 -0
- package/dist/extension/pipeline/runState.js +11 -0
- package/dist/extension/pipeline/runner.d.ts +19 -0
- package/dist/extension/pipeline/runner.js +13 -1
- package/dist/extension/pipeline/scrubSecrets.js +2 -2
- package/dist/extension/pipeline/stages.d.ts +3 -1
- package/dist/extension/pipeline/stages.js +3 -1
- package/dist/extension/pipeline/types.d.ts +7 -4
- package/dist/extension/pipeline/verify.js +6 -1
- package/dist/extension/pipeline/worktree.js +3 -1
- package/dist/extension/provider.d.ts +7 -1
- package/dist/extension/provider.js +8 -1
- package/dist/extension/recall.js +5 -2
- package/dist/extension/rerouteNotice.d.ts +42 -0
- package/dist/extension/rerouteNotice.js +67 -0
- package/dist/extension/sessionRuns.d.ts +45 -0
- package/dist/extension/sessionRuns.js +77 -0
- package/dist/extension/subagents.d.ts +17 -7
- package/dist/extension/subagents.js +52 -7
- package/dist/launch.d.ts +17 -3
- package/dist/launch.js +22 -9
- package/dist/login.d.ts +7 -0
- package/dist/login.js +3 -1
- package/package.json +2 -2
|
@@ -45,15 +45,32 @@ export declare const JUDGMENT_STAGE_IDS: readonly ["plan", "review"];
|
|
|
45
45
|
* moves exactly one variable.
|
|
46
46
|
*/
|
|
47
47
|
export declare function withJudgmentTier(stages: PipelineStage[] | undefined, tier: ModelTier): PipelineStage[];
|
|
48
|
+
/**
|
|
49
|
+
* The map stage's model choice IS the map-tier decision. Kept as data,
|
|
50
|
+
* mirroring {@link JUDGMENT_STAGE_IDS}, so the transform below and any future
|
|
51
|
+
* lane agree on what "map tier" means. Currently a single stage, but an array
|
|
52
|
+
* so a future lane that widens "map" to more than one stage is a data change.
|
|
53
|
+
*/
|
|
54
|
+
export declare const MAP_STAGE_IDS: readonly ["map"];
|
|
55
|
+
/**
|
|
56
|
+
* Set the map stage to a given tier, leaving every other stage untouched.
|
|
57
|
+
* PURE: returns new stage objects, mirroring {@link withJudgmentTier}'s shape.
|
|
58
|
+
*
|
|
59
|
+
* This is the seam for the Task 13 gate: does map on `standard` produce a
|
|
60
|
+
* meaningfully better repo brief than map on `efficient`, or is the cheaper
|
|
61
|
+
* tier good enough for a step whose job is compression, not judgment.
|
|
62
|
+
*/
|
|
63
|
+
export declare function withMapTier(stages: PipelineStage[] | undefined, tier: ModelTier): PipelineStage[];
|
|
48
64
|
/**
|
|
49
65
|
* Which lane an outcome belongs to.
|
|
50
66
|
*
|
|
51
|
-
*
|
|
52
|
-
* beat guessing)
|
|
53
|
-
* plan+review produce better changes)
|
|
54
|
-
*
|
|
67
|
+
* Three comparisons ride this engine today: grounded-vs-blind (does grounding
|
|
68
|
+
* beat guessing), peak-vs-advanced judgment (does the stronger model on
|
|
69
|
+
* plan+review produce better changes), and standard-vs-efficient map (does
|
|
70
|
+
* the pricier tier on the map stage produce better changes). All are two-lane
|
|
71
|
+
* A/Bs scored the same way, so the lane is DATA rather than a hardcoded union.
|
|
55
72
|
*/
|
|
56
|
-
export type Lane = "grounded" | "blind" | "peak_judgment" | "advanced_judgment";
|
|
73
|
+
export type Lane = "grounded" | "blind" | "peak_judgment" | "advanced_judgment" | "map_standard" | "map_efficient";
|
|
57
74
|
/** A lane's id plus how it is named in the report. */
|
|
58
75
|
export interface LaneSpec {
|
|
59
76
|
id: Lane;
|
|
@@ -81,6 +98,23 @@ export declare const JUDGMENT_LANES: {
|
|
|
81
98
|
readonly label: "Advanced judgment";
|
|
82
99
|
};
|
|
83
100
|
};
|
|
101
|
+
/**
|
|
102
|
+
* The map-tier question (Task 13): standard on map against efficient on map.
|
|
103
|
+
* `standard` is the treatment (the pricier, currently-shipped lane) so the
|
|
104
|
+
* verdict answers "does paying for standard over efficient on map pay off",
|
|
105
|
+
* matching how {@link JUDGMENT_LANES} always puts the more expensive lane in
|
|
106
|
+
* `treatment`.
|
|
107
|
+
*/
|
|
108
|
+
export declare const MAP_TIER_LANES: {
|
|
109
|
+
readonly treatment: {
|
|
110
|
+
readonly id: "map_standard";
|
|
111
|
+
readonly label: "Map standard";
|
|
112
|
+
};
|
|
113
|
+
readonly baseline: {
|
|
114
|
+
readonly id: "map_efficient";
|
|
115
|
+
readonly label: "Map efficient";
|
|
116
|
+
};
|
|
117
|
+
};
|
|
84
118
|
/**
|
|
85
119
|
* A lane's business-fit score, produced by running the GROUNDED
|
|
86
120
|
* review_business_match against that lane's final diff (so both lanes are judged
|
|
@@ -139,6 +173,7 @@ export interface ComparisonCopy {
|
|
|
139
173
|
}
|
|
140
174
|
export declare const GROUNDED_COPY: ComparisonCopy;
|
|
141
175
|
export declare const JUDGMENT_COPY: ComparisonCopy;
|
|
176
|
+
export declare const MAP_TIER_COPY: ComparisonCopy;
|
|
142
177
|
/**
|
|
143
178
|
* Render the comparative report a human reads. Leads with the headline conflict
|
|
144
179
|
* delta (the GATE number), then each lane's stop reason, finding tally, and fit
|
|
@@ -165,4 +200,6 @@ export declare function makeComparisonEval(seams: ComparisonSeams, lanes: {
|
|
|
165
200
|
export declare function makeGroundedVsBlindEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
|
|
166
201
|
/** The judgment-tier question: peak on plan+review against advanced (YAG-380). */
|
|
167
202
|
export declare function makeJudgmentTierEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
|
|
203
|
+
/** The map-tier question: standard on map against efficient on map (Task 13). */
|
|
204
|
+
export declare function makeMapTierEval(seams: ComparisonSeams): (ticket: string) => Promise<ComparisonReport>;
|
|
168
205
|
//# sourceMappingURL=eval.d.ts.map
|
|
@@ -61,6 +61,25 @@ export function withJudgmentTier(stages = selectStages("full"), tier) {
|
|
|
61
61
|
const judgment = new Set(JUDGMENT_STAGE_IDS);
|
|
62
62
|
return stages.map((s) => (judgment.has(s.id) ? { ...s, model: tier } : s));
|
|
63
63
|
}
|
|
64
|
+
/**
|
|
65
|
+
* The map stage's model choice IS the map-tier decision. Kept as data,
|
|
66
|
+
* mirroring {@link JUDGMENT_STAGE_IDS}, so the transform below and any future
|
|
67
|
+
* lane agree on what "map tier" means. Currently a single stage, but an array
|
|
68
|
+
* so a future lane that widens "map" to more than one stage is a data change.
|
|
69
|
+
*/
|
|
70
|
+
export const MAP_STAGE_IDS = ["map"];
|
|
71
|
+
/**
|
|
72
|
+
* Set the map stage to a given tier, leaving every other stage untouched.
|
|
73
|
+
* PURE: returns new stage objects, mirroring {@link withJudgmentTier}'s shape.
|
|
74
|
+
*
|
|
75
|
+
* This is the seam for the Task 13 gate: does map on `standard` produce a
|
|
76
|
+
* meaningfully better repo brief than map on `efficient`, or is the cheaper
|
|
77
|
+
* tier good enough for a step whose job is compression, not judgment.
|
|
78
|
+
*/
|
|
79
|
+
export function withMapTier(stages = selectStages("full"), tier) {
|
|
80
|
+
const mapStages = new Set(MAP_STAGE_IDS);
|
|
81
|
+
return stages.map((s) => (mapStages.has(s.id) ? { ...s, model: tier } : s));
|
|
82
|
+
}
|
|
64
83
|
/** The grounding GATE: the real /go against a grounding-suppressed twin. */
|
|
65
84
|
export const GROUNDED_LANES = {
|
|
66
85
|
treatment: { id: "grounded", label: "Grounded" },
|
|
@@ -71,6 +90,17 @@ export const JUDGMENT_LANES = {
|
|
|
71
90
|
treatment: { id: "peak_judgment", label: "Peak judgment" },
|
|
72
91
|
baseline: { id: "advanced_judgment", label: "Advanced judgment" },
|
|
73
92
|
};
|
|
93
|
+
/**
|
|
94
|
+
* The map-tier question (Task 13): standard on map against efficient on map.
|
|
95
|
+
* `standard` is the treatment (the pricier, currently-shipped lane) so the
|
|
96
|
+
* verdict answers "does paying for standard over efficient on map pay off",
|
|
97
|
+
* matching how {@link JUDGMENT_LANES} always puts the more expensive lane in
|
|
98
|
+
* `treatment`.
|
|
99
|
+
*/
|
|
100
|
+
export const MAP_TIER_LANES = {
|
|
101
|
+
treatment: { id: "map_standard", label: "Map standard" },
|
|
102
|
+
baseline: { id: "map_efficient", label: "Map efficient" },
|
|
103
|
+
};
|
|
74
104
|
export const GROUNDED_COPY = {
|
|
75
105
|
title: "Grounded vs blind eval",
|
|
76
106
|
deltaLabel: "FIT delta (blind conflicts minus grounded conflicts)",
|
|
@@ -91,6 +121,16 @@ export const JUDGMENT_COPY = {
|
|
|
91
121
|
: "Peak and advanced judgment tied on flagged conflicts for this ticket.",
|
|
92
122
|
footer: "Report only. This comparison is never wired to model selection, tiers, or routing. Only plan and review differ between the lanes, so the delta isolates the judgment tier. One ticket is an anecdote; run several before concluding.",
|
|
93
123
|
};
|
|
124
|
+
export const MAP_TIER_COPY = {
|
|
125
|
+
title: "Standard vs efficient map eval",
|
|
126
|
+
deltaLabel: "FIT delta (efficient-map conflicts minus standard-map conflicts)",
|
|
127
|
+
verdict: (favor) => favor > 0
|
|
128
|
+
? `Standard on map REDUCED business-fit conflicts by ${favor} on this ticket.`
|
|
129
|
+
: favor < 0
|
|
130
|
+
? `Standard on map showed ${Math.abs(favor)} MORE flagged conflict${Math.abs(favor) === 1 ? "" : "s"} on this ticket. The map tier upgrade did not pay for itself here.`
|
|
131
|
+
: "Standard and efficient map tied on flagged conflicts for this ticket.",
|
|
132
|
+
footer: "Report only. This comparison is never wired to model selection, tiers, or routing. Only the map stage differs between the lanes, so the delta isolates the map tier. One ticket is an anecdote; run several before concluding.",
|
|
133
|
+
};
|
|
94
134
|
const blocking = (findings) => findings.filter((f) => f.severity === "critical" || f.severity === "high").length;
|
|
95
135
|
/**
|
|
96
136
|
* Render the comparative report a human reads. Leads with the headline conflict
|
|
@@ -179,4 +219,8 @@ export function makeGroundedVsBlindEval(seams) {
|
|
|
179
219
|
export function makeJudgmentTierEval(seams) {
|
|
180
220
|
return makeComparisonEval(seams, JUDGMENT_LANES, JUDGMENT_COPY);
|
|
181
221
|
}
|
|
222
|
+
/** The map-tier question: standard on map against efficient on map (Task 13). */
|
|
223
|
+
export function makeMapTierEval(seams) {
|
|
224
|
+
return makeComparisonEval(seams, MAP_TIER_LANES, MAP_TIER_COPY);
|
|
225
|
+
}
|
|
182
226
|
//# sourceMappingURL=eval.js.map
|
|
@@ -58,6 +58,8 @@
|
|
|
58
58
|
* is fail-soft, so recording can never break `/go`.
|
|
59
59
|
*/
|
|
60
60
|
import type { ExtensionAPI, ExtensionCommandContext } from "@earendil-works/pi-coding-agent";
|
|
61
|
+
import type { SpendResponse } from "../costHud.js";
|
|
62
|
+
import { type CrashReporter } from "../crashReport.js";
|
|
61
63
|
import { runFinish as defaultRunFinish } from "./finish.js";
|
|
62
64
|
import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
63
65
|
import { makeRunSession as defaultMakeRunSession } from "./runSession.js";
|
|
@@ -97,12 +99,28 @@ export interface RegisterGoDeps {
|
|
|
97
99
|
runFinish?: typeof defaultRunFinish;
|
|
98
100
|
/** Injectable journal read for cross-process liveness (default: the file checkpoint store). */
|
|
99
101
|
loadJournal?: (sessionKey: string) => CheckpointRecord[];
|
|
102
|
+
/**
|
|
103
|
+
* Task 8: fetch the server-authoritative per-stage spend for one /go run
|
|
104
|
+
* (`GET /api/yagni-code/spend?runId=`, wired in index.ts reusing Task 7's
|
|
105
|
+
* fetch machinery). Absent (no dep wired), throwing, or resolving null all
|
|
106
|
+
* fall back to the local `runCostNote` client estimate — see
|
|
107
|
+
* `resolveCostText` below. No default here: unlike every other network dep
|
|
108
|
+
* on this interface, index.ts is the ONLY real wiring (there is nothing
|
|
109
|
+
* sensible to default to without a baseUrl/token), and every test that
|
|
110
|
+
* cares injects its own fake.
|
|
111
|
+
*/
|
|
112
|
+
fetchRunSpend?: (runId: string, signal?: AbortSignal) => Promise<SpendResponse | null>;
|
|
100
113
|
/** Injectable fs existence check (worktree adoption on resume). */
|
|
101
114
|
exists?: (path: string) => boolean;
|
|
102
115
|
/** Clock seam for registry rows + staleness. */
|
|
103
116
|
now?: () => number;
|
|
104
117
|
baseUrl?: string;
|
|
105
118
|
getToken?: () => string | undefined;
|
|
119
|
+
/**
|
|
120
|
+
* Crash reporter for the run's terminal catch (a THROWN error, not a
|
|
121
|
+
* verdict-shaped failure). Fail-soft and fire-and-forget by contract.
|
|
122
|
+
*/
|
|
123
|
+
reportCrash?: CrashReporter;
|
|
106
124
|
}
|
|
107
125
|
/**
|
|
108
126
|
* Per-run UI key prefix: the widget/status keys are `yagni-go:<runShortId>` so
|
|
@@ -65,8 +65,11 @@ import { eventToLine } from "./activity.js";
|
|
|
65
65
|
import { ActivityFeed, SPINNER_FRAMES } from "./activityFeed.js";
|
|
66
66
|
import { RunState } from "./runState.js";
|
|
67
67
|
import { aggregateRunUsage } from "./budget.js";
|
|
68
|
+
import { formatRunCostTable } from "./runCostTable.js";
|
|
68
69
|
import { makeCombinedCheckpointStore, makeFileCheckpointStore, makePiJournalCheckpointStore, } from "./checkpoint.js";
|
|
69
70
|
import { getToken as defaultGetToken, resolveBaseUrl } from "../config.js";
|
|
71
|
+
import { makeCrashReporter } from "../crashReport.js";
|
|
72
|
+
import { scrubSecrets } from "./scrubSecrets.js";
|
|
70
73
|
import { isDesktopSurface } from "../surface.js";
|
|
71
74
|
import { runFinish as defaultRunFinish, verifyTrailerValue, } from "./finish.js";
|
|
72
75
|
import { GO_USAGE, parseGoArgs } from "./goFlags.js";
|
|
@@ -75,6 +78,7 @@ import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
|
75
78
|
import { planResume } from "./resume.js";
|
|
76
79
|
import { activeRunCount, beginRun, classifyRunLiveness, findActiveRunByTicket, isRunInFlight, isTerminalStatus, lastJournalTs, loadRegistryRows, MAX_CONCURRENT_RUNS, settleRun, trackRunPromise, worktreesDir, } from "./runRegistry.js";
|
|
77
80
|
import { makeRunSession as defaultMakeRunSession } from "./runSession.js";
|
|
81
|
+
import { recordSessionRun } from "../sessionRuns.js";
|
|
78
82
|
import { resolveTicketBrief as defaultResolveTicketBrief } from "./ticketResolution.js";
|
|
79
83
|
import { snapshotWorkspace as defaultSnapshotWorkspace } from "./workspace.js";
|
|
80
84
|
import { parseChangedPaths } from "./verify.js";
|
|
@@ -168,9 +172,9 @@ function finishNote(result, finish) {
|
|
|
168
172
|
return "";
|
|
169
173
|
return `\n\n${parts.join("\n")}`;
|
|
170
174
|
}
|
|
171
|
-
/** Blocking-finding detail appended to the DURABLE handoff (spec: surface findings, not counts). */
|
|
172
|
-
function formatHandoff(result, ticket) {
|
|
173
|
-
const summary =
|
|
175
|
+
/** Blocking-finding detail appended to the DURABLE handoff (spec: surface findings, not counts). `costText` — see `resolveCostText` (Task 8). */
|
|
176
|
+
function formatHandoff(result, ticket, costText) {
|
|
177
|
+
const summary = formatSummaryForChat(result, costText);
|
|
174
178
|
const degraded = `${degradedLensNote(result)}${verifyNote(result)}`;
|
|
175
179
|
const blocking = result.findings.filter((f) => f.severity === "critical" || f.severity === "high");
|
|
176
180
|
if (blocking.length === 0)
|
|
@@ -235,23 +239,41 @@ function outcomeFor(stopReason) {
|
|
|
235
239
|
return "failed";
|
|
236
240
|
return "completed"; // clean | round_cap are both real completions
|
|
237
241
|
}
|
|
238
|
-
/**
|
|
239
|
-
|
|
240
|
-
* "YAGNI Code /go finished:" prefix. This is what rides `finish` as the
|
|
241
|
-
* `run_outcome` activity body. The Work-page timeline already labels that beat
|
|
242
|
-
* "Run outcome", so prefixing here would double up into "Run outcome: YAGNI
|
|
243
|
-
* Code /go finished: ...". `formatSummary` re-adds the prefix for the standalone
|
|
244
|
-
* toast + scrollback line, where the context is useful on its own.
|
|
245
|
-
*/
|
|
246
|
-
function recapBody(result) {
|
|
242
|
+
/** Stop reason + the review tally, with NO cost segment and no prefix — the shared core of `recapBody`/`recapBodyForChat` below. */
|
|
243
|
+
function tallyLine(result) {
|
|
247
244
|
const rounds = result.rounds.length;
|
|
248
245
|
const total = result.findings.length;
|
|
249
246
|
const blocking = result.findings.filter((f) => f.severity === "critical" || f.severity === "high").length;
|
|
250
247
|
const reason = STOP_REASON_COPY[result.stopReason];
|
|
251
248
|
return (`${reason}. ` +
|
|
252
249
|
`${rounds} review round${rounds === 1 ? "" : "s"}, ` +
|
|
253
|
-
`${total} finding${total === 1 ? "" : "s"} (${blocking} blocking).`
|
|
254
|
-
|
|
250
|
+
`${total} finding${total === 1 ? "" : "s"} (${blocking} blocking).`);
|
|
251
|
+
}
|
|
252
|
+
/**
|
|
253
|
+
* The bare "what was done" recap — stop reason + the review tally + the LOCAL
|
|
254
|
+
* cost estimate — with NO "YAGNI Code /go finished:" prefix. This is what
|
|
255
|
+
* rides `finish` as the `run_outcome` activity body (session.finish's
|
|
256
|
+
* `outcomeSummary`), which must stay synchronous: it is recorded as part of
|
|
257
|
+
* ending the run and must never wait on the Task 8 server spend fetch. The
|
|
258
|
+
* chat-facing summary (`recapBodyForChat`, below) is the one that prefers the
|
|
259
|
+
* server-priced table when it is available. The Work-page timeline already
|
|
260
|
+
* labels this beat "Run outcome", so prefixing here would double up into "Run
|
|
261
|
+
* outcome: YAGNI Code /go finished: ...". `formatSummaryForChat` re-adds the
|
|
262
|
+
* prefix for the standalone toast + scrollback line, where the context is
|
|
263
|
+
* useful on its own.
|
|
264
|
+
*/
|
|
265
|
+
function recapBody(result) {
|
|
266
|
+
return `${tallyLine(result)}${runCostNote(result)}`;
|
|
267
|
+
}
|
|
268
|
+
/**
|
|
269
|
+
* Chat-facing recap: the same tally as `recapBody`, but the cost segment is
|
|
270
|
+
* `costText` — precomputed by `resolveCostText` (Task 8), which is either the
|
|
271
|
+
* multi-line server-priced table or the labeled local-estimate fallback.
|
|
272
|
+
* Kept distinct from `recapBody` because `costText` requires an async fetch
|
|
273
|
+
* that `recapBody`'s caller (session.finish's outcomeSummary) cannot wait on.
|
|
274
|
+
*/
|
|
275
|
+
function recapBodyForChat(result, costText) {
|
|
276
|
+
return `${tallyLine(result)}${costText}`;
|
|
255
277
|
}
|
|
256
278
|
/**
|
|
257
279
|
* What the run actually cost.
|
|
@@ -271,14 +293,62 @@ function runCostNote(result) {
|
|
|
271
293
|
return ` Run cost: $${usage.cost.toFixed(2)} over ${tokens} tokens.`;
|
|
272
294
|
}
|
|
273
295
|
/**
|
|
274
|
-
*
|
|
275
|
-
*
|
|
276
|
-
*
|
|
277
|
-
*
|
|
278
|
-
*
|
|
296
|
+
* Task 8: resolve the cost segment for the CHAT-FACING summary (the toast +
|
|
297
|
+
* the durable handoff) — the server-priced per-stage table
|
|
298
|
+
* (`formatRunCostTable`) when this run is tracked and the fetch succeeds,
|
|
299
|
+
* else today's local `runCostNote` estimate, explicitly labeled "(client
|
|
300
|
+
* estimate)" so it is never mistaken for the priced total. Mirrors
|
|
301
|
+
* `registerCostCommand`'s own fail-soft preference order in costHud.ts.
|
|
302
|
+
*
|
|
303
|
+
* Never throws and never blocks long: `fetchRunSpend` (index.ts) already
|
|
304
|
+
* carries its own request timeout (same 5s ceiling as /cost's fetches), and
|
|
305
|
+
* this wraps the ENTIRE body — not just the fetch — in try/catch. That is
|
|
306
|
+
* deliberate, not belt-and-braces: the caller (see `runToCompletion` below)
|
|
307
|
+
* starts this promise and leaves it un-awaited for a while so session.finish
|
|
308
|
+
* never waits on it, so a throw ANYWHERE in here between promise creation and
|
|
309
|
+
* that later `await` — including from `runCostNote`/`formatRunCostTable` on a
|
|
310
|
+
* malformed result, not just from the fetch itself — would otherwise surface
|
|
311
|
+
* as a guaranteed unhandled rejection during that gap, independent of
|
|
312
|
+
* whatever the caller does with the eventual value.
|
|
313
|
+
*
|
|
314
|
+
* An empty-rows response is treated the same as "unavailable" (mirrors
|
|
315
|
+
* registerCostCommand's `serverLooksStale` guard): a run that billed real
|
|
316
|
+
* spend but whose rows have not landed yet would otherwise render a
|
|
317
|
+
* misleading "$0.00 (server)" table instead of the honest local estimate.
|
|
279
318
|
*/
|
|
280
|
-
function
|
|
281
|
-
|
|
319
|
+
async function resolveCostText(result, runId, fetchRunSpend, signal) {
|
|
320
|
+
try {
|
|
321
|
+
if (runId && fetchRunSpend) {
|
|
322
|
+
try {
|
|
323
|
+
const spend = await fetchRunSpend(runId, signal);
|
|
324
|
+
if (spend && spend.rows.length > 0)
|
|
325
|
+
return formatRunCostTable(spend);
|
|
326
|
+
}
|
|
327
|
+
catch {
|
|
328
|
+
/* fail-soft: fall through to the local client estimate below */
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
const local = runCostNote(result);
|
|
332
|
+
return local ? `${local} (client estimate)` : "";
|
|
333
|
+
}
|
|
334
|
+
catch {
|
|
335
|
+
// Truly can't say anything honest (even the local estimate blew up):
|
|
336
|
+
// no cost text at all, matching runCostNote's own "nothing to report" ""
|
|
337
|
+
// convention rather than fabricating a number.
|
|
338
|
+
return "";
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
/**
|
|
342
|
+
* One-line-or-more, human-readable outcome for the session + a UI toast. This
|
|
343
|
+
* is the DURABLE record (sent via `sendUserMessage`, fired even on headless
|
|
344
|
+
* runs where no `ActivityFeed` is constructed), so it is intentionally kept
|
|
345
|
+
* separate from `ActivityFeed.finalLine` (the feed's compact, UI-only outcome
|
|
346
|
+
* glyph-line) rather than the two sharing one outcome-copy source. `costText`
|
|
347
|
+
* carries the Task 8 cost segment (`resolveCostText`'s result: the
|
|
348
|
+
* server-priced table, or its labeled local-estimate fallback).
|
|
349
|
+
*/
|
|
350
|
+
function formatSummaryForChat(result, costText) {
|
|
351
|
+
return `YAGNI Code /go finished: ${recapBodyForChat(result, costText)}`;
|
|
282
352
|
}
|
|
283
353
|
/**
|
|
284
354
|
* User-facing notice when a /go run is NOT being recorded on the Work page.
|
|
@@ -351,8 +421,17 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
351
421
|
const loadJournal = deps.loadJournal ?? ((sessionKey) => makeFileCheckpointStore(sessionKey).load());
|
|
352
422
|
const exists = deps.exists ?? existsSync;
|
|
353
423
|
const now = deps.now ?? Date.now;
|
|
424
|
+
// Task 8: no default — see the RegisterGoDeps doc comment for why (only
|
|
425
|
+
// index.ts's real wiring makes sense here; absent, `resolveCostText` falls
|
|
426
|
+
// back to the local client estimate).
|
|
427
|
+
const fetchRunSpend = deps.fetchRunSpend;
|
|
354
428
|
const resolveTicketBrief = deps.resolveTicketBrief ??
|
|
355
429
|
((rawArg) => defaultResolveTicketBrief({ baseUrl: deps.baseUrl ?? resolveBaseUrl(), getToken: deps.getToken ?? defaultGetToken }, rawArg));
|
|
430
|
+
const reportCrash = deps.reportCrash ??
|
|
431
|
+
makeCrashReporter({
|
|
432
|
+
baseUrl: deps.baseUrl ?? resolveBaseUrl(),
|
|
433
|
+
getToken: deps.getToken ?? defaultGetToken,
|
|
434
|
+
});
|
|
356
435
|
pi.registerCommand("go", {
|
|
357
436
|
description: "Run the grounded multi-agent pipeline (map → plan → implement → review → fix) on a ticket and produce a reviewed change.",
|
|
358
437
|
handler: async (args, ctx) => {
|
|
@@ -667,6 +746,22 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
667
746
|
...(repoCtx.repo ? { repo: repoCtx.repo } : {}),
|
|
668
747
|
...(repoCtx.branch ? { branch: repoCtx.branch } : {}),
|
|
669
748
|
};
|
|
749
|
+
// YAG-383: this run's dispatches bill under handle.runId, not this
|
|
750
|
+
// driver session's id (see attributionHeaders in config.ts), so /cost's
|
|
751
|
+
// server-authoritative spend fetch would never see them without being
|
|
752
|
+
// told this run id explicitly. Record it the moment it is known (same
|
|
753
|
+
// gate as the attribution header above: only when the run is tracked).
|
|
754
|
+
if (handle.runId)
|
|
755
|
+
recordSessionRun(handle.runId);
|
|
756
|
+
// Desktop parity (Task 9): the desktop's /spend join needs this same
|
|
757
|
+
// server UUID (`run.runId` on the wire record is the LOCAL 8-char short
|
|
758
|
+
// id used for widget/status keys, never the join key). Stamp it the
|
|
759
|
+
// moment it is known and paint IMMEDIATELY — not the trailing-edge
|
|
760
|
+
// debounce — so a fast-finishing run's last few paints still carry it.
|
|
761
|
+
if (handle.runId && run) {
|
|
762
|
+
run.setServerRunId(handle.runId);
|
|
763
|
+
paint(true);
|
|
764
|
+
}
|
|
670
765
|
// Mark the journal terminal (fail-soft). A run_finish makes the key read as
|
|
671
766
|
// a COMPLETED run (so the next /go starts fresh, not a bogus resume); its
|
|
672
767
|
// absence is exactly what marks a crashed run resumable.
|
|
@@ -892,8 +987,9 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
892
987
|
const outcome = outcomeFor(finalResult.stopReason);
|
|
893
988
|
// The end-of-run recap: "what was done" (rounds, findings, stop reason).
|
|
894
989
|
// The bare `recapBody` rides `finish` as the run_outcome activity (the
|
|
895
|
-
// timeline labels that beat "Run outcome"); the prefixed
|
|
896
|
-
// is the standalone toast + scrollback line
|
|
990
|
+
// timeline labels that beat "Run outcome"); the prefixed
|
|
991
|
+
// `formatSummaryForChat` is the standalone toast + scrollback line
|
|
992
|
+
// shown to the user below.
|
|
897
993
|
// Non-clean terminal outcome in worktree mode: preserve the work as a
|
|
898
994
|
// WIP commit so it is never lost and the worktree becomes removable.
|
|
899
995
|
// (A clean outcome's commit is the FINISH stage's, captured above.)
|
|
@@ -902,7 +998,13 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
902
998
|
// both misattribute the commit and spuriously advance a work-less run).
|
|
903
999
|
const wipSha = finalResult.stopReason === "clean" ? undefined : await preserveWip(finalResult.stopReason);
|
|
904
1000
|
const runCommitSha = finishInfo?.commitSha ?? wipSha;
|
|
905
|
-
|
|
1001
|
+
// Task 8: START the server-priced fetch here but do NOT await it —
|
|
1002
|
+
// session.finish (the durable finish POST) and everything gated on
|
|
1003
|
+
// it below (recordRunFinish's journal write, settleRun clearing the
|
|
1004
|
+
// in-flight guard) must land regardless of how long this fetch
|
|
1005
|
+
// takes. Only the chat-facing summary/handoff, built after, needs
|
|
1006
|
+
// the resolved value (awaited just below, right before it is used).
|
|
1007
|
+
const costTextPromise = resolveCostText(finalResult, handle.runId, fetchRunSpend, ctx.signal);
|
|
906
1008
|
await session.finish({
|
|
907
1009
|
outcome,
|
|
908
1010
|
stopReason: outcome === "completed" ? undefined : STOP_REASON_COPY[finalResult.stopReason],
|
|
@@ -919,15 +1021,26 @@ export function registerGoCommand(pi, deps = {}) {
|
|
|
919
1021
|
...(runCommitSha ? { commitSha: runCommitSha } : {}),
|
|
920
1022
|
...(finishInfo?.prUrl ? { prUrl: finishInfo.prUrl } : {}),
|
|
921
1023
|
});
|
|
1024
|
+
// Only the chat-facing summary/handoff waits on the cost fetch,
|
|
1025
|
+
// now that the durable finish above has already landed.
|
|
1026
|
+
const costText = await costTextPromise;
|
|
1027
|
+
const summary = formatSummaryForChat(finalResult, costText);
|
|
922
1028
|
notify(summary, summaryTone(finalResult.stopReason));
|
|
923
1029
|
// The DURABLE record surfaces the unresolved blocking findings IN FULL
|
|
924
1030
|
// (file:line + message), so a later turn never has to rediscover them.
|
|
925
|
-
await deliverHandoff(`${formatHandoff(finalResult, ticket)}${finishNote(finalResult, finish)}${worktreeNote(worktreePath, repoCtx.branch)}`, summaryTone(finalResult.stopReason));
|
|
1031
|
+
await deliverHandoff(`${formatHandoff(finalResult, ticket, costText)}${finishNote(finalResult, finish)}${worktreeNote(worktreePath, repoCtx.branch)}`, summaryTone(finalResult.stopReason));
|
|
926
1032
|
}
|
|
927
1033
|
catch (err) {
|
|
928
1034
|
clearUI();
|
|
929
1035
|
const message = err instanceof Error ? err.message : String(err);
|
|
930
|
-
|
|
1036
|
+
// A THROWN error ending the run is crash-shaped (verdict failures
|
|
1037
|
+
// return results instead) — report it, fire-and-forget. The default
|
|
1038
|
+
// reporter never rejects; the catch guards an injected one.
|
|
1039
|
+
void reportCrash(err, "go", runCwd).catch(() => { });
|
|
1040
|
+
// The stopReason travels to the backend run record; scrub it like
|
|
1041
|
+
// every other captured text (a raw error can echo a connection
|
|
1042
|
+
// string or key).
|
|
1043
|
+
await session.finish({ outcome: "failed", stopReason: scrubSecrets(message) });
|
|
931
1044
|
recordRunFinish("failed");
|
|
932
1045
|
const wipSha = await preserveWip("failed");
|
|
933
1046
|
settleRun(runId, { status: "failed", ...(wipSha ? { commitSha: wipSha } : {}) });
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only)
|
|
2
|
+
* Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only),
|
|
3
|
+
* plus two opt-in tier lanes: `--judgment` (peak vs advanced on plan+review) and
|
|
4
|
+
* `--map-tier` (standard vs efficient on map).
|
|
3
5
|
*
|
|
4
6
|
* It runs the same ticket through the pipeline TWICE in isolated git worktrees
|
|
5
7
|
* (so the user's tree is never touched): once grounded (the real /go) and once
|
|
6
8
|
* blind (grounding tools stripped + grounding-free personas), scores each change
|
|
7
9
|
* for business fit against this company's recorded judgment via the grounded
|
|
8
|
-
* review_business_match endpoint, and prints a comparative report.
|
|
10
|
+
* review_business_match endpoint, and prints a comparative report. The
|
|
11
|
+
* `--judgment` and `--map-tier` modes instead run both lanes fully grounded and
|
|
12
|
+
* vary only their one stage's tier, isolating that upgrade's fit delta.
|
|
9
13
|
*
|
|
10
14
|
* The report is the deliverable; nothing here is wired to model selection or
|
|
11
15
|
* routing. The live double-run is dogfood (it spawns real children + models), so
|
|
@@ -16,7 +20,7 @@ import type { ExtensionAPI, ExtensionCommandContext } from "@earendil-works/pi-c
|
|
|
16
20
|
import { type ComparisonReport, type Lane } from "./eval.js";
|
|
17
21
|
import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
18
22
|
/** Which two-lane comparison a /go-compare invocation runs. */
|
|
19
|
-
export type CompareMode = "grounding" | "judgment";
|
|
23
|
+
export type CompareMode = "grounding" | "judgment" | "map-tier";
|
|
20
24
|
export interface RegisterGoCompareDeps {
|
|
21
25
|
/** The whole eval runner, injectable so the handler is unit-tested without spawning. */
|
|
22
26
|
compare?: (ticket: string, ctx: ExtensionCommandContext, mode: CompareMode) => Promise<ComparisonReport>;
|
|
@@ -26,10 +30,15 @@ export interface RegisterGoCompareDeps {
|
|
|
26
30
|
git?: GitCommand;
|
|
27
31
|
}
|
|
28
32
|
/**
|
|
29
|
-
* Parse the leading `--judgment` flag. PURE.
|
|
33
|
+
* Parse the leading `--judgment` / `--map-tier` flag. PURE.
|
|
30
34
|
*
|
|
31
35
|
* Default stays `grounding` so the existing command is byte-identical for
|
|
32
|
-
* callers who pass only a ticket.
|
|
36
|
+
* callers who pass only a ticket. The two flags are mutually exclusive in
|
|
37
|
+
* practice (only one is ever passed); if both appear, `--judgment` wins, but
|
|
38
|
+
* BOTH regexes are stripped unconditionally regardless of which one decided
|
|
39
|
+
* the mode, so the losing flag can never leak into the ticket ref (e.g.
|
|
40
|
+
* `--judgment --map-tier YAG-1` must yield ticket `"YAG-1"`, not
|
|
41
|
+
* `"--map-tier YAG-1"`).
|
|
33
42
|
*/
|
|
34
43
|
export declare function parseCompareArgs(args: string): {
|
|
35
44
|
mode: CompareMode;
|
|
@@ -38,9 +47,10 @@ export declare function parseCompareArgs(args: string): {
|
|
|
38
47
|
/**
|
|
39
48
|
* The per-lane stage list.
|
|
40
49
|
*
|
|
41
|
-
* Grounding mode strips the grounding tools from the blind lane. Judgment
|
|
42
|
-
*
|
|
43
|
-
*
|
|
50
|
+
* Grounding mode strips the grounding tools from the blind lane. Judgment and
|
|
51
|
+
* map-tier modes run BOTH lanes fully grounded and vary only their one tier
|
|
52
|
+
* (plan/review for judgment, map for map-tier), so the delta isolates that one
|
|
53
|
+
* upgrade rather than confounding it with grounding.
|
|
44
54
|
*/
|
|
45
55
|
export declare function stagesForLane(lane: Lane): import("./types.js").PipelineStage[];
|
|
46
56
|
export type GitCommand = (args: string[], cwd: string | undefined, signal?: AbortSignal) => Promise<string>;
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only)
|
|
2
|
+
* Registers `/go-compare <ticket>` - the M6 grounded-vs-blind eval (report-only),
|
|
3
|
+
* plus two opt-in tier lanes: `--judgment` (peak vs advanced on plan+review) and
|
|
4
|
+
* `--map-tier` (standard vs efficient on map).
|
|
3
5
|
*
|
|
4
6
|
* It runs the same ticket through the pipeline TWICE in isolated git worktrees
|
|
5
7
|
* (so the user's tree is never touched): once grounded (the real /go) and once
|
|
6
8
|
* blind (grounding tools stripped + grounding-free personas), scores each change
|
|
7
9
|
* for business fit against this company's recorded judgment via the grounded
|
|
8
|
-
* review_business_match endpoint, and prints a comparative report.
|
|
10
|
+
* review_business_match endpoint, and prints a comparative report. The
|
|
11
|
+
* `--judgment` and `--map-tier` modes instead run both lanes fully grounded and
|
|
12
|
+
* vary only their one stage's tier, isolating that upgrade's fit delta.
|
|
9
13
|
*
|
|
10
14
|
* The report is the deliverable; nothing here is wired to model selection or
|
|
11
15
|
* routing. The live double-run is dogfood (it spawns real children + models), so
|
|
@@ -17,32 +21,38 @@ import { mkdtempSync, rmSync } from "node:fs";
|
|
|
17
21
|
import { tmpdir } from "node:os";
|
|
18
22
|
import { join } from "node:path";
|
|
19
23
|
import { getToken as defaultGetToken, resolveBaseUrl } from "../config.js";
|
|
20
|
-
import { blindStages, makeGroundedVsBlindEval, makeJudgmentTierEval, reportOnlyStages, withJudgmentTier, } from "./eval.js";
|
|
24
|
+
import { blindStages, makeGroundedVsBlindEval, makeJudgmentTierEval, makeMapTierEval, reportOnlyStages, withJudgmentTier, withMapTier, } from "./eval.js";
|
|
21
25
|
import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
22
26
|
/**
|
|
23
|
-
* Parse the leading `--judgment` flag. PURE.
|
|
27
|
+
* Parse the leading `--judgment` / `--map-tier` flag. PURE.
|
|
24
28
|
*
|
|
25
29
|
* Default stays `grounding` so the existing command is byte-identical for
|
|
26
|
-
* callers who pass only a ticket.
|
|
30
|
+
* callers who pass only a ticket. The two flags are mutually exclusive in
|
|
31
|
+
* practice (only one is ever passed); if both appear, `--judgment` wins, but
|
|
32
|
+
* BOTH regexes are stripped unconditionally regardless of which one decided
|
|
33
|
+
* the mode, so the losing flag can never leak into the ticket ref (e.g.
|
|
34
|
+
* `--judgment --map-tier YAG-1` must yield ticket `"YAG-1"`, not
|
|
35
|
+
* `"--map-tier YAG-1"`).
|
|
27
36
|
*/
|
|
28
37
|
export function parseCompareArgs(args) {
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
// Accept the flag anywhere leading, so `/go-compare --judgment YAG-1` and
|
|
38
|
+
const trimmed = args.trim();
|
|
39
|
+
// Accept either flag anywhere leading, so `/go-compare --judgment YAG-1` and
|
|
32
40
|
// `/go-compare YAG-1 --judgment` both work.
|
|
33
|
-
const
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
41
|
+
const judgmentFlag = /(^|\s)--judgment(\s|$)/;
|
|
42
|
+
const mapTierFlag = /(^|\s)--map-tier(\s|$)/;
|
|
43
|
+
const hasJudgment = judgmentFlag.test(trimmed);
|
|
44
|
+
const hasMapTier = mapTierFlag.test(trimmed);
|
|
45
|
+
const mode = hasJudgment ? "judgment" : hasMapTier ? "map-tier" : "grounding";
|
|
46
|
+
const ticket = trimmed.replace(judgmentFlag, " ").replace(mapTierFlag, " ").trim();
|
|
47
|
+
return { mode, ticket };
|
|
39
48
|
}
|
|
40
49
|
/**
|
|
41
50
|
* The per-lane stage list.
|
|
42
51
|
*
|
|
43
|
-
* Grounding mode strips the grounding tools from the blind lane. Judgment
|
|
44
|
-
*
|
|
45
|
-
*
|
|
52
|
+
* Grounding mode strips the grounding tools from the blind lane. Judgment and
|
|
53
|
+
* map-tier modes run BOTH lanes fully grounded and vary only their one tier
|
|
54
|
+
* (plan/review for judgment, map for map-tier), so the delta isolates that one
|
|
55
|
+
* upgrade rather than confounding it with grounding.
|
|
46
56
|
*/
|
|
47
57
|
export function stagesForLane(lane) {
|
|
48
58
|
switch (lane) {
|
|
@@ -52,6 +62,10 @@ export function stagesForLane(lane) {
|
|
|
52
62
|
return withJudgmentTier(reportOnlyStages(), "peak");
|
|
53
63
|
case "advanced_judgment":
|
|
54
64
|
return withJudgmentTier(reportOnlyStages(), "advanced");
|
|
65
|
+
case "map_standard":
|
|
66
|
+
return withMapTier(reportOnlyStages(), "standard");
|
|
67
|
+
case "map_efficient":
|
|
68
|
+
return withMapTier(reportOnlyStages(), "efficient");
|
|
55
69
|
default:
|
|
56
70
|
return reportOnlyStages();
|
|
57
71
|
}
|
|
@@ -94,8 +108,9 @@ function defaultCompare(deps) {
|
|
|
94
108
|
const result = await runPipeline(ticket, {
|
|
95
109
|
cwd: tree,
|
|
96
110
|
signal: ctx.signal,
|
|
97
|
-
// Only the blind lane runs ungrounded. Both judgment lanes
|
|
98
|
-
// grounded so the delta isolates the tier
|
|
111
|
+
// Only the blind lane runs ungrounded. Both judgment lanes and both
|
|
112
|
+
// map-tier lanes stay grounded so the delta isolates the one tier
|
|
113
|
+
// under test, not the grounding.
|
|
99
114
|
grounded: lane !== "blind",
|
|
100
115
|
stages: stagesForLane(lane),
|
|
101
116
|
childEnv: evalEnv,
|
|
@@ -144,14 +159,16 @@ function defaultCompare(deps) {
|
|
|
144
159
|
};
|
|
145
160
|
const evalRunner = mode === "judgment"
|
|
146
161
|
? makeJudgmentTierEval({ runLane, scoreFit })
|
|
147
|
-
:
|
|
162
|
+
: mode === "map-tier"
|
|
163
|
+
? makeMapTierEval({ runLane, scoreFit })
|
|
164
|
+
: makeGroundedVsBlindEval({ runLane, scoreFit });
|
|
148
165
|
return evalRunner(ticket);
|
|
149
166
|
};
|
|
150
167
|
}
|
|
151
168
|
export function registerGoCompareCommand(pi, deps = {}) {
|
|
152
169
|
const compare = deps.compare ?? defaultCompare(deps);
|
|
153
170
|
pi.registerCommand("go-compare", {
|
|
154
|
-
description: "Eval only: run a ticket through two lane configurations in isolated worktrees and report the business-fit delta. Default compares grounded vs grounding-suppressed; --judgment compares peak vs advanced on plan+review. Never affects routing.",
|
|
171
|
+
description: "Eval only: run a ticket through two lane configurations in isolated worktrees and report the business-fit delta. Default compares grounded vs grounding-suppressed; --judgment compares peak vs advanced on plan+review; --map-tier compares standard vs efficient on map. Never affects routing.",
|
|
155
172
|
handler: async (args, ctx) => {
|
|
156
173
|
const notify = (message, type) => {
|
|
157
174
|
if (ctx.hasUI)
|
|
@@ -159,7 +176,7 @@ export function registerGoCompareCommand(pi, deps = {}) {
|
|
|
159
176
|
};
|
|
160
177
|
const { mode, ticket } = parseCompareArgs(args);
|
|
161
178
|
if (!ticket) {
|
|
162
|
-
notify("Usage: /go-compare [--judgment] <ticket>", "warning");
|
|
179
|
+
notify("Usage: /go-compare [--judgment | --map-tier] <ticket>", "warning");
|
|
163
180
|
return;
|
|
164
181
|
}
|
|
165
182
|
if (!ctx.isIdle()) {
|
|
@@ -168,7 +185,9 @@ export function registerGoCompareCommand(pi, deps = {}) {
|
|
|
168
185
|
}
|
|
169
186
|
notify(mode === "judgment"
|
|
170
187
|
? "Running the peak vs advanced judgment eval (two isolated passes). This takes a while."
|
|
171
|
-
:
|
|
188
|
+
: mode === "map-tier"
|
|
189
|
+
? "Running the standard vs efficient map eval (two isolated passes). This takes a while."
|
|
190
|
+
: "Running the grounded vs blind eval (two isolated passes). This takes a while.", "info");
|
|
172
191
|
try {
|
|
173
192
|
const out = await compare(ticket, ctx, mode);
|
|
174
193
|
await pi.sendUserMessage(out.report);
|