@cspeach/cli 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/loop.js +22 -9
- package/dist/approvals/op-labels.js +124 -0
- package/dist/approvals/render.js +42 -36
- package/dist/cli.js +15 -0
- package/dist/commands/compact.js +28 -2
- package/dist/commands/config-set.js +189 -0
- package/dist/commands/config-show.js +20 -0
- package/dist/commands/export-audit.js +43 -0
- package/dist/commands/help.js +5 -0
- package/dist/commands/plan-audit-evidence.js +266 -0
- package/dist/commands/plan-audit.js +692 -0
- package/dist/commands/plan-chain.js +671 -0
- package/dist/commands/plan-continue.js +179 -0
- package/dist/commands/plan-gate.js +154 -0
- package/dist/commands/plan-resume.js +588 -33
- package/dist/config/loader.js +128 -4
- package/dist/config/model-defaults.js +14 -0
- package/dist/cost/pricing.js +27 -1
- package/dist/doctor/checks/system-roles.js +41 -0
- package/dist/doctor/run.js +2 -0
- package/dist/models/resolve.js +61 -0
- package/dist/models/server-config.js +155 -0
- package/dist/one-shot.js +25 -3
- package/dist/projects/extract-cca.js +3 -1
- package/dist/projects/extract-modernize.js +3 -1
- package/dist/projects/extract-plan.js +60 -6
- package/dist/projects/extract-test-coverage.js +3 -1
- package/dist/projects/extract-upgrade.js +3 -1
- package/dist/projects/handover-md.js +195 -0
- package/dist/projects/index.js +1 -1
- package/dist/projects/plan-run.js +137 -13
- package/dist/projects/plan-schema.js +73 -0
- package/dist/projects/run-lease.js +157 -0
- package/dist/projects/save-command.js +26 -15
- package/dist/renderer/status-footer.js +22 -12
- package/dist/renderer/thinking-heartbeat.js +64 -8
- package/dist/renderer/todo-block.js +51 -0
- package/dist/renderer/tool-widget.js +37 -0
- package/dist/repl/bracketed-paste.js +28 -19
- package/dist/repl/builtin-commands.js +5 -0
- package/dist/repl/current-transport.js +10 -0
- package/dist/repl/history.js +86 -0
- package/dist/repl/ink-stdin-guard.js +64 -0
- package/dist/repl/mode-ceiling.js +16 -0
- package/dist/repl/mode-cycle.js +104 -0
- package/dist/repl/post-turn-status.js +24 -4
- package/dist/repl/slash-completer.js +5 -0
- package/dist/repl.js +954 -83
- package/dist/rewind/candidates.js +194 -0
- package/dist/rewind/cli.js +137 -0
- package/dist/rewind/format.js +27 -0
- package/dist/rewind/restore.js +245 -0
- package/dist/session/audit-export.js +459 -0
- package/dist/session/context-report.js +163 -0
- package/dist/session/recap.js +160 -0
- package/dist/skill-catalog.js +9 -3
- package/dist/skills/bundled-skills.js +59 -66
- package/dist/tools/approval.js +115 -7
- package/dist/tools/ask-question.js +304 -3
- package/dist/tools/extend-model/anchored-insert.js +604 -0
- package/dist/tools/extend-model/tool.js +162 -10
- package/dist/tools/fiori/fe-extend.js +76 -0
- package/dist/tools/fiori/fe-scaffold.js +29 -3
- package/dist/tools/fiori/floorplan-map.js +19 -0
- package/dist/tools/fiori/samples/data/index.json +13602 -0
- package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
- package/dist/tools/fiori/samples/loader.js +248 -0
- package/dist/tools/fiori/samples/search.js +63 -0
- package/dist/tools/fiori/samples/types.js +2 -0
- package/dist/tools/fiori/smoke/assertions.js +74 -0
- package/dist/tools/fiori/smoke/browser.js +52 -0
- package/dist/tools/fiori/smoke/driver.js +89 -0
- package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
- package/dist/tools/fiori/smoke/run-smoke.js +149 -0
- package/dist/tools/fiori/tools.js +328 -3
- package/dist/tools/local-build.js +11 -1
- package/dist/tools/sap-read.js +79 -11
- package/dist/tools/sap-write.js +24 -4
- package/dist/tools/snapshot.js +27 -1
- package/dist/tools/subagent/agent_run.js +27 -3
- package/dist/tools/todo.js +144 -0
- package/dist/ui/app.js +372 -19
- package/dist/ui/approval-modal.js +49 -16
- package/dist/ui/ask-question-emitter.js +14 -0
- package/dist/ui/context-grid.js +108 -0
- package/dist/ui/footer.js +109 -30
- package/dist/ui/header.js +7 -0
- package/dist/ui/line-resolution.js +18 -2
- package/dist/ui/rewind-emitter.js +10 -0
- package/dist/ui/rewind-panel.js +81 -0
- package/dist/ui/sap-state-store.js +1 -0
- package/dist/ui/status-line.js +43 -0
- package/dist/ui/text-input.js +72 -8
- package/dist/ui/todo-emitter.js +25 -0
- package/dist/ui/todo-panel.js +64 -0
- package/dist/ui/turn-status-emitter.js +50 -4
- package/dist/ui/turn-status.js +18 -3
- package/dist/ui/widgets/ask-form.js +242 -0
- package/dist/ui/widgets/ask-question-modal.js +17 -7
- package/package.json +4 -1
|
@@ -0,0 +1,671 @@
|
|
|
1
|
+
// cspeach-cli/src/commands/plan-chain.ts
|
|
2
|
+
//
|
|
3
|
+
// Task 6 (agentic-flow, 2026-07-03) — the shared post-phase sequence, called
|
|
4
|
+
// by BOTH repl.tsx plan-resume call sites (Ink + classic) so neither
|
|
5
|
+
// duplicates the block. Exact order (the crash-safety design):
|
|
6
|
+
//
|
|
7
|
+
// finishPlanResume persisted the phase WITH audit:'pending' (caller)
|
|
8
|
+
// → runPhaseAudit (fresh-context, tool-less judge turn)
|
|
9
|
+
// → applyAuditResult (verdict persisted as revision N+1)
|
|
10
|
+
// → verdict handling (failed ⇒ re-run/waive/block; infra ⇒ stop,
|
|
11
|
+
// 2nd consecutive infra ⇒ retry/waive/block)
|
|
12
|
+
// → decideNextPhaseGate (step/guarded transition decision)
|
|
13
|
+
// → offerNextPhaseAutoRun (prompt, or auto-queue in guarded mode)
|
|
14
|
+
//
|
|
15
|
+
// A crash anywhere before applyAuditResult leaves 'pending' on disk;
|
|
16
|
+
// preparePlanResume's re-audit-on-resume then runs this chain FIRST on the
|
|
17
|
+
// next --resume (executedPhaseId = the pending phase, forceAudit) instead of
|
|
18
|
+
// executing a new phase — the audit is never skipped.
|
|
19
|
+
//
|
|
20
|
+
// Everything I/O-heavy or interactive is injectable via PlanChainDeps so the
|
|
21
|
+
// chain is unit-testable without a provider, Ink, or inquirer.
|
|
22
|
+
import { basename } from 'node:path';
|
|
23
|
+
import chalk from 'chalk';
|
|
24
|
+
import { readProjectFile } from '../projects/status.js';
|
|
25
|
+
import { parsePlanContent } from '../projects/plan-schema.js';
|
|
26
|
+
import { statusesFromItems, computeNextPhase, isPhaseSatisfied, isAuditResolved, planCompletionLines, renderPlanTracker, } from '../projects/plan-run.js';
|
|
27
|
+
import { decideNextPhaseGate, getAndResetPlanPhaseSubagentDispatches, getAndResetPlanDeviation, } from './plan-gate.js';
|
|
28
|
+
import { runPhaseAudit } from './plan-audit.js';
|
|
29
|
+
import { applyAuditResult, applyPhaseResolution, buildResumeCommand, offerNextPhaseAutoRun, } from './plan-resume.js';
|
|
30
|
+
import { appendCostLine, buildEntry } from '../cost/cost-log.js';
|
|
31
|
+
import { startThinkingHeartbeat, AUDIT_HEARTBEAT_THRESHOLDS } from '../renderer/thinking-heartbeat.js';
|
|
32
|
+
import { computeCost, formatCost } from '../cost/pricing.js';
|
|
33
|
+
/** Two consecutive failed verdicts on the same phase force status 'blocked'. */
|
|
34
|
+
export const MAX_CONSECUTIVE_AUDIT_FAILURES = 2;
|
|
35
|
+
/** Infra error strings are model/HTTP noise — clamp when rendering (Task 5 review). */
|
|
36
|
+
const INFRA_RENDER_CHARS = 200;
|
|
37
|
+
/**
|
|
38
|
+
* Run the post-phase chain. Returns the LATEST saved envelope path (the
|
|
39
|
+
* caller's savedPath when nothing was written), or null when the chain could
|
|
40
|
+
* not even load the envelope.
|
|
41
|
+
*/
|
|
42
|
+
export async function runPostPhaseChain(args) {
|
|
43
|
+
let path = args.savedPath;
|
|
44
|
+
let loaded = loadPlan(path, args.log);
|
|
45
|
+
if (!loaded)
|
|
46
|
+
return null;
|
|
47
|
+
let phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
|
|
48
|
+
if (!phase) {
|
|
49
|
+
args.log(chalk.red(`[plan] phase '${args.executedPhaseId}' not found in ${basename(path)} — chain skipped.`));
|
|
50
|
+
return path;
|
|
51
|
+
}
|
|
52
|
+
// Per-run flags survive the chain: an explicitly flagged run keeps its mode
|
|
53
|
+
// AND its audit setting across every harness-built resume command (I2 — the
|
|
54
|
+
// audit flag mirrors modeFlag; --audit maps to itself, 'off' to --no-audit).
|
|
55
|
+
const flagSuffix = (args.modeFlag ? ` --${args.modeFlag}` : '')
|
|
56
|
+
+ (args.auditFlag === 'off' ? ' --no-audit' : args.auditFlag === 'on' ? ' --audit' : '');
|
|
57
|
+
const resumeCmd = (p) => buildResumeCommand(p) + flagSuffix;
|
|
58
|
+
let chainWrote = false;
|
|
59
|
+
// Task 7 — phase-end subagent report. agent_run counted every dispatch
|
|
60
|
+
// made while the plan-phase flag was on (all forced read-only). The REPL
|
|
61
|
+
// reads+resets the counter in its runTurn finally and threads the value
|
|
62
|
+
// here (fix-wave, item 1) — this chain can be skipped on error paths, so
|
|
63
|
+
// consuming the counter here alone would let counts leak into the next
|
|
64
|
+
// phase's report line. The read+reset below is only the fallback for
|
|
65
|
+
// callers that ran no phase turn (re-audit passes, direct test calls) —
|
|
66
|
+
// zero there in practice, so nothing is printed.
|
|
67
|
+
const subagentDispatches = args.subagentDispatches ?? getAndResetPlanPhaseSubagentDispatches();
|
|
68
|
+
if (subagentDispatches > 0) {
|
|
69
|
+
args.log(chalk.dim(`[phase] ${args.executedPhaseId}: ${subagentDispatches} subagent dispatch(es) — read-only enforced`));
|
|
70
|
+
}
|
|
71
|
+
/* ── 0. deviation backstop (Task 10) ──────────────────────────────────────
|
|
72
|
+
* approval.ts raised the flag mid-turn: this phase declared writes:false
|
|
73
|
+
* but requested a write approval anyway — the write was DENIED without
|
|
74
|
+
* prompting; here the chain records why and stops. Design choice (per the
|
|
75
|
+
* task's simpler-path option): SKIP the audit — the phase didn't complete
|
|
76
|
+
* honestly and was deliberately stopped, so auditing it wastes a turn. The
|
|
77
|
+
* finishPlanResume-injected 'pending' audit is dropped in the same write
|
|
78
|
+
* (clearAudit) or the next resume would re-audit the blocked phase. Same
|
|
79
|
+
* read+reset pattern as the subagent counter — the flag never leaks into
|
|
80
|
+
* the next chain leg. Guarded-only by construction: approval.ts only ever
|
|
81
|
+
* raises it while isGuardedRunActive(). */
|
|
82
|
+
const deviationDetail = getAndResetPlanDeviation();
|
|
83
|
+
if (deviationDetail) {
|
|
84
|
+
path = await applyPhaseResolution(path, phase.id, {
|
|
85
|
+
status: 'blocked',
|
|
86
|
+
lastStop: { phaseId: phase.id, reason: 'deviation', detail: deviationDetail, at: new Date().toISOString() },
|
|
87
|
+
summary: `guarded chain stopped: plan deviation in ${phase.id} — ${deviationDetail}`,
|
|
88
|
+
clearAudit: true,
|
|
89
|
+
log: args.log,
|
|
90
|
+
});
|
|
91
|
+
args.log(chalk.red(`✖ plan deviation in phase ${phase.id} — ${deviationDetail}`), stoppedBanner(`plan deviation — phase ${phase.id} blocked (undeclared write denied)`, resumeCmd(path)));
|
|
92
|
+
ringBell();
|
|
93
|
+
return path;
|
|
94
|
+
}
|
|
95
|
+
/* ── 1. audit (only when unresolved-pending/infra, or forced) ─────────── */
|
|
96
|
+
// Off-switch: audits off skips the NORMAL post-phase audit (no injected
|
|
97
|
+
// pending exists to catch anyway). A forced re-audit still runs — a persisted
|
|
98
|
+
// pending/infra record from an earlier ON run must resolve regardless.
|
|
99
|
+
const auditState = phase.audit?.state;
|
|
100
|
+
const auditsOff = args.auditMode === 'off';
|
|
101
|
+
// audit-infra-continue (2026-07-09): 'infra_failed' is now RESOLVED and no
|
|
102
|
+
// longer auto-re-audited — an audit that could not run already let the chain
|
|
103
|
+
// continue, so it must not spontaneously re-run on a later pass. Only a
|
|
104
|
+
// persisted 'pending' (crash between phase-save and verdict-save) or an
|
|
105
|
+
// explicit forceAudit (hand-edit re-verification) triggers an audit here.
|
|
106
|
+
const needsAudit = args.forceAudit
|
|
107
|
+
|| (!auditsOff && auditState === 'pending');
|
|
108
|
+
if (needsAudit) {
|
|
109
|
+
args.log('', chalk.dim(`[audit] independent audit of phase ${phase.id} — fresh context, judged against the tool-call evidence…`));
|
|
110
|
+
// Evidence lives with the session that EXECUTED the phase; on a re-audit
|
|
111
|
+
// in a fresh CLI session that id was recorded on the pending audit.
|
|
112
|
+
const evidenceSessionId = phase.audit?.sessionId ?? args.sessionId;
|
|
113
|
+
const auditFn = args.deps?.runPhaseAuditFn ?? runPhaseAudit;
|
|
114
|
+
let cost;
|
|
115
|
+
// D-B — window the evidence to THIS attempt only. Live post-phase audit:
|
|
116
|
+
// the repl threads the turn-start it captured (attemptStartedAt).
|
|
117
|
+
// RE-audit (review item 1): the OLD session's JSONL can hold MULTIPLE
|
|
118
|
+
// attempts (fail → rerun in the same process → die before verdict →
|
|
119
|
+
// resume re-audits), so an unwindowed re-audit re-creates the exact
|
|
120
|
+
// cross-attempt false-ordering on the standard recovery path — use the
|
|
121
|
+
// window finishPlanResume persisted on the audit record
|
|
122
|
+
// (audit.attemptStartedAt, carried through verdict writes). A legacy
|
|
123
|
+
// record without the field runs unwindowed (fail-open, no window claim
|
|
124
|
+
// is made to the auditor either — see buildAuditPrompt).
|
|
125
|
+
const sinceIso = args.attemptStartedAt ?? phase.audit?.attemptStartedAt;
|
|
126
|
+
// D-F item 4 — liveness. The audit is a single fresh-context model turn
|
|
127
|
+
// (10-40s) with no streaming into this chain, so the [audit] header sat
|
|
128
|
+
// silent. Reuse the house-style heartbeat (same primitive loop.ts uses),
|
|
129
|
+
// routed through args.log so it works under both Ink and classic.
|
|
130
|
+
// audit-timeout (2026-07-06): use the audit-specific schedule — the audit
|
|
131
|
+
// is a PROXY call (api.cspeach.dev), so the generic "check the VPN" 5m copy
|
|
132
|
+
// is the wrong diagnosis. runPhaseAudit now hard-caps each attempt at 120s,
|
|
133
|
+
// so this can never sit stuck forever. Attempt transitions are surfaced via
|
|
134
|
+
// onRetry below. Always stopped in finally.
|
|
135
|
+
// audit-lean FIX #2 — a LIVE token count threaded from the auditor's stream
|
|
136
|
+
// into the heartbeat suffix, so a stall (`· 0 tokens`) reads differently
|
|
137
|
+
// from slow-but-working progress. The suffix closure reads the latest value
|
|
138
|
+
// at each heartbeat tick; onProgress updates it as chunks arrive.
|
|
139
|
+
let auditTokens = 0;
|
|
140
|
+
const heartbeat = startThinkingHeartbeat({
|
|
141
|
+
thresholds: AUDIT_HEARTBEAT_THRESHOLDS,
|
|
142
|
+
emit: (line) => args.log(line),
|
|
143
|
+
suffix: () => ` · ${auditTokens.toLocaleString()} token${auditTokens === 1 ? '' : 's'}`,
|
|
144
|
+
});
|
|
145
|
+
let verdict;
|
|
146
|
+
try {
|
|
147
|
+
verdict = await auditFn({
|
|
148
|
+
envelope: loaded.envelope,
|
|
149
|
+
phaseId: phase.id,
|
|
150
|
+
sessionId: evidenceSessionId,
|
|
151
|
+
cwd: args.cwd,
|
|
152
|
+
...(sinceIso ? { sinceIso } : {}),
|
|
153
|
+
onCost: (c) => { cost = c; },
|
|
154
|
+
// Best-effort: the heartbeat reads auditTokens at each tick.
|
|
155
|
+
onProgress: (n) => { auditTokens = n; },
|
|
156
|
+
// audit-timeout — an honest attempt-transition line when a prior
|
|
157
|
+
// attempt hit its deadline (or otherwise failed) and we retry in a
|
|
158
|
+
// fresh session. Keeps the timeline truthful against the 120s×3 ceiling.
|
|
159
|
+
onRetry: (r) => {
|
|
160
|
+
const why = r.reason === 'timeout' ? 'previous attempt timed out'
|
|
161
|
+
: r.reason === 'unparseable' ? 'previous attempt returned no verdict'
|
|
162
|
+
: 'previous attempt errored';
|
|
163
|
+
args.log(chalk.dim(` [audit] attempt ${r.attempt} of ${r.totalAttempts} (${why})`));
|
|
164
|
+
},
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
finally {
|
|
168
|
+
heartbeat.stop();
|
|
169
|
+
}
|
|
170
|
+
if (cost)
|
|
171
|
+
renderAuditCost(args, cost);
|
|
172
|
+
// Fix-wave (item 10a) — a persisted 'failed' reaching a re-audit means
|
|
173
|
+
// the session died AT the failure menu (verdict written, resolution
|
|
174
|
+
// never was). A failed verdict then RE-CONFIRMS the same failure: hold
|
|
175
|
+
// the counter instead of incrementing, and re-render the menu — the
|
|
176
|
+
// developer must still get the re-run/waive/block choice they lost.
|
|
177
|
+
// Delta-review edge: an interleaved infra outage launders the same case
|
|
178
|
+
// — failed(1) → menu death → re-audit hits infra (the counter is carried
|
|
179
|
+
// onto infra_failed) → the NEXT re-audit's failed verdict would count as
|
|
180
|
+
// consecutive and silently force-block. A persisted infra_failed that
|
|
181
|
+
// CARRIES a failure counter is therefore also a reconfirmation: the
|
|
182
|
+
// menu was owed and never shown.
|
|
183
|
+
const reconfirmingFailed = auditState === 'failed'
|
|
184
|
+
|| (auditState === 'infra_failed' && (phase.audit?.consecutiveFailures ?? 0) >= 1);
|
|
185
|
+
// Consecutive-failure decision happens BEFORE the write so the forced
|
|
186
|
+
// block lands in the SAME revision as the verdict.
|
|
187
|
+
const priorFailures = phase.audit?.consecutiveFailures ?? 0;
|
|
188
|
+
const countedFailures = reconfirmingFailed
|
|
189
|
+
? Math.max(priorFailures, 1)
|
|
190
|
+
: priorFailures + 1;
|
|
191
|
+
const forceBlock = verdict.state === 'failed'
|
|
192
|
+
&& countedFailures >= MAX_CONSECUTIVE_AUDIT_FAILURES;
|
|
193
|
+
const nowIso = new Date().toISOString();
|
|
194
|
+
// Only a real 'failed' verdict records a stop. 'infra_failed' no longer
|
|
195
|
+
// stops (audit-infra-continue) — like warn/pass it leaves no lastStop, so
|
|
196
|
+
// applyAuditResult clears any stale one (see its resolved-verdict branch).
|
|
197
|
+
const lastStop = verdict.state === 'failed'
|
|
198
|
+
? { phaseId: phase.id, reason: 'audit-failed', detail: clampInfra(verdict.findings[0] ?? ''), at: nowIso }
|
|
199
|
+
: undefined;
|
|
200
|
+
// Confidence-tiered redesign (2026-07-08) — the GATE. A `warn` verdict is
|
|
201
|
+
// NON-BLOCKING: it persists LITERALLY (state:'warn', findings carried),
|
|
202
|
+
// clears the failure counter (a warn is not a failure), and the chain
|
|
203
|
+
// continues exactly like a pass — no menu, no lastStop, forceBlock can
|
|
204
|
+
// never fire on it (forceBlock keys off verdict.state === 'failed' above).
|
|
205
|
+
// fail / infra_failed / pass persistence is byte-unchanged.
|
|
206
|
+
path = await applyAuditResult(path, phase.id, {
|
|
207
|
+
state: verdict.state,
|
|
208
|
+
findings: verdict.findings,
|
|
209
|
+
}, {
|
|
210
|
+
lastStop,
|
|
211
|
+
...(forceBlock ? { statusOverride: 'blocked' } : {}),
|
|
212
|
+
...(reconfirmingFailed ? { reconfirmedFailure: true } : {}),
|
|
213
|
+
log: args.log,
|
|
214
|
+
});
|
|
215
|
+
chainWrote = true;
|
|
216
|
+
loaded = loadPlan(path, args.log);
|
|
217
|
+
if (!loaded)
|
|
218
|
+
return path;
|
|
219
|
+
phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
|
|
220
|
+
// D-E — the progress board renders HERE, AFTER the verdict lands, so the
|
|
221
|
+
// tally is audit-aware and never contradicts a later verdict. (finishPlanResume
|
|
222
|
+
// now prints only a neutral "audit pending…" line; the board it used to
|
|
223
|
+
// print counted the just-run phase as done because it erased the pending
|
|
224
|
+
// audit for display — a validated phase with a pending audit is NOT
|
|
225
|
+
// satisfied, so the pre-audit board read "2/3" for what was really "1/3".)
|
|
226
|
+
// Suppressed on mid-pass re-audit legs (runReauditPass renders once at the
|
|
227
|
+
// end via its final, unsuppressed leg).
|
|
228
|
+
if (!args.suppressOffer) {
|
|
229
|
+
const boardStatuses = statusesFromItems(loaded.envelope.interaction.items, loaded.content.phases);
|
|
230
|
+
args.log('', renderPlanTracker({
|
|
231
|
+
title: loaded.envelope.title,
|
|
232
|
+
version: loaded.envelope.version,
|
|
233
|
+
content: loaded.content,
|
|
234
|
+
statuses: boardStatuses,
|
|
235
|
+
currentId: null,
|
|
236
|
+
auditsOff,
|
|
237
|
+
}), '');
|
|
238
|
+
}
|
|
239
|
+
if (verdict.state === 'infra_failed') {
|
|
240
|
+
// audit-infra-continue (2026-07-09): the audit COULD NOT RUN — after the
|
|
241
|
+
// in-turn 3 attempts (runPhaseAudit's 120s×3 ceiling) the model call
|
|
242
|
+
// returned no output. That is an audit-INFRASTRUCTURE failure, not a
|
|
243
|
+
// finding against the phase, whose own exit gate (activation + ATC)
|
|
244
|
+
// already passed. So we do NOT hard-stop verified work: exactly like a
|
|
245
|
+
// warn/pass, render a LOUD, DISTINCT "could not run / NOT verified" note,
|
|
246
|
+
// keep the infra_failed state recorded (tracker keeps ⚠ audit not run;
|
|
247
|
+
// the final summary lists it as unverified) and CONTINUE to the gate +
|
|
248
|
+
// offer below. isAuditResolved treats infra_failed as resolved, so
|
|
249
|
+
// dependants unlock and the failure menu is never shown.
|
|
250
|
+
renderAuditNotRun(args.log, phase.id, clampInfra(verdict.findings[0] ?? 'auditor infrastructure failure'), resumeCmd(path));
|
|
251
|
+
}
|
|
252
|
+
else if (verdict.state === 'failed') {
|
|
253
|
+
renderFindings(args.log, phase.id, verdict.findings);
|
|
254
|
+
if (forceBlock) {
|
|
255
|
+
args.log(chalk.red(`✖ phase ${phase.id} failed its audit ${countedFailures} times in a row — marked blocked.`), stoppedBanner('two consecutive failed audits — phase blocked', resumeCmd(path)));
|
|
256
|
+
ringBell();
|
|
257
|
+
return path;
|
|
258
|
+
}
|
|
259
|
+
const select = args.deps?.selectFailureAction ?? defaultSelectFailureAction;
|
|
260
|
+
const action = await select({ phaseId: phase.id, findings: verdict.findings });
|
|
261
|
+
if (action === 'none') {
|
|
262
|
+
// D-C — fail CLOSED. 'none' is headless OR an interactive decline/cancel
|
|
263
|
+
// (Esc, no listener, or an unrecognised menu answer). The phase must NOT
|
|
264
|
+
// auto-re-run — a re-run is an explicit menu choice, never a default.
|
|
265
|
+
// Stop with the banner; resuming re-audits and re-offers the menu (the
|
|
266
|
+
// persisted 'failed' on a still-satisfied phase is re-caught by
|
|
267
|
+
// preparePlanResume's died-at-menu scan).
|
|
268
|
+
args.log(stoppedBanner(`audit failed for phase ${phase.id} — resolve interactively (re-run / waive / block)`, resumeCmd(path)));
|
|
269
|
+
ringBell();
|
|
270
|
+
return path;
|
|
271
|
+
}
|
|
272
|
+
if (action === 'blocked') {
|
|
273
|
+
path = await applyPhaseResolution(path, phase.id, {
|
|
274
|
+
status: 'blocked',
|
|
275
|
+
lastStop: { phaseId: phase.id, reason: 'audit-failed', detail: 'user marked the phase blocked after a failed audit', at: new Date().toISOString() },
|
|
276
|
+
summary: `audit ${phase.id}: failed — user marked phase blocked`,
|
|
277
|
+
log: args.log,
|
|
278
|
+
});
|
|
279
|
+
args.log(stoppedBanner(`phase ${phase.id} marked blocked after failed audit`, resumeCmd(path)));
|
|
280
|
+
ringBell();
|
|
281
|
+
return path;
|
|
282
|
+
}
|
|
283
|
+
if (action === 'rerun') {
|
|
284
|
+
// Status back to 'todo' — computeNextPhase re-picks it, and the flow
|
|
285
|
+
// ALWAYS audits an executed phase, so a re-run is re-audited by
|
|
286
|
+
// construction. The failed audit + counter stay on the phase until
|
|
287
|
+
// the re-run's finishPlanResume replaces them with 'pending'.
|
|
288
|
+
path = await applyPhaseResolution(path, phase.id, {
|
|
289
|
+
status: 'todo',
|
|
290
|
+
summary: `audit ${phase.id}: failed — user chose re-run (status → todo)`,
|
|
291
|
+
log: args.log,
|
|
292
|
+
});
|
|
293
|
+
args.log(`↩ re-running phase ${phase.id} — it will be re-audited after execution.`);
|
|
294
|
+
args.queueDispatch(resumeCmd(path));
|
|
295
|
+
return path;
|
|
296
|
+
}
|
|
297
|
+
// action === 'waive'
|
|
298
|
+
const reasonRaw = (await args.prompt(`Waive reason for ${phase.id} (one line): `)).trim();
|
|
299
|
+
const reason = reasonRaw.length > 0 ? reasonRaw : 'waived by user (no reason given)';
|
|
300
|
+
// §10 Q2 — a light 2-choice category alongside the reason: was the audit
|
|
301
|
+
// WRONG (false-positive) or is the finding real-but-accepted
|
|
302
|
+
// (accepted-risk)? Persisted so the audit's FP rate is measurable.
|
|
303
|
+
// Headless / cancel → accepted-risk (never a false-positive unasserted).
|
|
304
|
+
const category = await (args.deps?.selectWaiveCategory ?? defaultSelectWaiveCategory)({ phaseId: phase.id });
|
|
305
|
+
path = await applyAuditResult(path, phase.id, { state: 'waived', waivedReason: reason, waiveCategory: category }, { log: args.log });
|
|
306
|
+
loaded = loadPlan(path, args.log);
|
|
307
|
+
if (!loaded)
|
|
308
|
+
return path;
|
|
309
|
+
phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
|
|
310
|
+
args.log(chalk.yellow(`✎ audit waived (${category}) for ${phase.id} — reason recorded in the envelope.`));
|
|
311
|
+
}
|
|
312
|
+
else if (verdict.state === 'warn') {
|
|
313
|
+
// Non-blocking tier: ALWAYS shown, concise (Open-Q decision — warns are
|
|
314
|
+
// never swallowed). The chain continues to the gate + offer below just
|
|
315
|
+
// like a pass; the notes are already recorded on phase.audit.findings.
|
|
316
|
+
renderWarnings(args.log, phase.id, verdict.findings);
|
|
317
|
+
}
|
|
318
|
+
else {
|
|
319
|
+
args.log(chalk.green(`✔ audit passed — ${phase.id} verified against the evidence.`));
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
/* ── 2. gate + offer ──────────────────────────────────────────────────── */
|
|
323
|
+
// Review I-2 — mid-pass re-audits stop after the verdict: the pass owner
|
|
324
|
+
// (runReauditPass) issues one offer at the end.
|
|
325
|
+
if (args.suppressOffer)
|
|
326
|
+
return path;
|
|
327
|
+
const statuses = statusesFromItems(loaded.envelope.interaction.items, loaded.content.phases);
|
|
328
|
+
const next = computeNextPhase(loaded.content.phases, statuses);
|
|
329
|
+
if (!next) {
|
|
330
|
+
// Terminal messages only when this chain changed something — otherwise
|
|
331
|
+
// finishPlanResume already printed the same conclusion (no duplicates).
|
|
332
|
+
if (chainWrote) {
|
|
333
|
+
if (loaded.content.phases.every((p) => isPhaseSatisfied(statuses[p.id], p.audit))) {
|
|
334
|
+
args.log(...planCompletionLines(loaded.content.phases));
|
|
335
|
+
}
|
|
336
|
+
else {
|
|
337
|
+
const blocked = loaded.content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
|
|
338
|
+
const unresolved = loaded.content.phases
|
|
339
|
+
.filter((p) => isPhaseSatisfied(statuses[p.id]) && !isAuditResolved(p.audit))
|
|
340
|
+
.map((p) => `${p.id} (audit ${p.audit?.state ?? 'unknown'})`);
|
|
341
|
+
const unresolvedNote = unresolved.length > 0 ? ` Unresolved audits: ${unresolved.join(', ')}.` : '';
|
|
342
|
+
args.log(`No eligible next phase. Blocked: ${blocked.join(', ') || '(none)'}.${unresolvedNote}`);
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
return path;
|
|
346
|
+
}
|
|
347
|
+
const decision = decideNextPhaseGate({
|
|
348
|
+
mode: args.mode,
|
|
349
|
+
nextPhase: { id: next.id, writes: next.writes },
|
|
350
|
+
prevPhaseAudit: phase?.audit,
|
|
351
|
+
});
|
|
352
|
+
if (decision.action === 'stop') {
|
|
353
|
+
path = await applyPhaseResolution(path, args.executedPhaseId, {
|
|
354
|
+
lastStop: {
|
|
355
|
+
phaseId: args.executedPhaseId,
|
|
356
|
+
reason: decision.stopReason === 'audit-infra' ? 'audit-infra' : 'audit-failed',
|
|
357
|
+
at: new Date().toISOString(),
|
|
358
|
+
},
|
|
359
|
+
summary: `guarded chain stopped: ${decision.stopReason}`,
|
|
360
|
+
log: args.log,
|
|
361
|
+
});
|
|
362
|
+
args.log(substitutePlanToken(decision.banner ?? stoppedBanner('guarded chain stopped', resumeCmd(path)), path));
|
|
363
|
+
ringBell();
|
|
364
|
+
return path;
|
|
365
|
+
}
|
|
366
|
+
if (decision.action === 'prompt' && decision.stopReason === 'write-phase') {
|
|
367
|
+
// The prompt IS the stop (guarded semantics) — record it crash-safe
|
|
368
|
+
// BEFORE blocking on the user, then render the banner + bell.
|
|
369
|
+
path = await applyPhaseResolution(path, args.executedPhaseId, {
|
|
370
|
+
lastStop: { phaseId: next.id, reason: 'write-phase', at: new Date().toISOString() },
|
|
371
|
+
summary: `guarded chain stopped: write-phase consent for ${next.id}`,
|
|
372
|
+
log: args.log,
|
|
373
|
+
});
|
|
374
|
+
args.log(substitutePlanToken(decision.banner ?? '', path));
|
|
375
|
+
ringBell();
|
|
376
|
+
}
|
|
377
|
+
await offerNextPhaseAutoRun({
|
|
378
|
+
outcome: {
|
|
379
|
+
savedPath: path,
|
|
380
|
+
resumeCommand: resumeCmd(path),
|
|
381
|
+
nextPhaseId: next.id,
|
|
382
|
+
nextDelegateTo: next.delegateTo,
|
|
383
|
+
},
|
|
384
|
+
prompt: args.prompt,
|
|
385
|
+
queueDispatch: args.queueDispatch,
|
|
386
|
+
log: args.log,
|
|
387
|
+
gate: decision,
|
|
388
|
+
});
|
|
389
|
+
return path;
|
|
390
|
+
}
|
|
391
|
+
/**
|
|
392
|
+
* Review I-2 — a re-audit PASS over multiple phases in ONE dispatch. Used by
|
|
393
|
+
* the repl's re-audit-on-resume branch: usually a single pending/infra phase,
|
|
394
|
+
* but a guarded hand-edit re-verification lists EVERY hand-edited satisfied
|
|
395
|
+
* phase (design: "an audit pass over hand-edited statuses"). Each phase runs
|
|
396
|
+
* the full chain with the offer suppressed except on the last; the saved path
|
|
397
|
+
* feeds forward so every verdict lands on the latest revision.
|
|
398
|
+
*
|
|
399
|
+
* Abort semantics: when a chain iteration queues a dispatch (the user picked
|
|
400
|
+
* re-run, or the final offer auto/consented), the pass stops — continuing to
|
|
401
|
+
* audit while a dispatch is pending would race the re-entry. Any phases left
|
|
402
|
+
* un-audited are named in a warning: hand-edit detection is history-author
|
|
403
|
+
* based and the pass's own writes wash it away, so they will NOT be
|
|
404
|
+
* re-detected automatically on the next resume. Stops (blocked/infra) do NOT
|
|
405
|
+
* abort the pass — infra persists as `infra_failed` (re-caught by the
|
|
406
|
+
* pending/infra scan next resume) and auditing the remaining phases is still
|
|
407
|
+
* useful information.
|
|
408
|
+
*/
|
|
409
|
+
export async function runReauditPass(args) {
|
|
410
|
+
// Task 10 review follow-up — a re-audit dispatch runs NO model turn, so a
|
|
411
|
+
// raised deviation flag here is never legitimate: it can only be stale
|
|
412
|
+
// residue from a chain leg that crashed before consuming it (possibly for
|
|
413
|
+
// a DIFFERENT plan in the same process). Discard it — mirroring the repl's
|
|
414
|
+
// phase-turn-start discard — or the deviation step inside runPostPhaseChain
|
|
415
|
+
// would wrongly block the re-audited phase and clearAudit would destroy
|
|
416
|
+
// the very 'pending' audit crash-safety depends on.
|
|
417
|
+
getAndResetPlanDeviation();
|
|
418
|
+
const { phaseIds, ...chainArgs } = args;
|
|
419
|
+
if (phaseIds.length === 0)
|
|
420
|
+
return args.savedPath;
|
|
421
|
+
if (phaseIds.length > 1) {
|
|
422
|
+
args.log(chalk.dim(`[audit] re-audit pass over ${phaseIds.length} phases: ${phaseIds.join(', ')}`));
|
|
423
|
+
}
|
|
424
|
+
let path = args.savedPath;
|
|
425
|
+
let queued = false;
|
|
426
|
+
const queueDispatch = (cmd) => { queued = true; args.queueDispatch(cmd); };
|
|
427
|
+
for (let i = 0; i < phaseIds.length; i++) {
|
|
428
|
+
const result = await runPostPhaseChain({
|
|
429
|
+
...chainArgs,
|
|
430
|
+
savedPath: path,
|
|
431
|
+
executedPhaseId: phaseIds[i],
|
|
432
|
+
forceAudit: true,
|
|
433
|
+
suppressOffer: i < phaseIds.length - 1,
|
|
434
|
+
queueDispatch,
|
|
435
|
+
});
|
|
436
|
+
if (result)
|
|
437
|
+
path = result;
|
|
438
|
+
if (queued) {
|
|
439
|
+
const remaining = phaseIds.slice(i + 1);
|
|
440
|
+
if (remaining.length > 0) {
|
|
441
|
+
args.log(chalk.yellow(`⚠ ${remaining.length} hand-edited phase(s) not re-audited yet: ${remaining.join(', ')} — `
|
|
442
|
+
+ 'they are not auto-detected after this write; verify them (or re-run their phases) once the queued dispatch completes.'));
|
|
443
|
+
}
|
|
444
|
+
break;
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
return path;
|
|
448
|
+
}
|
|
449
|
+
/* ── helpers ───────────────────────────────────────────────────────────── */
|
|
450
|
+
function loadPlan(path, log) {
|
|
451
|
+
try {
|
|
452
|
+
const envelope = readProjectFile(path);
|
|
453
|
+
if (envelope.artefactType !== 'plan') {
|
|
454
|
+
log(chalk.red(`[plan] ${basename(path)} is not a plan envelope — chain skipped.`));
|
|
455
|
+
return null;
|
|
456
|
+
}
|
|
457
|
+
const pc = parsePlanContent(envelope.content);
|
|
458
|
+
if (!pc.ok) {
|
|
459
|
+
log(chalk.red(`[plan] ${basename(path)} failed content validation — chain skipped: ${pc.errors[0] ?? ''}`));
|
|
460
|
+
return null;
|
|
461
|
+
}
|
|
462
|
+
return { envelope, content: pc.content };
|
|
463
|
+
}
|
|
464
|
+
catch (e) {
|
|
465
|
+
log(chalk.red(`[plan] could not load ${basename(path)} — ${e instanceof Error ? e.message : String(e)}`));
|
|
466
|
+
return null;
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
/**
|
|
470
|
+
* Terminal bell with the STOPPED banner — TTY only (a piped/CI stream must
|
|
471
|
+
* never receive a raw BEL byte).
|
|
472
|
+
*/
|
|
473
|
+
function ringBell() {
|
|
474
|
+
if (process.stdout.isTTY)
|
|
475
|
+
process.stdout.write('\u0007');
|
|
476
|
+
}
|
|
477
|
+
/** STOPPED banner in the plan-gate style, with the REAL resume command. */
|
|
478
|
+
function stoppedBanner(reason, resumeCommand) {
|
|
479
|
+
return `\n■ STOPPED — ${reason}\n Resume with: ${resumeCommand}\n`;
|
|
480
|
+
}
|
|
481
|
+
/**
|
|
482
|
+
* Task 3 left the gate banner's `@<plan>` placeholder literal (the pure
|
|
483
|
+
* module doesn't know the saved path) — substitute the real basename here.
|
|
484
|
+
*/
|
|
485
|
+
export function substitutePlanToken(banner, savedPath) {
|
|
486
|
+
return banner.replace(/@<plan>/g, `@${basename(savedPath)}`);
|
|
487
|
+
}
|
|
488
|
+
function clampInfra(msg) {
|
|
489
|
+
const flat = msg.replace(/\s+/g, ' ').trim();
|
|
490
|
+
return flat.length > INFRA_RENDER_CHARS ? flat.slice(0, INFRA_RENDER_CHARS - 1).trimEnd() + '…' : flat;
|
|
491
|
+
}
|
|
492
|
+
function renderFindings(log, phaseId, findings) {
|
|
493
|
+
log('', chalk.red(`✖ audit FAILED for phase ${phaseId}:`));
|
|
494
|
+
const shown = findings.length > 0 ? findings : ['(auditor returned no findings text)'];
|
|
495
|
+
shown.forEach((f, i) => log(` ${i + 1}. ${f}`));
|
|
496
|
+
}
|
|
497
|
+
/**
|
|
498
|
+
* Confidence-tiered redesign (2026-07-08) — the concise, always-shown render
|
|
499
|
+
* for a NON-BLOCKING warn verdict. Mirrors renderFindings' shape but frames the
|
|
500
|
+
* phase as passed-with-notes (the chain continues). Each finding is already
|
|
501
|
+
* formatted "[W1] detail" by the auditor (plan-audit.renderFinding).
|
|
502
|
+
*/
|
|
503
|
+
function renderWarnings(log, phaseId, findings) {
|
|
504
|
+
const n = findings.length;
|
|
505
|
+
log('', chalk.yellow(`⚠ audit passed with ${n} note${n === 1 ? '' : 's'} for phase ${phaseId}:`));
|
|
506
|
+
const shown = n > 0 ? findings : ['(auditor returned a warn verdict with no note detail)'];
|
|
507
|
+
shown.forEach((f, i) => log(` ${i + 1}. ${f}`));
|
|
508
|
+
}
|
|
509
|
+
/**
|
|
510
|
+
* audit-infra-continue (2026-07-09) — the LOUD, always-shown note for an
|
|
511
|
+
* audit that COULD NOT RUN. Distinct from warn's "passed with N notes": this
|
|
512
|
+
* frames the phase as unverified-but-proceeding, is explicit that the audit
|
|
513
|
+
* INFRASTRUCTURE failed (not the work), reassures that the phase's own exit
|
|
514
|
+
* gate passed, and points at the off-switch for a persistent outage — a
|
|
515
|
+
* pointer, not a forced prompt. Never swallowed (the chain continues, so
|
|
516
|
+
* nothing else surfaces this).
|
|
517
|
+
*/
|
|
518
|
+
function renderAuditNotRun(log, phaseId, reason, resumeCommand) {
|
|
519
|
+
log('', chalk.yellow(`⚠ audit could not run for phase ${phaseId} (audit infrastructure failed after 3 attempts — the model call returned no output).`), chalk.yellow(' This phase is NOT audit-verified; its own exit gate (activation + ATC) passed. Proceeding.'), chalk.dim(` reason: ${reason}`), chalk.dim(` If the auditor keeps failing, run with --no-audit (or set audit off in config) to stop retrying; resume as usual: ${resumeCommand}`));
|
|
520
|
+
}
|
|
521
|
+
/**
|
|
522
|
+
* Shared menu surface for the audit-resolution selects. Mirrors
|
|
523
|
+
* safety-confirm.ts: headless → no decision (fail-closed, nothing written);
|
|
524
|
+
* Ink → askQuestionEmitter choice modal; classic → inquirer select.
|
|
525
|
+
*/
|
|
526
|
+
async function selectAuditAction(args) {
|
|
527
|
+
const { isHeadless, shouldUseInk } = await import('../renderer/tty.js');
|
|
528
|
+
if (isHeadless())
|
|
529
|
+
return 'none';
|
|
530
|
+
if (shouldUseInk()) {
|
|
531
|
+
// D-C — this modal opens in the POST-PHASE chain, after the phase turn
|
|
532
|
+
// ended. Its sibling gate presentSafetyConfirmation (repl/safety-confirm.ts)
|
|
533
|
+
// documents the exact failure this caused live: opening the askQuestion
|
|
534
|
+
// modal while another render source (the in-place tool spinner, a ticking
|
|
535
|
+
// turn-status row) still repaints the Ink frame lets the first keystroke
|
|
536
|
+
// LEAK past the modal and auto-select its FIRST choice. Here the first
|
|
537
|
+
// choice is "Re-run phase", so the menu never visibly rendered and the
|
|
538
|
+
// phase silently re-ran — the reported symptom. Claim exclusive input
|
|
539
|
+
// focus the same way safety-confirm does: stop the spinner, freeze the
|
|
540
|
+
// status row, restore both in the finally.
|
|
541
|
+
const { clearActiveSpinner } = await import('../renderer/tool-widget.js');
|
|
542
|
+
const { turnStatusEmitter } = await import('../ui/turn-status-emitter.js');
|
|
543
|
+
const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
|
|
544
|
+
clearActiveSpinner();
|
|
545
|
+
const statusWasActive = turnStatusEmitter.snapshot().active;
|
|
546
|
+
if (statusWasActive)
|
|
547
|
+
turnStatusEmitter.pause();
|
|
548
|
+
try {
|
|
549
|
+
const res = await askQuestionEmitter.request({
|
|
550
|
+
id: args.id,
|
|
551
|
+
question: args.question,
|
|
552
|
+
context: args.context,
|
|
553
|
+
kind: 'choice',
|
|
554
|
+
choices: args.choices,
|
|
555
|
+
});
|
|
556
|
+
if (res.cancelled || res.answer == null)
|
|
557
|
+
return 'none';
|
|
558
|
+
return coerceAction(res.answer, args.choices);
|
|
559
|
+
}
|
|
560
|
+
finally {
|
|
561
|
+
if (statusWasActive)
|
|
562
|
+
turnStatusEmitter.resume();
|
|
563
|
+
}
|
|
564
|
+
}
|
|
565
|
+
const { select } = await import('@inquirer/prompts');
|
|
566
|
+
const { withInquirer } = await import('../repl/inquirer-guard.js');
|
|
567
|
+
const { inquirerTheme } = await import('../repl/inquirer-theme.js');
|
|
568
|
+
const picked = await withInquirer(() => select({
|
|
569
|
+
message: args.question,
|
|
570
|
+
theme: inquirerTheme,
|
|
571
|
+
choices: args.choices.map((c) => ({ value: c.value, name: c.label })),
|
|
572
|
+
}));
|
|
573
|
+
return coerceAction(picked, args.choices);
|
|
574
|
+
}
|
|
575
|
+
/**
|
|
576
|
+
* D-C — fail CLOSED on any answer that is not one of this menu's own choice
|
|
577
|
+
* values. `'' ?? 'none'` did NOT catch an empty-string answer (nullish
|
|
578
|
+
* coalescing only guards null/undefined), so a stray/blank answer used to fall
|
|
579
|
+
* through the runPostPhaseChain switch to a silent default. Anything the menu
|
|
580
|
+
* did not explicitly offer maps to 'none' (stop, no auto-action).
|
|
581
|
+
*/
|
|
582
|
+
function coerceAction(answer, choices) {
|
|
583
|
+
return choices.some((c) => c.value === answer) ? answer : 'none';
|
|
584
|
+
}
|
|
585
|
+
/** Production failed-verdict menu (see selectAuditAction for the surface). */
|
|
586
|
+
async function defaultSelectFailureAction(a) {
|
|
587
|
+
return selectAuditAction({
|
|
588
|
+
id: 'plan-audit-failed',
|
|
589
|
+
question: `Audit failed for phase ${a.phaseId} — what now?`,
|
|
590
|
+
context: a.findings.map((f, i) => `${i + 1}. ${f}`).join('\n'),
|
|
591
|
+
choices: [
|
|
592
|
+
{ value: 'rerun', label: 'Re-run phase (re-audited after execution)' },
|
|
593
|
+
{ value: 'waive', label: 'Waive audit (record a one-line reason)' },
|
|
594
|
+
{ value: 'blocked', label: 'Mark phase blocked' },
|
|
595
|
+
],
|
|
596
|
+
});
|
|
597
|
+
}
|
|
598
|
+
/**
|
|
599
|
+
* §10 Q2 (audit-confidence redesign) — production waive-category menu. A LIGHT
|
|
600
|
+
* 2-choice shown right after the waive-reason prompt so the audit's real
|
|
601
|
+
* false-positive rate is measurable. Same surface as selectAuditAction, but
|
|
602
|
+
* headless / cancel / stray answer → 'accepted-risk' (the CONSERVATIVE default:
|
|
603
|
+
* a waive no human explicitly labelled a false-positive is recorded as
|
|
604
|
+
* accepted-risk, so the measured FP rate never over-counts). Never hangs.
|
|
605
|
+
*/
|
|
606
|
+
async function defaultSelectWaiveCategory(a) {
|
|
607
|
+
const { isHeadless, shouldUseInk } = await import('../renderer/tty.js');
|
|
608
|
+
if (isHeadless())
|
|
609
|
+
return 'accepted-risk';
|
|
610
|
+
const question = `Waiving ${a.phaseId} — why? (measures the audit's false-positive rate)`;
|
|
611
|
+
const choices = [
|
|
612
|
+
{ value: 'accepted-risk', label: 'Accepted risk — the finding is real, I accept it and proceed' },
|
|
613
|
+
{ value: 'false-positive', label: 'False positive — the audit was wrong, the work is actually fine' },
|
|
614
|
+
];
|
|
615
|
+
const coerce = (answer) => choices.some((c) => c.value === answer) ? answer : 'accepted-risk';
|
|
616
|
+
if (shouldUseInk()) {
|
|
617
|
+
// Same input-focus discipline as selectAuditAction — this modal opens in
|
|
618
|
+
// the post-phase chain, so a leaked keystroke must not auto-select choice 1.
|
|
619
|
+
const { clearActiveSpinner } = await import('../renderer/tool-widget.js');
|
|
620
|
+
const { turnStatusEmitter } = await import('../ui/turn-status-emitter.js');
|
|
621
|
+
const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
|
|
622
|
+
clearActiveSpinner();
|
|
623
|
+
const statusWasActive = turnStatusEmitter.snapshot().active;
|
|
624
|
+
if (statusWasActive)
|
|
625
|
+
turnStatusEmitter.pause();
|
|
626
|
+
try {
|
|
627
|
+
const res = await askQuestionEmitter.request({
|
|
628
|
+
id: 'plan-waive-category',
|
|
629
|
+
question,
|
|
630
|
+
kind: 'choice',
|
|
631
|
+
choices,
|
|
632
|
+
});
|
|
633
|
+
if (res.cancelled || res.answer == null)
|
|
634
|
+
return 'accepted-risk';
|
|
635
|
+
return coerce(res.answer);
|
|
636
|
+
}
|
|
637
|
+
finally {
|
|
638
|
+
if (statusWasActive)
|
|
639
|
+
turnStatusEmitter.resume();
|
|
640
|
+
}
|
|
641
|
+
}
|
|
642
|
+
const { select } = await import('@inquirer/prompts');
|
|
643
|
+
const { withInquirer } = await import('../repl/inquirer-guard.js');
|
|
644
|
+
const { inquirerTheme } = await import('../repl/inquirer-theme.js');
|
|
645
|
+
const picked = await withInquirer(() => select({
|
|
646
|
+
message: question,
|
|
647
|
+
theme: inquirerTheme,
|
|
648
|
+
choices: choices.map((c) => ({ value: c.value, name: c.label })),
|
|
649
|
+
}));
|
|
650
|
+
return coerce(picked);
|
|
651
|
+
}
|
|
652
|
+
/**
|
|
653
|
+
* Audit spend surfaces twice: a dim line next to the turn footer (the
|
|
654
|
+
* `audit` entry) and a labelled turn-0 line in the session cost JSONL —
|
|
655
|
+
* the audit runs in its own isolated session, so without this its tokens
|
|
656
|
+
* would be invisible to the user's cost trail. Best-effort, never throws.
|
|
657
|
+
*/
|
|
658
|
+
function renderAuditCost(args, cost) {
|
|
659
|
+
try {
|
|
660
|
+
const tok = cost.tokens.input + cost.tokens.output;
|
|
661
|
+
const usd = computeCost(cost.model, cost.tokens);
|
|
662
|
+
args.log(chalk.dim(`[audit] ${tok.toLocaleString()} tok · ${formatCost(usd)} (${cost.model})`));
|
|
663
|
+
void appendCostLine(args.sessionId, {
|
|
664
|
+
...buildEntry({ turn: 0, model: cost.model, tokens: cost.tokens, duration_ms: cost.durationMs }),
|
|
665
|
+
label: 'audit',
|
|
666
|
+
});
|
|
667
|
+
}
|
|
668
|
+
catch {
|
|
669
|
+
// observability only
|
|
670
|
+
}
|
|
671
|
+
}
|