@cspeach/cli 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/dist/agent/loop.js +22 -9
  2. package/dist/approvals/op-labels.js +124 -0
  3. package/dist/approvals/render.js +42 -36
  4. package/dist/cli.js +15 -0
  5. package/dist/commands/compact.js +28 -2
  6. package/dist/commands/config-set.js +189 -0
  7. package/dist/commands/config-show.js +20 -0
  8. package/dist/commands/export-audit.js +43 -0
  9. package/dist/commands/help.js +5 -0
  10. package/dist/commands/plan-audit-evidence.js +266 -0
  11. package/dist/commands/plan-audit.js +692 -0
  12. package/dist/commands/plan-chain.js +671 -0
  13. package/dist/commands/plan-continue.js +179 -0
  14. package/dist/commands/plan-gate.js +154 -0
  15. package/dist/commands/plan-resume.js +588 -33
  16. package/dist/config/loader.js +128 -4
  17. package/dist/config/model-defaults.js +14 -0
  18. package/dist/cost/pricing.js +27 -1
  19. package/dist/doctor/checks/system-roles.js +41 -0
  20. package/dist/doctor/run.js +2 -0
  21. package/dist/models/resolve.js +61 -0
  22. package/dist/models/server-config.js +155 -0
  23. package/dist/one-shot.js +25 -3
  24. package/dist/projects/extract-cca.js +3 -1
  25. package/dist/projects/extract-modernize.js +3 -1
  26. package/dist/projects/extract-plan.js +60 -6
  27. package/dist/projects/extract-test-coverage.js +3 -1
  28. package/dist/projects/extract-upgrade.js +3 -1
  29. package/dist/projects/handover-md.js +195 -0
  30. package/dist/projects/index.js +1 -1
  31. package/dist/projects/plan-run.js +137 -13
  32. package/dist/projects/plan-schema.js +73 -0
  33. package/dist/projects/run-lease.js +157 -0
  34. package/dist/projects/save-command.js +26 -15
  35. package/dist/renderer/status-footer.js +22 -12
  36. package/dist/renderer/thinking-heartbeat.js +64 -8
  37. package/dist/renderer/todo-block.js +51 -0
  38. package/dist/renderer/tool-widget.js +37 -0
  39. package/dist/repl/bracketed-paste.js +28 -19
  40. package/dist/repl/builtin-commands.js +5 -0
  41. package/dist/repl/current-transport.js +10 -0
  42. package/dist/repl/history.js +86 -0
  43. package/dist/repl/ink-stdin-guard.js +64 -0
  44. package/dist/repl/mode-ceiling.js +16 -0
  45. package/dist/repl/mode-cycle.js +104 -0
  46. package/dist/repl/post-turn-status.js +24 -4
  47. package/dist/repl/slash-completer.js +5 -0
  48. package/dist/repl.js +954 -83
  49. package/dist/rewind/candidates.js +194 -0
  50. package/dist/rewind/cli.js +137 -0
  51. package/dist/rewind/format.js +27 -0
  52. package/dist/rewind/restore.js +245 -0
  53. package/dist/session/audit-export.js +459 -0
  54. package/dist/session/context-report.js +163 -0
  55. package/dist/session/recap.js +160 -0
  56. package/dist/skill-catalog.js +9 -3
  57. package/dist/skills/bundled-skills.js +59 -66
  58. package/dist/tools/approval.js +115 -7
  59. package/dist/tools/ask-question.js +304 -3
  60. package/dist/tools/extend-model/anchored-insert.js +604 -0
  61. package/dist/tools/extend-model/tool.js +162 -10
  62. package/dist/tools/fiori/fe-extend.js +76 -0
  63. package/dist/tools/fiori/fe-scaffold.js +29 -3
  64. package/dist/tools/fiori/floorplan-map.js +19 -0
  65. package/dist/tools/fiori/samples/data/index.json +13602 -0
  66. package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
  67. package/dist/tools/fiori/samples/loader.js +248 -0
  68. package/dist/tools/fiori/samples/search.js +63 -0
  69. package/dist/tools/fiori/samples/types.js +2 -0
  70. package/dist/tools/fiori/smoke/assertions.js +74 -0
  71. package/dist/tools/fiori/smoke/browser.js +52 -0
  72. package/dist/tools/fiori/smoke/driver.js +89 -0
  73. package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
  74. package/dist/tools/fiori/smoke/run-smoke.js +149 -0
  75. package/dist/tools/fiori/tools.js +328 -3
  76. package/dist/tools/local-build.js +11 -1
  77. package/dist/tools/sap-read.js +79 -11
  78. package/dist/tools/sap-write.js +24 -4
  79. package/dist/tools/snapshot.js +27 -1
  80. package/dist/tools/subagent/agent_run.js +27 -3
  81. package/dist/tools/todo.js +144 -0
  82. package/dist/ui/app.js +372 -19
  83. package/dist/ui/approval-modal.js +49 -16
  84. package/dist/ui/ask-question-emitter.js +14 -0
  85. package/dist/ui/context-grid.js +108 -0
  86. package/dist/ui/footer.js +109 -30
  87. package/dist/ui/header.js +7 -0
  88. package/dist/ui/line-resolution.js +18 -2
  89. package/dist/ui/rewind-emitter.js +10 -0
  90. package/dist/ui/rewind-panel.js +81 -0
  91. package/dist/ui/sap-state-store.js +1 -0
  92. package/dist/ui/status-line.js +43 -0
  93. package/dist/ui/text-input.js +72 -8
  94. package/dist/ui/todo-emitter.js +25 -0
  95. package/dist/ui/todo-panel.js +64 -0
  96. package/dist/ui/turn-status-emitter.js +50 -4
  97. package/dist/ui/turn-status.js +18 -3
  98. package/dist/ui/widgets/ask-form.js +242 -0
  99. package/dist/ui/widgets/ask-question-modal.js +17 -7
  100. package/package.json +4 -1
@@ -0,0 +1,671 @@
1
+ // cspeach-cli/src/commands/plan-chain.ts
2
+ //
3
+ // Task 6 (agentic-flow, 2026-07-03) — the shared post-phase sequence, called
4
+ // by BOTH repl.tsx plan-resume call sites (Ink + classic) so neither
5
+ // duplicates the block. Exact order (the crash-safety design):
6
+ //
7
+ // finishPlanResume persisted the phase WITH audit:'pending' (caller)
8
+ // → runPhaseAudit (fresh-context, tool-less judge turn)
9
+ // → applyAuditResult (verdict persisted as revision N+1)
10
+ // → verdict handling (failed ⇒ re-run/waive/block; infra ⇒ stop,
11
+ // 2nd consecutive infra ⇒ retry/waive/block)
12
+ // → decideNextPhaseGate (step/guarded transition decision)
13
+ // → offerNextPhaseAutoRun (prompt, or auto-queue in guarded mode)
14
+ //
15
+ // A crash anywhere before applyAuditResult leaves 'pending' on disk;
16
+ // preparePlanResume's re-audit-on-resume then runs this chain FIRST on the
17
+ // next --resume (executedPhaseId = the pending phase, forceAudit) instead of
18
+ // executing a new phase — the audit is never skipped.
19
+ //
20
+ // Everything I/O-heavy or interactive is injectable via PlanChainDeps so the
21
+ // chain is unit-testable without a provider, Ink, or inquirer.
22
+ import { basename } from 'node:path';
23
+ import chalk from 'chalk';
24
+ import { readProjectFile } from '../projects/status.js';
25
+ import { parsePlanContent } from '../projects/plan-schema.js';
26
+ import { statusesFromItems, computeNextPhase, isPhaseSatisfied, isAuditResolved, planCompletionLines, renderPlanTracker, } from '../projects/plan-run.js';
27
+ import { decideNextPhaseGate, getAndResetPlanPhaseSubagentDispatches, getAndResetPlanDeviation, } from './plan-gate.js';
28
+ import { runPhaseAudit } from './plan-audit.js';
29
+ import { applyAuditResult, applyPhaseResolution, buildResumeCommand, offerNextPhaseAutoRun, } from './plan-resume.js';
30
+ import { appendCostLine, buildEntry } from '../cost/cost-log.js';
31
+ import { startThinkingHeartbeat, AUDIT_HEARTBEAT_THRESHOLDS } from '../renderer/thinking-heartbeat.js';
32
+ import { computeCost, formatCost } from '../cost/pricing.js';
33
+ /** Two consecutive failed verdicts on the same phase force status 'blocked'. */
34
+ export const MAX_CONSECUTIVE_AUDIT_FAILURES = 2;
35
+ /** Infra error strings are model/HTTP noise — clamp when rendering (Task 5 review). */
36
+ const INFRA_RENDER_CHARS = 200;
37
+ /**
38
+ * Run the post-phase chain. Returns the LATEST saved envelope path (the
39
+ * caller's savedPath when nothing was written), or null when the chain could
40
+ * not even load the envelope.
41
+ */
42
+ export async function runPostPhaseChain(args) {
43
+ let path = args.savedPath;
44
+ let loaded = loadPlan(path, args.log);
45
+ if (!loaded)
46
+ return null;
47
+ let phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
48
+ if (!phase) {
49
+ args.log(chalk.red(`[plan] phase '${args.executedPhaseId}' not found in ${basename(path)} — chain skipped.`));
50
+ return path;
51
+ }
52
+ // Per-run flags survive the chain: an explicitly flagged run keeps its mode
53
+ // AND its audit setting across every harness-built resume command (I2 — the
54
+ // audit flag mirrors modeFlag; --audit maps to itself, 'off' to --no-audit).
55
+ const flagSuffix = (args.modeFlag ? ` --${args.modeFlag}` : '')
56
+ + (args.auditFlag === 'off' ? ' --no-audit' : args.auditFlag === 'on' ? ' --audit' : '');
57
+ const resumeCmd = (p) => buildResumeCommand(p) + flagSuffix;
58
+ let chainWrote = false;
59
+ // Task 7 — phase-end subagent report. agent_run counted every dispatch
60
+ // made while the plan-phase flag was on (all forced read-only). The REPL
61
+ // reads+resets the counter in its runTurn finally and threads the value
62
+ // here (fix-wave, item 1) — this chain can be skipped on error paths, so
63
+ // consuming the counter here alone would let counts leak into the next
64
+ // phase's report line. The read+reset below is only the fallback for
65
+ // callers that ran no phase turn (re-audit passes, direct test calls) —
66
+ // zero there in practice, so nothing is printed.
67
+ const subagentDispatches = args.subagentDispatches ?? getAndResetPlanPhaseSubagentDispatches();
68
+ if (subagentDispatches > 0) {
69
+ args.log(chalk.dim(`[phase] ${args.executedPhaseId}: ${subagentDispatches} subagent dispatch(es) — read-only enforced`));
70
+ }
71
+ /* ── 0. deviation backstop (Task 10) ──────────────────────────────────────
72
+ * approval.ts raised the flag mid-turn: this phase declared writes:false
73
+ * but requested a write approval anyway — the write was DENIED without
74
+ * prompting; here the chain records why and stops. Design choice (per the
75
+ * task's simpler-path option): SKIP the audit — the phase didn't complete
76
+ * honestly and was deliberately stopped, so auditing it wastes a turn. The
77
+ * finishPlanResume-injected 'pending' audit is dropped in the same write
78
+ * (clearAudit) or the next resume would re-audit the blocked phase. Same
79
+ * read+reset pattern as the subagent counter — the flag never leaks into
80
+ * the next chain leg. Guarded-only by construction: approval.ts only ever
81
+ * raises it while isGuardedRunActive(). */
82
+ const deviationDetail = getAndResetPlanDeviation();
83
+ if (deviationDetail) {
84
+ path = await applyPhaseResolution(path, phase.id, {
85
+ status: 'blocked',
86
+ lastStop: { phaseId: phase.id, reason: 'deviation', detail: deviationDetail, at: new Date().toISOString() },
87
+ summary: `guarded chain stopped: plan deviation in ${phase.id} — ${deviationDetail}`,
88
+ clearAudit: true,
89
+ log: args.log,
90
+ });
91
+ args.log(chalk.red(`✖ plan deviation in phase ${phase.id} — ${deviationDetail}`), stoppedBanner(`plan deviation — phase ${phase.id} blocked (undeclared write denied)`, resumeCmd(path)));
92
+ ringBell();
93
+ return path;
94
+ }
95
+ /* ── 1. audit (only when unresolved-pending/infra, or forced) ─────────── */
96
+ // Off-switch: audits off skips the NORMAL post-phase audit (no injected
97
+ // pending exists to catch anyway). A forced re-audit still runs — a persisted
98
+ // pending/infra record from an earlier ON run must resolve regardless.
99
+ const auditState = phase.audit?.state;
100
+ const auditsOff = args.auditMode === 'off';
101
+ // audit-infra-continue (2026-07-09): 'infra_failed' is now RESOLVED and no
102
+ // longer auto-re-audited — an audit that could not run already let the chain
103
+ // continue, so it must not spontaneously re-run on a later pass. Only a
104
+ // persisted 'pending' (crash between phase-save and verdict-save) or an
105
+ // explicit forceAudit (hand-edit re-verification) triggers an audit here.
106
+ const needsAudit = args.forceAudit
107
+ || (!auditsOff && auditState === 'pending');
108
+ if (needsAudit) {
109
+ args.log('', chalk.dim(`[audit] independent audit of phase ${phase.id} — fresh context, judged against the tool-call evidence…`));
110
+ // Evidence lives with the session that EXECUTED the phase; on a re-audit
111
+ // in a fresh CLI session that id was recorded on the pending audit.
112
+ const evidenceSessionId = phase.audit?.sessionId ?? args.sessionId;
113
+ const auditFn = args.deps?.runPhaseAuditFn ?? runPhaseAudit;
114
+ let cost;
115
+ // D-B — window the evidence to THIS attempt only. Live post-phase audit:
116
+ // the repl threads the turn-start it captured (attemptStartedAt).
117
+ // RE-audit (review item 1): the OLD session's JSONL can hold MULTIPLE
118
+ // attempts (fail → rerun in the same process → die before verdict →
119
+ // resume re-audits), so an unwindowed re-audit re-creates the exact
120
+ // cross-attempt false-ordering on the standard recovery path — use the
121
+ // window finishPlanResume persisted on the audit record
122
+ // (audit.attemptStartedAt, carried through verdict writes). A legacy
123
+ // record without the field runs unwindowed (fail-open, no window claim
124
+ // is made to the auditor either — see buildAuditPrompt).
125
+ const sinceIso = args.attemptStartedAt ?? phase.audit?.attemptStartedAt;
126
+ // D-F item 4 — liveness. The audit is a single fresh-context model turn
127
+ // (10-40s) with no streaming into this chain, so the [audit] header sat
128
+ // silent. Reuse the house-style heartbeat (same primitive loop.ts uses),
129
+ // routed through args.log so it works under both Ink and classic.
130
+ // audit-timeout (2026-07-06): use the audit-specific schedule — the audit
131
+ // is a PROXY call (api.cspeach.dev), so the generic "check the VPN" 5m copy
132
+ // is the wrong diagnosis. runPhaseAudit now hard-caps each attempt at 120s,
133
+ // so this can never sit stuck forever. Attempt transitions are surfaced via
134
+ // onRetry below. Always stopped in finally.
135
+ // audit-lean FIX #2 — a LIVE token count threaded from the auditor's stream
136
+ // into the heartbeat suffix, so a stall (`· 0 tokens`) reads differently
137
+ // from slow-but-working progress. The suffix closure reads the latest value
138
+ // at each heartbeat tick; onProgress updates it as chunks arrive.
139
+ let auditTokens = 0;
140
+ const heartbeat = startThinkingHeartbeat({
141
+ thresholds: AUDIT_HEARTBEAT_THRESHOLDS,
142
+ emit: (line) => args.log(line),
143
+ suffix: () => ` · ${auditTokens.toLocaleString()} token${auditTokens === 1 ? '' : 's'}`,
144
+ });
145
+ let verdict;
146
+ try {
147
+ verdict = await auditFn({
148
+ envelope: loaded.envelope,
149
+ phaseId: phase.id,
150
+ sessionId: evidenceSessionId,
151
+ cwd: args.cwd,
152
+ ...(sinceIso ? { sinceIso } : {}),
153
+ onCost: (c) => { cost = c; },
154
+ // Best-effort: the heartbeat reads auditTokens at each tick.
155
+ onProgress: (n) => { auditTokens = n; },
156
+ // audit-timeout — an honest attempt-transition line when a prior
157
+ // attempt hit its deadline (or otherwise failed) and we retry in a
158
+ // fresh session. Keeps the timeline truthful against the 120s×3 ceiling.
159
+ onRetry: (r) => {
160
+ const why = r.reason === 'timeout' ? 'previous attempt timed out'
161
+ : r.reason === 'unparseable' ? 'previous attempt returned no verdict'
162
+ : 'previous attempt errored';
163
+ args.log(chalk.dim(` [audit] attempt ${r.attempt} of ${r.totalAttempts} (${why})`));
164
+ },
165
+ });
166
+ }
167
+ finally {
168
+ heartbeat.stop();
169
+ }
170
+ if (cost)
171
+ renderAuditCost(args, cost);
172
+ // Fix-wave (item 10a) — a persisted 'failed' reaching a re-audit means
173
+ // the session died AT the failure menu (verdict written, resolution
174
+ // never was). A failed verdict then RE-CONFIRMS the same failure: hold
175
+ // the counter instead of incrementing, and re-render the menu — the
176
+ // developer must still get the re-run/waive/block choice they lost.
177
+ // Delta-review edge: an interleaved infra outage launders the same case
178
+ // — failed(1) → menu death → re-audit hits infra (the counter is carried
179
+ // onto infra_failed) → the NEXT re-audit's failed verdict would count as
180
+ // consecutive and silently force-block. A persisted infra_failed that
181
+ // CARRIES a failure counter is therefore also a reconfirmation: the
182
+ // menu was owed and never shown.
183
+ const reconfirmingFailed = auditState === 'failed'
184
+ || (auditState === 'infra_failed' && (phase.audit?.consecutiveFailures ?? 0) >= 1);
185
+ // Consecutive-failure decision happens BEFORE the write so the forced
186
+ // block lands in the SAME revision as the verdict.
187
+ const priorFailures = phase.audit?.consecutiveFailures ?? 0;
188
+ const countedFailures = reconfirmingFailed
189
+ ? Math.max(priorFailures, 1)
190
+ : priorFailures + 1;
191
+ const forceBlock = verdict.state === 'failed'
192
+ && countedFailures >= MAX_CONSECUTIVE_AUDIT_FAILURES;
193
+ const nowIso = new Date().toISOString();
194
+ // Only a real 'failed' verdict records a stop. 'infra_failed' no longer
195
+ // stops (audit-infra-continue) — like warn/pass it leaves no lastStop, so
196
+ // applyAuditResult clears any stale one (see its resolved-verdict branch).
197
+ const lastStop = verdict.state === 'failed'
198
+ ? { phaseId: phase.id, reason: 'audit-failed', detail: clampInfra(verdict.findings[0] ?? ''), at: nowIso }
199
+ : undefined;
200
+ // Confidence-tiered redesign (2026-07-08) — the GATE. A `warn` verdict is
201
+ // NON-BLOCKING: it persists LITERALLY (state:'warn', findings carried),
202
+ // clears the failure counter (a warn is not a failure), and the chain
203
+ // continues exactly like a pass — no menu, no lastStop, forceBlock can
204
+ // never fire on it (forceBlock keys off verdict.state === 'failed' above).
205
+ // fail / infra_failed / pass persistence is byte-unchanged.
206
+ path = await applyAuditResult(path, phase.id, {
207
+ state: verdict.state,
208
+ findings: verdict.findings,
209
+ }, {
210
+ lastStop,
211
+ ...(forceBlock ? { statusOverride: 'blocked' } : {}),
212
+ ...(reconfirmingFailed ? { reconfirmedFailure: true } : {}),
213
+ log: args.log,
214
+ });
215
+ chainWrote = true;
216
+ loaded = loadPlan(path, args.log);
217
+ if (!loaded)
218
+ return path;
219
+ phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
220
+ // D-E — the progress board renders HERE, AFTER the verdict lands, so the
221
+ // tally is audit-aware and never contradicts a later verdict. (finishPlanResume
222
+ // now prints only a neutral "audit pending…" line; the board it used to
223
+ // print counted the just-run phase as done because it erased the pending
224
+ // audit for display — a validated phase with a pending audit is NOT
225
+ // satisfied, so the pre-audit board read "2/3" for what was really "1/3".)
226
+ // Suppressed on mid-pass re-audit legs (runReauditPass renders once at the
227
+ // end via its final, unsuppressed leg).
228
+ if (!args.suppressOffer) {
229
+ const boardStatuses = statusesFromItems(loaded.envelope.interaction.items, loaded.content.phases);
230
+ args.log('', renderPlanTracker({
231
+ title: loaded.envelope.title,
232
+ version: loaded.envelope.version,
233
+ content: loaded.content,
234
+ statuses: boardStatuses,
235
+ currentId: null,
236
+ auditsOff,
237
+ }), '');
238
+ }
239
+ if (verdict.state === 'infra_failed') {
240
+ // audit-infra-continue (2026-07-09): the audit COULD NOT RUN — after the
241
+ // in-turn 3 attempts (runPhaseAudit's 120s×3 ceiling) the model call
242
+ // returned no output. That is an audit-INFRASTRUCTURE failure, not a
243
+ // finding against the phase, whose own exit gate (activation + ATC)
244
+ // already passed. So we do NOT hard-stop verified work: exactly like a
245
+ // warn/pass, render a LOUD, DISTINCT "could not run / NOT verified" note,
246
+ // keep the infra_failed state recorded (tracker keeps ⚠ audit not run;
247
+ // the final summary lists it as unverified) and CONTINUE to the gate +
248
+ // offer below. isAuditResolved treats infra_failed as resolved, so
249
+ // dependants unlock and the failure menu is never shown.
250
+ renderAuditNotRun(args.log, phase.id, clampInfra(verdict.findings[0] ?? 'auditor infrastructure failure'), resumeCmd(path));
251
+ }
252
+ else if (verdict.state === 'failed') {
253
+ renderFindings(args.log, phase.id, verdict.findings);
254
+ if (forceBlock) {
255
+ args.log(chalk.red(`✖ phase ${phase.id} failed its audit ${countedFailures} times in a row — marked blocked.`), stoppedBanner('two consecutive failed audits — phase blocked', resumeCmd(path)));
256
+ ringBell();
257
+ return path;
258
+ }
259
+ const select = args.deps?.selectFailureAction ?? defaultSelectFailureAction;
260
+ const action = await select({ phaseId: phase.id, findings: verdict.findings });
261
+ if (action === 'none') {
262
+ // D-C — fail CLOSED. 'none' is headless OR an interactive decline/cancel
263
+ // (Esc, no listener, or an unrecognised menu answer). The phase must NOT
264
+ // auto-re-run — a re-run is an explicit menu choice, never a default.
265
+ // Stop with the banner; resuming re-audits and re-offers the menu (the
266
+ // persisted 'failed' on a still-satisfied phase is re-caught by
267
+ // preparePlanResume's died-at-menu scan).
268
+ args.log(stoppedBanner(`audit failed for phase ${phase.id} — resolve interactively (re-run / waive / block)`, resumeCmd(path)));
269
+ ringBell();
270
+ return path;
271
+ }
272
+ if (action === 'blocked') {
273
+ path = await applyPhaseResolution(path, phase.id, {
274
+ status: 'blocked',
275
+ lastStop: { phaseId: phase.id, reason: 'audit-failed', detail: 'user marked the phase blocked after a failed audit', at: new Date().toISOString() },
276
+ summary: `audit ${phase.id}: failed — user marked phase blocked`,
277
+ log: args.log,
278
+ });
279
+ args.log(stoppedBanner(`phase ${phase.id} marked blocked after failed audit`, resumeCmd(path)));
280
+ ringBell();
281
+ return path;
282
+ }
283
+ if (action === 'rerun') {
284
+ // Status back to 'todo' — computeNextPhase re-picks it, and the flow
285
+ // ALWAYS audits an executed phase, so a re-run is re-audited by
286
+ // construction. The failed audit + counter stay on the phase until
287
+ // the re-run's finishPlanResume replaces them with 'pending'.
288
+ path = await applyPhaseResolution(path, phase.id, {
289
+ status: 'todo',
290
+ summary: `audit ${phase.id}: failed — user chose re-run (status → todo)`,
291
+ log: args.log,
292
+ });
293
+ args.log(`↩ re-running phase ${phase.id} — it will be re-audited after execution.`);
294
+ args.queueDispatch(resumeCmd(path));
295
+ return path;
296
+ }
297
+ // action === 'waive'
298
+ const reasonRaw = (await args.prompt(`Waive reason for ${phase.id} (one line): `)).trim();
299
+ const reason = reasonRaw.length > 0 ? reasonRaw : 'waived by user (no reason given)';
300
+ // §10 Q2 — a light 2-choice category alongside the reason: was the audit
301
+ // WRONG (false-positive) or is the finding real-but-accepted
302
+ // (accepted-risk)? Persisted so the audit's FP rate is measurable.
303
+ // Headless / cancel → accepted-risk (never a false-positive unasserted).
304
+ const category = await (args.deps?.selectWaiveCategory ?? defaultSelectWaiveCategory)({ phaseId: phase.id });
305
+ path = await applyAuditResult(path, phase.id, { state: 'waived', waivedReason: reason, waiveCategory: category }, { log: args.log });
306
+ loaded = loadPlan(path, args.log);
307
+ if (!loaded)
308
+ return path;
309
+ phase = loaded.content.phases.find((p) => p.id === args.executedPhaseId);
310
+ args.log(chalk.yellow(`✎ audit waived (${category}) for ${phase.id} — reason recorded in the envelope.`));
311
+ }
312
+ else if (verdict.state === 'warn') {
313
+ // Non-blocking tier: ALWAYS shown, concise (Open-Q decision — warns are
314
+ // never swallowed). The chain continues to the gate + offer below just
315
+ // like a pass; the notes are already recorded on phase.audit.findings.
316
+ renderWarnings(args.log, phase.id, verdict.findings);
317
+ }
318
+ else {
319
+ args.log(chalk.green(`✔ audit passed — ${phase.id} verified against the evidence.`));
320
+ }
321
+ }
322
+ /* ── 2. gate + offer ──────────────────────────────────────────────────── */
323
+ // Review I-2 — mid-pass re-audits stop after the verdict: the pass owner
324
+ // (runReauditPass) issues one offer at the end.
325
+ if (args.suppressOffer)
326
+ return path;
327
+ const statuses = statusesFromItems(loaded.envelope.interaction.items, loaded.content.phases);
328
+ const next = computeNextPhase(loaded.content.phases, statuses);
329
+ if (!next) {
330
+ // Terminal messages only when this chain changed something — otherwise
331
+ // finishPlanResume already printed the same conclusion (no duplicates).
332
+ if (chainWrote) {
333
+ if (loaded.content.phases.every((p) => isPhaseSatisfied(statuses[p.id], p.audit))) {
334
+ args.log(...planCompletionLines(loaded.content.phases));
335
+ }
336
+ else {
337
+ const blocked = loaded.content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
338
+ const unresolved = loaded.content.phases
339
+ .filter((p) => isPhaseSatisfied(statuses[p.id]) && !isAuditResolved(p.audit))
340
+ .map((p) => `${p.id} (audit ${p.audit?.state ?? 'unknown'})`);
341
+ const unresolvedNote = unresolved.length > 0 ? ` Unresolved audits: ${unresolved.join(', ')}.` : '';
342
+ args.log(`No eligible next phase. Blocked: ${blocked.join(', ') || '(none)'}.${unresolvedNote}`);
343
+ }
344
+ }
345
+ return path;
346
+ }
347
+ const decision = decideNextPhaseGate({
348
+ mode: args.mode,
349
+ nextPhase: { id: next.id, writes: next.writes },
350
+ prevPhaseAudit: phase?.audit,
351
+ });
352
+ if (decision.action === 'stop') {
353
+ path = await applyPhaseResolution(path, args.executedPhaseId, {
354
+ lastStop: {
355
+ phaseId: args.executedPhaseId,
356
+ reason: decision.stopReason === 'audit-infra' ? 'audit-infra' : 'audit-failed',
357
+ at: new Date().toISOString(),
358
+ },
359
+ summary: `guarded chain stopped: ${decision.stopReason}`,
360
+ log: args.log,
361
+ });
362
+ args.log(substitutePlanToken(decision.banner ?? stoppedBanner('guarded chain stopped', resumeCmd(path)), path));
363
+ ringBell();
364
+ return path;
365
+ }
366
+ if (decision.action === 'prompt' && decision.stopReason === 'write-phase') {
367
+ // The prompt IS the stop (guarded semantics) — record it crash-safe
368
+ // BEFORE blocking on the user, then render the banner + bell.
369
+ path = await applyPhaseResolution(path, args.executedPhaseId, {
370
+ lastStop: { phaseId: next.id, reason: 'write-phase', at: new Date().toISOString() },
371
+ summary: `guarded chain stopped: write-phase consent for ${next.id}`,
372
+ log: args.log,
373
+ });
374
+ args.log(substitutePlanToken(decision.banner ?? '', path));
375
+ ringBell();
376
+ }
377
+ await offerNextPhaseAutoRun({
378
+ outcome: {
379
+ savedPath: path,
380
+ resumeCommand: resumeCmd(path),
381
+ nextPhaseId: next.id,
382
+ nextDelegateTo: next.delegateTo,
383
+ },
384
+ prompt: args.prompt,
385
+ queueDispatch: args.queueDispatch,
386
+ log: args.log,
387
+ gate: decision,
388
+ });
389
+ return path;
390
+ }
391
+ /**
392
+ * Review I-2 — a re-audit PASS over multiple phases in ONE dispatch. Used by
393
+ * the repl's re-audit-on-resume branch: usually a single pending/infra phase,
394
+ * but a guarded hand-edit re-verification lists EVERY hand-edited satisfied
395
+ * phase (design: "an audit pass over hand-edited statuses"). Each phase runs
396
+ * the full chain with the offer suppressed except on the last; the saved path
397
+ * feeds forward so every verdict lands on the latest revision.
398
+ *
399
+ * Abort semantics: when a chain iteration queues a dispatch (the user picked
400
+ * re-run, or the final offer auto/consented), the pass stops — continuing to
401
+ * audit while a dispatch is pending would race the re-entry. Any phases left
402
+ * un-audited are named in a warning: hand-edit detection is history-author
403
+ * based and the pass's own writes wash it away, so they will NOT be
404
+ * re-detected automatically on the next resume. Stops (blocked/infra) do NOT
405
+ * abort the pass — infra persists as `infra_failed` (re-caught by the
406
+ * pending/infra scan next resume) and auditing the remaining phases is still
407
+ * useful information.
408
+ */
409
+ export async function runReauditPass(args) {
410
+ // Task 10 review follow-up — a re-audit dispatch runs NO model turn, so a
411
+ // raised deviation flag here is never legitimate: it can only be stale
412
+ // residue from a chain leg that crashed before consuming it (possibly for
413
+ // a DIFFERENT plan in the same process). Discard it — mirroring the repl's
414
+ // phase-turn-start discard — or the deviation step inside runPostPhaseChain
415
+ // would wrongly block the re-audited phase and clearAudit would destroy
416
+ // the very 'pending' audit crash-safety depends on.
417
+ getAndResetPlanDeviation();
418
+ const { phaseIds, ...chainArgs } = args;
419
+ if (phaseIds.length === 0)
420
+ return args.savedPath;
421
+ if (phaseIds.length > 1) {
422
+ args.log(chalk.dim(`[audit] re-audit pass over ${phaseIds.length} phases: ${phaseIds.join(', ')}`));
423
+ }
424
+ let path = args.savedPath;
425
+ let queued = false;
426
+ const queueDispatch = (cmd) => { queued = true; args.queueDispatch(cmd); };
427
+ for (let i = 0; i < phaseIds.length; i++) {
428
+ const result = await runPostPhaseChain({
429
+ ...chainArgs,
430
+ savedPath: path,
431
+ executedPhaseId: phaseIds[i],
432
+ forceAudit: true,
433
+ suppressOffer: i < phaseIds.length - 1,
434
+ queueDispatch,
435
+ });
436
+ if (result)
437
+ path = result;
438
+ if (queued) {
439
+ const remaining = phaseIds.slice(i + 1);
440
+ if (remaining.length > 0) {
441
+ args.log(chalk.yellow(`⚠ ${remaining.length} hand-edited phase(s) not re-audited yet: ${remaining.join(', ')} — `
442
+ + 'they are not auto-detected after this write; verify them (or re-run their phases) once the queued dispatch completes.'));
443
+ }
444
+ break;
445
+ }
446
+ }
447
+ return path;
448
+ }
449
+ /* ── helpers ───────────────────────────────────────────────────────────── */
450
+ function loadPlan(path, log) {
451
+ try {
452
+ const envelope = readProjectFile(path);
453
+ if (envelope.artefactType !== 'plan') {
454
+ log(chalk.red(`[plan] ${basename(path)} is not a plan envelope — chain skipped.`));
455
+ return null;
456
+ }
457
+ const pc = parsePlanContent(envelope.content);
458
+ if (!pc.ok) {
459
+ log(chalk.red(`[plan] ${basename(path)} failed content validation — chain skipped: ${pc.errors[0] ?? ''}`));
460
+ return null;
461
+ }
462
+ return { envelope, content: pc.content };
463
+ }
464
+ catch (e) {
465
+ log(chalk.red(`[plan] could not load ${basename(path)} — ${e instanceof Error ? e.message : String(e)}`));
466
+ return null;
467
+ }
468
+ }
469
+ /**
470
+ * Terminal bell with the STOPPED banner — TTY only (a piped/CI stream must
471
+ * never receive a raw BEL byte).
472
+ */
473
+ function ringBell() {
474
+ if (process.stdout.isTTY)
475
+ process.stdout.write('\u0007');
476
+ }
477
+ /** STOPPED banner in the plan-gate style, with the REAL resume command. */
478
+ function stoppedBanner(reason, resumeCommand) {
479
+ return `\n■ STOPPED — ${reason}\n Resume with: ${resumeCommand}\n`;
480
+ }
481
+ /**
482
+ * Task 3 left the gate banner's `@<plan>` placeholder literal (the pure
483
+ * module doesn't know the saved path) — substitute the real basename here.
484
+ */
485
+ export function substitutePlanToken(banner, savedPath) {
486
+ return banner.replace(/@<plan>/g, `@${basename(savedPath)}`);
487
+ }
488
+ function clampInfra(msg) {
489
+ const flat = msg.replace(/\s+/g, ' ').trim();
490
+ return flat.length > INFRA_RENDER_CHARS ? flat.slice(0, INFRA_RENDER_CHARS - 1).trimEnd() + '…' : flat;
491
+ }
492
+ function renderFindings(log, phaseId, findings) {
493
+ log('', chalk.red(`✖ audit FAILED for phase ${phaseId}:`));
494
+ const shown = findings.length > 0 ? findings : ['(auditor returned no findings text)'];
495
+ shown.forEach((f, i) => log(` ${i + 1}. ${f}`));
496
+ }
497
+ /**
498
+ * Confidence-tiered redesign (2026-07-08) — the concise, always-shown render
499
+ * for a NON-BLOCKING warn verdict. Mirrors renderFindings' shape but frames the
500
+ * phase as passed-with-notes (the chain continues). Each finding is already
501
+ * formatted "[W1] detail" by the auditor (plan-audit.renderFinding).
502
+ */
503
+ function renderWarnings(log, phaseId, findings) {
504
+ const n = findings.length;
505
+ log('', chalk.yellow(`⚠ audit passed with ${n} note${n === 1 ? '' : 's'} for phase ${phaseId}:`));
506
+ const shown = n > 0 ? findings : ['(auditor returned a warn verdict with no note detail)'];
507
+ shown.forEach((f, i) => log(` ${i + 1}. ${f}`));
508
+ }
509
+ /**
510
+ * audit-infra-continue (2026-07-09) — the LOUD, always-shown note for an
511
+ * audit that COULD NOT RUN. Distinct from warn's "passed with N notes": this
512
+ * frames the phase as unverified-but-proceeding, is explicit that the audit
513
+ * INFRASTRUCTURE failed (not the work), reassures that the phase's own exit
514
+ * gate passed, and points at the off-switch for a persistent outage — a
515
+ * pointer, not a forced prompt. Never swallowed (the chain continues, so
516
+ * nothing else surfaces this).
517
+ */
518
+ function renderAuditNotRun(log, phaseId, reason, resumeCommand) {
519
+ log('', chalk.yellow(`⚠ audit could not run for phase ${phaseId} (audit infrastructure failed after 3 attempts — the model call returned no output).`), chalk.yellow(' This phase is NOT audit-verified; its own exit gate (activation + ATC) passed. Proceeding.'), chalk.dim(` reason: ${reason}`), chalk.dim(` If the auditor keeps failing, run with --no-audit (or set audit off in config) to stop retrying; resume as usual: ${resumeCommand}`));
520
+ }
521
+ /**
522
+ * Shared menu surface for the audit-resolution selects. Mirrors
523
+ * safety-confirm.ts: headless → no decision (fail-closed, nothing written);
524
+ * Ink → askQuestionEmitter choice modal; classic → inquirer select.
525
+ */
526
+ async function selectAuditAction(args) {
527
+ const { isHeadless, shouldUseInk } = await import('../renderer/tty.js');
528
+ if (isHeadless())
529
+ return 'none';
530
+ if (shouldUseInk()) {
531
+ // D-C — this modal opens in the POST-PHASE chain, after the phase turn
532
+ // ended. Its sibling gate presentSafetyConfirmation (repl/safety-confirm.ts)
533
+ // documents the exact failure this caused live: opening the askQuestion
534
+ // modal while another render source (the in-place tool spinner, a ticking
535
+ // turn-status row) still repaints the Ink frame lets the first keystroke
536
+ // LEAK past the modal and auto-select its FIRST choice. Here the first
537
+ // choice is "Re-run phase", so the menu never visibly rendered and the
538
+ // phase silently re-ran — the reported symptom. Claim exclusive input
539
+ // focus the same way safety-confirm does: stop the spinner, freeze the
540
+ // status row, restore both in the finally.
541
+ const { clearActiveSpinner } = await import('../renderer/tool-widget.js');
542
+ const { turnStatusEmitter } = await import('../ui/turn-status-emitter.js');
543
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
544
+ clearActiveSpinner();
545
+ const statusWasActive = turnStatusEmitter.snapshot().active;
546
+ if (statusWasActive)
547
+ turnStatusEmitter.pause();
548
+ try {
549
+ const res = await askQuestionEmitter.request({
550
+ id: args.id,
551
+ question: args.question,
552
+ context: args.context,
553
+ kind: 'choice',
554
+ choices: args.choices,
555
+ });
556
+ if (res.cancelled || res.answer == null)
557
+ return 'none';
558
+ return coerceAction(res.answer, args.choices);
559
+ }
560
+ finally {
561
+ if (statusWasActive)
562
+ turnStatusEmitter.resume();
563
+ }
564
+ }
565
+ const { select } = await import('@inquirer/prompts');
566
+ const { withInquirer } = await import('../repl/inquirer-guard.js');
567
+ const { inquirerTheme } = await import('../repl/inquirer-theme.js');
568
+ const picked = await withInquirer(() => select({
569
+ message: args.question,
570
+ theme: inquirerTheme,
571
+ choices: args.choices.map((c) => ({ value: c.value, name: c.label })),
572
+ }));
573
+ return coerceAction(picked, args.choices);
574
+ }
575
+ /**
576
+ * D-C — fail CLOSED on any answer that is not one of this menu's own choice
577
+ * values. `'' ?? 'none'` did NOT catch an empty-string answer (nullish
578
+ * coalescing only guards null/undefined), so a stray/blank answer used to fall
579
+ * through the runPostPhaseChain switch to a silent default. Anything the menu
580
+ * did not explicitly offer maps to 'none' (stop, no auto-action).
581
+ */
582
+ function coerceAction(answer, choices) {
583
+ return choices.some((c) => c.value === answer) ? answer : 'none';
584
+ }
585
+ /** Production failed-verdict menu (see selectAuditAction for the surface). */
586
+ async function defaultSelectFailureAction(a) {
587
+ return selectAuditAction({
588
+ id: 'plan-audit-failed',
589
+ question: `Audit failed for phase ${a.phaseId} — what now?`,
590
+ context: a.findings.map((f, i) => `${i + 1}. ${f}`).join('\n'),
591
+ choices: [
592
+ { value: 'rerun', label: 'Re-run phase (re-audited after execution)' },
593
+ { value: 'waive', label: 'Waive audit (record a one-line reason)' },
594
+ { value: 'blocked', label: 'Mark phase blocked' },
595
+ ],
596
+ });
597
+ }
598
+ /**
599
+ * §10 Q2 (audit-confidence redesign) — production waive-category menu. A LIGHT
600
+ * 2-choice shown right after the waive-reason prompt so the audit's real
601
+ * false-positive rate is measurable. Same surface as selectAuditAction, but
602
+ * headless / cancel / stray answer → 'accepted-risk' (the CONSERVATIVE default:
603
+ * a waive no human explicitly labelled a false-positive is recorded as
604
+ * accepted-risk, so the measured FP rate never over-counts). Never hangs.
605
+ */
606
+ async function defaultSelectWaiveCategory(a) {
607
+ const { isHeadless, shouldUseInk } = await import('../renderer/tty.js');
608
+ if (isHeadless())
609
+ return 'accepted-risk';
610
+ const question = `Waiving ${a.phaseId} — why? (measures the audit's false-positive rate)`;
611
+ const choices = [
612
+ { value: 'accepted-risk', label: 'Accepted risk — the finding is real, I accept it and proceed' },
613
+ { value: 'false-positive', label: 'False positive — the audit was wrong, the work is actually fine' },
614
+ ];
615
+ const coerce = (answer) => choices.some((c) => c.value === answer) ? answer : 'accepted-risk';
616
+ if (shouldUseInk()) {
617
+ // Same input-focus discipline as selectAuditAction — this modal opens in
618
+ // the post-phase chain, so a leaked keystroke must not auto-select choice 1.
619
+ const { clearActiveSpinner } = await import('../renderer/tool-widget.js');
620
+ const { turnStatusEmitter } = await import('../ui/turn-status-emitter.js');
621
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
622
+ clearActiveSpinner();
623
+ const statusWasActive = turnStatusEmitter.snapshot().active;
624
+ if (statusWasActive)
625
+ turnStatusEmitter.pause();
626
+ try {
627
+ const res = await askQuestionEmitter.request({
628
+ id: 'plan-waive-category',
629
+ question,
630
+ kind: 'choice',
631
+ choices,
632
+ });
633
+ if (res.cancelled || res.answer == null)
634
+ return 'accepted-risk';
635
+ return coerce(res.answer);
636
+ }
637
+ finally {
638
+ if (statusWasActive)
639
+ turnStatusEmitter.resume();
640
+ }
641
+ }
642
+ const { select } = await import('@inquirer/prompts');
643
+ const { withInquirer } = await import('../repl/inquirer-guard.js');
644
+ const { inquirerTheme } = await import('../repl/inquirer-theme.js');
645
+ const picked = await withInquirer(() => select({
646
+ message: question,
647
+ theme: inquirerTheme,
648
+ choices: choices.map((c) => ({ value: c.value, name: c.label })),
649
+ }));
650
+ return coerce(picked);
651
+ }
652
+ /**
653
+ * Audit spend surfaces twice: a dim line next to the turn footer (the
654
+ * `audit` entry) and a labelled turn-0 line in the session cost JSONL —
655
+ * the audit runs in its own isolated session, so without this its tokens
656
+ * would be invisible to the user's cost trail. Best-effort, never throws.
657
+ */
658
+ function renderAuditCost(args, cost) {
659
+ try {
660
+ const tok = cost.tokens.input + cost.tokens.output;
661
+ const usd = computeCost(cost.model, cost.tokens);
662
+ args.log(chalk.dim(`[audit] ${tok.toLocaleString()} tok · ${formatCost(usd)} (${cost.model})`));
663
+ void appendCostLine(args.sessionId, {
664
+ ...buildEntry({ turn: 0, model: cost.model, tokens: cost.tokens, duration_ms: cost.durationMs }),
665
+ label: 'audit',
666
+ });
667
+ }
668
+ catch {
669
+ // observability only
670
+ }
671
+ }